oxedyne/fe2o3/fe2o3_mail/src/message.rs
32.8 KiB, 1 run
created by r1870400018:35629, which is this file's identity for as long as the history lasts, whatever it is later renamed to
download · who wrote it · its history
| 1 | //! Sans-io RFC 5322 message reading and draft building. |
| 2 | //! |
| 3 | //! This module owns no socket and reads no clock. A message is parsed from bytes a caller |
| 4 | //! already holds, and a draft is built into bytes the caller does something with -- so the |
| 5 | //! whole of it is a pure transform that is proven natively, in the tests below, rather than |
| 6 | //! only against a live server. The two facts a builder would otherwise reach for the |
| 7 | //! environment for -- the `Date` header and the `Message-ID` -- are inputs to |
| 8 | //! [`DraftMessage`], for the same reason: a builder that read the clock could not be tested |
| 9 | //! for the bytes it emits. |
| 10 | //! |
| 11 | //! A DRAFT IS NOT A SEND. [`DraftMessage::build`] produces the RFC 5322 document that a |
| 12 | //! message would be, and stops there. Nothing here opens a connection, and there is |
| 13 | //! deliberately no counterpart that puts the bytes on the wire: putting a finished document |
| 14 | //! in front of a person to send is the whole of the contribution, and the send stays theirs. |
| 15 | //! |
| 16 | //! # Character sets |
| 17 | //! |
| 18 | //! A body or a header word in a charset other than UTF-8 is decoded as UTF-8 with the |
| 19 | //! invalid sequences replaced, rather than transcoded, because a transcoder is a table per |
| 20 | //! charset and this crate carries no such table. The common case -- UTF-8, and the ASCII |
| 21 | //! subset every charset shares -- is exact; a legacy ISO-8859 body degrades to readable |
| 22 | //! text with the odd replacement character rather than failing. |
| 23 | |
| 24 | use oxedyne_fe2o3_core::prelude::*; |
| 25 | use oxedyne_fe2o3_text::base64; |
| 26 | |
| 27 | |
| 28 | /// A parsed message, reduced to what a reader needs: the headers that identify it and the |
| 29 | /// readable text of its body. |
| 30 | /// |
| 31 | /// The address and subject fields are decoded for display -- RFC 2047 encoded-words are |
| 32 | /// turned back into the characters they stand for -- so what a reader sees is the name and |
| 33 | /// the subject rather than `=?utf-8?B?...?=`. The raw header block is not kept; a caller |
| 34 | /// that wants a header this struct does not name can call [`headers`] on the bytes itself. |
| 35 | #[derive(Clone, Debug, Default)] |
| 36 | pub struct ParsedMessage { |
| 37 | pub from: String, // decoded display, e.g. "Ada <ada@x.example>" |
| 38 | pub to: String, |
| 39 | pub cc: String, |
| 40 | pub subject: String, // RFC 2047 decoded |
| 41 | pub date: String, // the Date header, verbatim |
| 42 | pub message_id: String, |
| 43 | pub in_reply_to: String, |
| 44 | pub references: String, |
| 45 | pub body: String, // readable text of the message |
| 46 | pub attachments: Vec<String>, // filenames of the non-text parts |
| 47 | } |
| 48 | |
| 49 | impl ParsedMessage { |
| 50 | /// Read a whole message from its RFC 5322 bytes. |
| 51 | /// |
| 52 | /// Infallible on purpose: a mailbox holds whatever a sender put in it, and a parser that |
| 53 | /// refused a malformed message would leave a reader unable to see the very message they |
| 54 | /// most need to. A field that cannot be found is empty, and a body that cannot be |
| 55 | /// decoded degrades rather than failing. |
| 56 | pub fn parse(raw: &[u8]) -> Self { |
| 57 | let text = String::from_utf8_lossy(raw); |
| 58 | let hs = headers(&text); |
| 59 | Self { |
| 60 | from: decode_words(&hget(&hs, "from")), |
| 61 | to: decode_words(&hget(&hs, "to")), |
| 62 | cc: decode_words(&hget(&hs, "cc")), |
| 63 | subject: decode_words(&hget(&hs, "subject")), |
| 64 | date: hget(&hs, "date"), |
| 65 | message_id: hget(&hs, "message-id"), |
| 66 | in_reply_to: hget(&hs, "in-reply-to"), |
| 67 | references: hget(&hs, "references"), |
| 68 | body: readable_text(&text), |
| 69 | attachments: attachment_names(&text), |
| 70 | } |
| 71 | } |
| 72 | } |
| 73 | |
| 74 | |
| 75 | // ── Reading ───────────────────────────────────────────────────────────── |
| 76 | |
| 77 | /// Split the header block from the body at the first blank line, unfold continuation lines, |
| 78 | /// and return the headers as an ordered list of `(lowercased-name, value)`. |
| 79 | /// |
| 80 | /// The name is lowercased because a header is looked up by a caller who did not write it and |
| 81 | /// cannot know whether the sender wrote `Message-ID` or `Message-Id`; the value is left as it |
| 82 | /// was, since its case may matter. |
| 83 | pub fn headers(text: &str) -> Vec<(String, String)> { |
| 84 | let block = match find_blank_line(text) { |
| 85 | Some(i) => &text[..i], |
| 86 | None => text, |
| 87 | }; |
| 88 | let mut out: Vec<(String, String)> = Vec::new(); |
| 89 | for line in block.split('\n') { |
| 90 | let line = line.strip_suffix('\r').unwrap_or(line); |
| 91 | // A line that begins with a space or tab continues the header above it. |
| 92 | if line.starts_with(' ') || line.starts_with('\t') { |
| 93 | if let Some(last) = out.last_mut() { |
| 94 | last.1.push(' '); |
| 95 | last.1.push_str(line.trim()); |
| 96 | continue; |
| 97 | } |
| 98 | } |
| 99 | if let Some(i) = line.find(':') { |
| 100 | let name = line[..i].trim().to_lowercase(); |
| 101 | let val = line[i + 1..].trim().to_string(); |
| 102 | if !name.is_empty() { |
| 103 | out.push((name, val)); |
| 104 | } |
| 105 | } |
| 106 | } |
| 107 | out |
| 108 | } |
| 109 | |
| 110 | /// The first value of a header, or the empty string when it is absent. |
| 111 | fn hget(hs: &[(String, String)], name: &str) -> String { |
| 112 | hs.iter() |
| 113 | .find(|(n, _)| n == name) |
| 114 | .map(|(_, v)| v.clone()) |
| 115 | .unwrap_or_default() |
| 116 | } |
| 117 | |
| 118 | /// The byte offset of the blank line that ends the header block, `\r\n\r\n` or `\n\n`. |
| 119 | fn find_blank_line(text: &str) -> Option<usize> { |
| 120 | let b = text.as_bytes(); |
| 121 | let mut i = 0; |
| 122 | while i + 1 < b.len() { |
| 123 | if b[i] == b'\n' && b[i + 1] == b'\n' { |
| 124 | return Some(i); |
| 125 | } |
| 126 | if i + 3 < b.len() && b[i] == b'\r' && b[i + 1] == b'\n' |
| 127 | && b[i + 2] == b'\r' && b[i + 3] == b'\n' |
| 128 | { |
| 129 | return Some(i); |
| 130 | } |
| 131 | i += 1; |
| 132 | } |
| 133 | None |
| 134 | } |
| 135 | |
| 136 | /// The body, which is everything after the blank line that ends the headers. |
| 137 | fn body_of(text: &str) -> &str { |
| 138 | match find_blank_line(text) { |
| 139 | Some(i) => { |
| 140 | // Step over the separator itself, whichever spelling it was. |
| 141 | let b = text.as_bytes(); |
| 142 | if i + 3 < b.len() && b[i] == b'\r' { |
| 143 | &text[i + 4..] |
| 144 | } else { |
| 145 | &text[i + 2..] |
| 146 | } |
| 147 | }, |
| 148 | None => "", |
| 149 | } |
| 150 | } |
| 151 | |
| 152 | /// Decode RFC 2047 encoded-words (`=?charset?B?...?=` or `?Q?`) in a header value. |
| 153 | /// |
| 154 | /// Adjacent encoded-words separated only by whitespace are joined with none, which is what |
| 155 | /// the standard says to do so a name split across two words reads as one. |
| 156 | pub fn decode_words(s: &str) -> String { |
| 157 | let bytes = s.as_bytes(); |
| 158 | let mut out = String::with_capacity(s.len()); |
| 159 | let mut i = 0; |
| 160 | // Whether the previous token emitted was itself an encoded-word, so the whitespace |
| 161 | // before the next one can be dropped where that next one is also encoded. |
| 162 | let mut last_was_word = false; |
| 163 | while i < bytes.len() { |
| 164 | if bytes[i] == b'=' && i + 1 < bytes.len() && bytes[i + 1] == b'?' { |
| 165 | if let Some((decoded, next)) = decode_one_word(s, i) { |
| 166 | // Drop the run of whitespace this word was separated from the previous |
| 167 | // encoded-word by. |
| 168 | if last_was_word { |
| 169 | while out.ends_with(' ') || out.ends_with('\t') |
| 170 | || out.ends_with('\r') || out.ends_with('\n') |
| 171 | { |
| 172 | out.pop(); |
| 173 | } |
| 174 | } |
| 175 | out.push_str(&decoded); |
| 176 | i = next; |
| 177 | last_was_word = true; |
| 178 | continue; |
| 179 | } |
| 180 | } |
| 181 | // Any ordinary character. A non-whitespace one means the run of encoded-words has |
| 182 | // ended, so a later word no longer joins to an earlier one. |
| 183 | let ch = s[i..].chars().next().unwrap_or('\u{fffd}'); |
| 184 | if !ch.is_whitespace() { |
| 185 | last_was_word = false; |
| 186 | } |
| 187 | out.push(ch); |
| 188 | i += ch.len_utf8(); |
| 189 | } |
| 190 | out |
| 191 | } |
| 192 | |
| 193 | /// Decode the one encoded-word beginning at `start`, returning it and the offset just past |
| 194 | /// it, or `None` when the bytes there are not a well-formed encoded-word. |
| 195 | fn decode_one_word(s: &str, start: usize) -> Option<(String, usize)> { |
| 196 | // `=?charset?enc?text?=` |
| 197 | let rest = &s[start + 2..]; |
| 198 | let q1 = rest.find('?')?; |
| 199 | let charset = &rest[..q1]; |
| 200 | let after_cs = &rest[q1 + 1..]; |
| 201 | let q2 = after_cs.find('?')?; |
| 202 | let enc = &after_cs[..q2]; |
| 203 | let after_enc = &after_cs[q2 + 1..]; |
| 204 | let q3 = after_enc.find("?=")?; |
| 205 | let text = &after_enc[..q3]; |
| 206 | if charset.is_empty() || enc.len() != 1 { |
| 207 | return None; |
| 208 | } |
| 209 | let end = start + 2 + q1 + 1 + q2 + 1 + q3 + 2; |
| 210 | let raw = match enc.as_bytes()[0].to_ascii_lowercase() { |
| 211 | b'b' => decode_base64_lenient(text), |
| 212 | b'q' => decode_q(text), |
| 213 | _ => return None, |
| 214 | }; |
| 215 | Some((String::from_utf8_lossy(&raw).into_owned(), end)) |
| 216 | } |
| 217 | |
| 218 | /// Decode a `Q` encoded-word body: `_` is a space, `=HH` is a byte, everything else stands. |
| 219 | fn decode_q(s: &str) -> Vec<u8> { |
| 220 | let b = s.as_bytes(); |
| 221 | let mut out = Vec::with_capacity(b.len()); |
| 222 | let mut i = 0; |
| 223 | while i < b.len() { |
| 224 | match b[i] { |
| 225 | b'_' => { out.push(b' '); i += 1; }, |
| 226 | b'=' if i + 2 < b.len() => { |
| 227 | match hex2(b[i + 1], b[i + 2]) { |
| 228 | Some(v) => { out.push(v); i += 3; }, |
| 229 | None => { out.push(b'='); i += 1; }, |
| 230 | } |
| 231 | }, |
| 232 | other => { out.push(other); i += 1; }, |
| 233 | } |
| 234 | } |
| 235 | out |
| 236 | } |
| 237 | |
| 238 | /// The readable text of a message: the `text/plain` part of a multipart, or the decoded |
| 239 | /// body, or `text/html` reduced to text where that is all there is. |
| 240 | /// |
| 241 | /// One level of multipart nesting is followed, which is the shape of a message with an |
| 242 | /// attachment -- a `multipart/mixed` whose first part is the `multipart/alternative` that |
| 243 | /// holds the text. |
| 244 | pub fn readable_text(raw: &str) -> String { |
| 245 | let hs = headers(raw); |
| 246 | let ctype = { |
| 247 | let c = hget(&hs, "content-type"); |
| 248 | if c.is_empty() { fmt!("text/plain") } else { c } |
| 249 | }; |
| 250 | let body = body_of(raw); |
| 251 | |
| 252 | if ctype.to_lowercase().contains("multipart") { |
| 253 | if let Some(boundary) = param(&ctype, "boundary") { |
| 254 | let marker = fmt!("--{}", boundary); |
| 255 | let mut plain: Option<String> = None; |
| 256 | let mut html: Option<String> = None; |
| 257 | for part in body.split(marker.as_str()) { |
| 258 | let part = part.strip_prefix('\r').unwrap_or(part); |
| 259 | let part = part.strip_prefix('\n').unwrap_or(part); |
| 260 | let phs = headers(part); |
| 261 | let pct = hget(&phs, "content-type"); |
| 262 | let pcte = hget(&phs, "content-transfer-encoding").to_lowercase(); |
| 263 | let pbody = body_of(part); |
| 264 | if pbody.trim().is_empty() { |
| 265 | continue; |
| 266 | } |
| 267 | let pctl = pct.to_lowercase(); |
| 268 | if pctl.contains("multipart") && plain.is_none() { |
| 269 | // A nested multipart: read one level down and take its text. |
| 270 | let inner = readable_text(&fmt!("Content-Type: {}\r\n\r\n{}", pct, pbody)); |
| 271 | if !inner.trim().is_empty() { |
| 272 | plain = Some(inner); |
| 273 | } |
| 274 | continue; |
| 275 | } |
| 276 | let decoded = decode_transfer(pbody, &pcte, param(&pct, "charset").as_deref()); |
| 277 | if pctl.contains("text/plain") && plain.is_none() { |
| 278 | plain = Some(decoded); |
| 279 | } else if pctl.contains("text/html") && html.is_none() { |
| 280 | html = Some(decoded); |
| 281 | } |
| 282 | } |
| 283 | if let Some(p) = plain { |
| 284 | return norm_lines(p.trim()); |
| 285 | } |
| 286 | if let Some(h) = html { |
| 287 | return norm_lines(strip_html(&h).trim()); |
| 288 | } |
| 289 | return String::new(); |
| 290 | } |
| 291 | } |
| 292 | |
| 293 | let cte = hget(&hs, "content-transfer-encoding").to_lowercase(); |
| 294 | let decoded = decode_transfer(body, &cte, param(&ctype, "charset").as_deref()); |
| 295 | if ctype.to_lowercase().contains("text/html") { |
| 296 | return norm_lines(strip_html(&decoded).trim()); |
| 297 | } |
| 298 | norm_lines(decoded.trim()) |
| 299 | } |
| 300 | |
| 301 | /// Line endings a reader wants: a message's `\r\n` and bare `\r` reduced to `\n`, so the |
| 302 | /// text a model is handed is the text and not the wire's carriage returns. |
| 303 | fn norm_lines(s: &str) -> String { |
| 304 | s.replace("\r\n", "\n").replace('\r', "\n") |
| 305 | } |
| 306 | |
| 307 | /// Decode one part's body out of its transfer encoding, then read the bytes as text. |
| 308 | fn decode_transfer(body: &str, cte: &str, charset: Option<&str>) -> String { |
| 309 | let bytes = match cte.trim() { |
| 310 | "base64" => decode_base64_lenient(body), |
| 311 | "quoted-printable" => decode_qp(body), |
| 312 | _ => body.as_bytes().to_vec(), |
| 313 | }; |
| 314 | // Every supported charset shares ASCII, and UTF-8 is the one this crate decodes exactly; |
| 315 | // anything else degrades (see the module header). `charset` is accepted so a future |
| 316 | // transcoder has the name to hand. |
| 317 | let _ = charset; |
| 318 | String::from_utf8_lossy(&bytes).into_owned() |
| 319 | } |
| 320 | |
| 321 | /// The filenames of the parts that are not the readable text -- what a reader means by "the |
| 322 | /// attachments". A part with a `filename` or a `name` parameter, or a `Content-Disposition` |
| 323 | /// of `attachment`, counts. |
| 324 | fn attachment_names(raw: &str) -> Vec<String> { |
| 325 | let hs = headers(raw); |
| 326 | let ctype = hget(&hs, "content-type"); |
| 327 | let mut out: Vec<String> = Vec::new(); |
| 328 | if let Some(boundary) = param(&ctype, "boundary") { |
| 329 | let marker = fmt!("--{}", boundary); |
| 330 | for part in body_of(raw).split(marker.as_str()) { |
| 331 | let phs = headers(part); |
| 332 | let disp = hget(&phs, "content-disposition"); |
| 333 | let pct = hget(&phs, "content-type"); |
| 334 | let name = param(&pct, "name") |
| 335 | .or_else(|| param(&disp, "filename")); |
| 336 | if let Some(n) = name { |
| 337 | if !n.trim().is_empty() { |
| 338 | out.push(decode_words(n.trim())); |
| 339 | } |
| 340 | } else if disp.to_lowercase().contains("attachment") { |
| 341 | out.push(fmt!("attachment")); |
| 342 | } |
| 343 | } |
| 344 | } |
| 345 | out |
| 346 | } |
| 347 | |
| 348 | /// The value of a `name="value"` parameter on a header, unquoted. |
| 349 | fn param(header: &str, name: &str) -> Option<String> { |
| 350 | let lower = header.to_lowercase(); |
| 351 | let key = fmt!("{}=", name.to_lowercase()); |
| 352 | let at = lower.find(&key)?; |
| 353 | let after = &header[at + key.len()..]; |
| 354 | let after = after.trim_start(); |
| 355 | if let Some(stripped) = after.strip_prefix('"') { |
| 356 | let end = stripped.find('"')?; |
| 357 | Some(stripped[..end].to_string()) |
| 358 | } else { |
| 359 | let end = after.find(|c: char| c == ';' || c == '\r' || c == '\n' || c == ' ') |
| 360 | .unwrap_or(after.len()); |
| 361 | let v = after[..end].trim(); |
| 362 | if v.is_empty() { None } else { Some(v.to_string()) } |
| 363 | } |
| 364 | } |
| 365 | |
| 366 | /// Decode a quoted-printable body to bytes: `=\n` soft breaks vanish, `=HH` is one byte. |
| 367 | pub fn decode_qp(s: &str) -> Vec<u8> { |
| 368 | let b = s.as_bytes(); |
| 369 | let mut out = Vec::with_capacity(b.len()); |
| 370 | let mut i = 0; |
| 371 | while i < b.len() { |
| 372 | if b[i] == b'=' { |
| 373 | // A soft line break: `=` at end of line. |
| 374 | if i + 1 < b.len() && b[i + 1] == b'\n' { i += 2; continue; } |
| 375 | if i + 2 < b.len() && b[i + 1] == b'\r' && b[i + 2] == b'\n' { i += 3; continue; } |
| 376 | if i + 2 < b.len() { |
| 377 | if let Some(v) = hex2(b[i + 1], b[i + 2]) { |
| 378 | out.push(v); |
| 379 | i += 3; |
| 380 | continue; |
| 381 | } |
| 382 | } |
| 383 | } |
| 384 | out.push(b[i]); |
| 385 | i += 1; |
| 386 | } |
| 387 | out |
| 388 | } |
| 389 | |
| 390 | /// Base64, tolerant of the line breaks and whitespace a MIME body wraps it in. |
| 391 | fn decode_base64_lenient(s: &str) -> Vec<u8> { |
| 392 | let cleaned: String = s.chars().filter(|c| !c.is_whitespace()).collect(); |
| 393 | if cleaned.is_empty() { |
| 394 | return Vec::new(); |
| 395 | } |
| 396 | match base64::decode(&cleaned) { |
| 397 | Ok(v) => v, |
| 398 | // A body that will not decode is handed back as its own bytes rather than lost: a |
| 399 | // reader seeing the encoded text is better served than one seeing nothing. |
| 400 | Err(_) => s.as_bytes().to_vec(), |
| 401 | } |
| 402 | } |
| 403 | |
| 404 | /// Two hex digits to a byte, or `None` when either is not hex. |
| 405 | fn hex2(a: u8, b: u8) -> Option<u8> { |
| 406 | let hi = (a as char).to_digit(16)?; |
| 407 | let lo = (b as char).to_digit(16)?; |
| 408 | Some((hi * 16 + lo) as u8) |
| 409 | } |
| 410 | |
| 411 | /// Reduce HTML to its readable text. A mail body is the least trustworthy string a reader |
| 412 | /// meets, so nothing here is ever treated as markup to render -- the tags are removed and |
| 413 | /// only the text between them is kept. |
| 414 | pub fn strip_html(html: &str) -> String { |
| 415 | let s0 = remove_blocks(html, "<style", "</style>"); |
| 416 | let s = remove_blocks(&s0, "<script", "</script>"); |
| 417 | |
| 418 | let mut out = String::with_capacity(s.len()); |
| 419 | let mut i = 0; |
| 420 | while i < s.len() { |
| 421 | if s.as_bytes()[i] == b'<' { |
| 422 | // A tag: skip to the matching '>'. A block-level close or a break becomes a |
| 423 | // newline so the text does not run together. |
| 424 | let tag_end = s[i..].find('>').map(|j| i + j + 1).unwrap_or(s.len()); |
| 425 | let tag = s[i..tag_end].to_lowercase(); |
| 426 | if is_break_tag(&tag) { |
| 427 | out.push('\n'); |
| 428 | } |
| 429 | i = tag_end; |
| 430 | } else { |
| 431 | let ch = s[i..].chars().next().unwrap_or('\u{fffd}'); |
| 432 | out.push(ch); |
| 433 | i += ch.len_utf8(); |
| 434 | } |
| 435 | } |
| 436 | collapse_blank_lines(&decode_entities(&out)) |
| 437 | } |
| 438 | |
| 439 | /// Does this tag end a block, so its removal should leave a line break behind? |
| 440 | fn is_break_tag(tag: &str) -> bool { |
| 441 | const BREAKERS: [&str; 12] = [ |
| 442 | "</p", "</div", "</tr", "</li", "</h1", "</h2", "</h3", "</h4", "</h5", "</h6", |
| 443 | "<br", "</table", |
| 444 | ]; |
| 445 | BREAKERS.iter().any(|b| tag.starts_with(b)) |
| 446 | } |
| 447 | |
| 448 | /// Remove every `<open ... close>` block, tag and content, case-insensitively. |
| 449 | fn remove_blocks(s: &str, open: &str, close: &str) -> String { |
| 450 | let lower = s.to_lowercase(); |
| 451 | let mut out = String::with_capacity(s.len()); |
| 452 | let mut i = 0; |
| 453 | while i < s.len() { |
| 454 | if lower[i..].starts_with(open) { |
| 455 | if let Some(rel) = lower[i..].find(close) { |
| 456 | i += rel + close.len(); |
| 457 | continue; |
| 458 | } |
| 459 | // An unterminated block: drop the rest. |
| 460 | break; |
| 461 | } |
| 462 | let ch = s[i..].chars().next().unwrap_or('\u{fffd}'); |
| 463 | out.push(ch); |
| 464 | i += ch.len_utf8(); |
| 465 | } |
| 466 | out |
| 467 | } |
| 468 | |
| 469 | /// Decode the handful of HTML entities that carry meaning in plain text. |
| 470 | fn decode_entities(s: &str) -> String { |
| 471 | s.replace("&", "&") |
| 472 | .replace("<", "<") |
| 473 | .replace(">", ">") |
| 474 | .replace(""", "\"") |
| 475 | .replace("'", "'") |
| 476 | .replace("'", "'") |
| 477 | .replace(" ", " ") |
| 478 | } |
| 479 | |
| 480 | /// Collapse three or more newlines to two, so the text keeps its paragraphs without the |
| 481 | /// gaps a stripped layout leaves. |
| 482 | fn collapse_blank_lines(s: &str) -> String { |
| 483 | let mut out = String::with_capacity(s.len()); |
| 484 | let mut runs = 0; |
| 485 | for ch in s.chars() { |
| 486 | if ch == '\n' { |
| 487 | runs += 1; |
| 488 | if runs <= 2 { |
| 489 | out.push('\n'); |
| 490 | } |
| 491 | } else { |
| 492 | runs = 0; |
| 493 | out.push(ch); |
| 494 | } |
| 495 | } |
| 496 | out |
| 497 | } |
| 498 | |
| 499 | |
| 500 | // ── Building a draft ───────────────────────────────────────────────────── |
| 501 | |
| 502 | /// A message being composed, before it is turned into the bytes a person would send. |
| 503 | /// |
| 504 | /// The `date` and `message_id` are fields rather than something the builder works out, |
| 505 | /// because a builder that read the clock or drew a random id could not be tested for the |
| 506 | /// bytes it emits (see the module header). A caller in a browser fills them from the page's |
| 507 | /// own clock and randomness; a test fills them with fixed strings. |
| 508 | #[derive(Clone, Debug, Default)] |
| 509 | pub struct DraftMessage { |
| 510 | pub from: String, // the bare address the message is from |
| 511 | pub from_name: String, // the display name, or empty |
| 512 | pub to: Vec<String>, // each "addr" or "Name <addr>" |
| 513 | pub cc: Vec<String>, |
| 514 | pub subject: String, |
| 515 | pub body: String, |
| 516 | pub in_reply_to: String, // the Message-ID this replies to, or empty |
| 517 | pub references: String, // the References header, or empty to derive from in_reply_to |
| 518 | pub date: String, // the Date header, spelled as RFC 5322 wants it |
| 519 | pub message_id: String, // this message's own Message-ID, angle brackets and all |
| 520 | } |
| 521 | |
| 522 | impl DraftMessage { |
| 523 | /// Build the RFC 5322 document this draft describes: a single `text/plain` part, its body |
| 524 | /// quoted-printable, with the headers that make a reply thread. |
| 525 | /// |
| 526 | /// This does NOT send. It hands back bytes; what becomes of them is the caller's, and in |
| 527 | /// Daimond the caller writes them to a drafts folder for a person to read and send. |
| 528 | pub fn build(&self) -> Outcome<Vec<u8>> { |
| 529 | if self.from.trim().is_empty() { |
| 530 | return Err(err!( |
| 531 | "A draft has no sender address, so it cannot be built."; Invalid, Input, Missing)); |
| 532 | } |
| 533 | if self.to.iter().all(|a| a.trim().is_empty()) { |
| 534 | return Err(err!( |
| 535 | "A draft has no recipient in its To list, so it cannot be built."; |
| 536 | Invalid, Input, Missing)); |
| 537 | } |
| 538 | if self.date.trim().is_empty() { |
| 539 | return Err(err!( |
| 540 | "A draft was built with no Date header; the caller must supply one."; |
| 541 | Invalid, Input, Missing)); |
| 542 | } |
| 543 | if self.message_id.trim().is_empty() { |
| 544 | return Err(err!( |
| 545 | "A draft was built with no Message-ID; the caller must supply one."; |
| 546 | Invalid, Input, Missing)); |
| 547 | } |
| 548 | |
| 549 | let mut h: Vec<String> = Vec::new(); |
| 550 | h.push(fmt!("Message-ID: {}", self.message_id.trim())); |
| 551 | h.push(fmt!("Date: {}", self.date.trim())); |
| 552 | h.push(fmt!("From: {}", encode_addr(&self.from_name, &self.from))); |
| 553 | h.push(fmt!("To: {}", join_addrs(&self.to))); |
| 554 | if !self.cc.iter().all(|a| a.trim().is_empty()) { |
| 555 | h.push(fmt!("Cc: {}", join_addrs(&self.cc))); |
| 556 | } |
| 557 | h.push(fmt!("Subject: {}", encode_word(&self.subject))); |
| 558 | if !self.in_reply_to.trim().is_empty() { |
| 559 | h.push(fmt!("In-Reply-To: {}", self.in_reply_to.trim())); |
| 560 | let refs = if self.references.trim().is_empty() { |
| 561 | self.in_reply_to.trim() |
| 562 | } else { |
| 563 | self.references.trim() |
| 564 | }; |
| 565 | h.push(fmt!("References: {}", refs)); |
| 566 | } |
| 567 | h.push(fmt!("MIME-Version: 1.0")); |
| 568 | h.push(fmt!("User-Agent: Daimond")); |
| 569 | h.push(fmt!("Content-Type: text/plain; charset=utf-8")); |
| 570 | h.push(fmt!("Content-Transfer-Encoding: quoted-printable")); |
| 571 | |
| 572 | let doc = fmt!("{}\r\n\r\n{}\r\n", h.join("\r\n"), encode_qp(&self.body)); |
| 573 | Ok(doc.into_bytes()) |
| 574 | } |
| 575 | } |
| 576 | |
| 577 | /// Join a list of addresses for a `To` or `Cc` header, each encoded and comma-separated. |
| 578 | fn join_addrs(list: &[String]) -> String { |
| 579 | list.iter() |
| 580 | .map(|a| a.trim()) |
| 581 | .filter(|a| !a.is_empty()) |
| 582 | .map(|a| { |
| 583 | let (name, addr) = split_addr(a); |
| 584 | encode_addr(&name, &addr) |
| 585 | }) |
| 586 | .collect::<Vec<String>>() |
| 587 | .join(", ") |
| 588 | } |
| 589 | |
| 590 | /// Split `Name <addr>` into its two parts; a bare address gives an empty name. |
| 591 | fn split_addr(s: &str) -> (String, String) { |
| 592 | if let Some(open) = s.rfind('<') { |
| 593 | if let Some(close) = s[open..].find('>') { |
| 594 | let addr = s[open + 1..open + close].trim().to_string(); |
| 595 | let name = s[..open].trim().trim_matches('"').trim().to_string(); |
| 596 | return (name, addr); |
| 597 | } |
| 598 | } |
| 599 | (String::new(), s.trim().to_string()) |
| 600 | } |
| 601 | |
| 602 | /// One address as a header writes it: `Name <addr>`, the name encoded if it is not ASCII and |
| 603 | /// quoted if it holds a character that would otherwise punctuate the header. |
| 604 | fn encode_addr(name: &str, addr: &str) -> String { |
| 605 | let addr = addr.trim(); |
| 606 | let name = name.trim(); |
| 607 | if addr.is_empty() { |
| 608 | return String::new(); |
| 609 | } |
| 610 | if name.is_empty() { |
| 611 | return addr.to_string(); |
| 612 | } |
| 613 | let shown = if is_ascii(name) { |
| 614 | if name.contains(|c| "(),:;<>@[]\".".contains(c)) { |
| 615 | fmt!("\"{}\"", name.replace('\\', "\\\\").replace('"', "\\\"")) |
| 616 | } else { |
| 617 | name.to_string() |
| 618 | } |
| 619 | } else { |
| 620 | encode_word(name) |
| 621 | }; |
| 622 | fmt!("{} <{}>", shown, addr) |
| 623 | } |
| 624 | |
| 625 | /// Whether every character is printable ASCII. |
| 626 | fn is_ascii(s: &str) -> bool { |
| 627 | s.bytes().all(|b| (0x20..=0x7e).contains(&b)) |
| 628 | } |
| 629 | |
| 630 | /// A header value with anything but plain ASCII in it, as one or more RFC 2047 base64 |
| 631 | /// encoded-words. The words are cut at a character boundary, never inside a multi-byte one, |
| 632 | /// so no recipient decodes a half character. |
| 633 | pub fn encode_word(s: &str) -> String { |
| 634 | if is_ascii(s) { |
| 635 | return s.to_string(); |
| 636 | } |
| 637 | let mut words: Vec<String> = Vec::new(); |
| 638 | let mut chunk = String::new(); |
| 639 | let mut bytes = 0usize; |
| 640 | for ch in s.chars() { |
| 641 | let n = ch.len_utf8(); |
| 642 | // 39 source bytes keeps one encoded-word's base64 under the 76-character line limit. |
| 643 | if bytes + n > 39 && !chunk.is_empty() { |
| 644 | words.push(fmt!("=?utf-8?B?{}?=", base64::encode(chunk.as_bytes()))); |
| 645 | chunk.clear(); |
| 646 | bytes = 0; |
| 647 | } |
| 648 | chunk.push(ch); |
| 649 | bytes += n; |
| 650 | } |
| 651 | if !chunk.is_empty() { |
| 652 | words.push(fmt!("=?utf-8?B?{}?=", base64::encode(chunk.as_bytes()))); |
| 653 | } |
| 654 | words.join("\r\n ") |
| 655 | } |
| 656 | |
| 657 | /// Quoted-printable over a body's UTF-8 bytes. |
| 658 | /// |
| 659 | /// The three rules that bite: a space or tab at the end of a line is invisible and would be |
| 660 | /// stripped in transit, so it is encoded; a line is folded with a soft break before it |
| 661 | /// reaches the 76-character limit; and a line that would begin `From ` is escaped, because |
| 662 | /// some software still reads one as the start of a new message. |
| 663 | pub fn encode_qp(text: &str) -> String { |
| 664 | let norm = text.replace("\r\n", "\n").replace('\r', "\n"); |
| 665 | let bytes = norm.as_bytes(); |
| 666 | let mut lines: Vec<String> = Vec::new(); |
| 667 | let mut line = String::new(); |
| 668 | let mut held: Option<u8> = None; // a pending space or tab, held in case a newline follows |
| 669 | |
| 670 | let mut i = 0; |
| 671 | while i < bytes.len() { |
| 672 | let b = bytes[i]; |
| 673 | if b == 0x0a { |
| 674 | if let Some(h) = held.take() { |
| 675 | qp_push(&mut lines, &mut line, if h == 0x20 { "=20" } else { "=09" }); |
| 676 | } |
| 677 | lines.push(std::mem::take(&mut line)); |
| 678 | i += 1; |
| 679 | continue; |
| 680 | } |
| 681 | if let Some(h) = held.take() { |
| 682 | qp_push(&mut lines, &mut line, &(h as char).to_string()); |
| 683 | } |
| 684 | if b == 0x20 { held = Some(0x20); i += 1; continue; } |
| 685 | if b == 0x09 { held = Some(0x09); i += 1; continue; } |
| 686 | if (33..=126).contains(&b) && b != 61 { |
| 687 | qp_push(&mut lines, &mut line, &(b as char).to_string()); |
| 688 | } else { |
| 689 | qp_push(&mut lines, &mut line, &fmt!("={:02X}", b)); |
| 690 | } |
| 691 | // A line may not begin "From ". |
| 692 | if line == "From" && i + 1 < bytes.len() && bytes[i + 1] == 0x20 { |
| 693 | line = fmt!("=46rom"); |
| 694 | } |
| 695 | i += 1; |
| 696 | } |
| 697 | if let Some(h) = held.take() { |
| 698 | qp_push(&mut lines, &mut line, if h == 0x20 { "=20" } else { "=09" }); |
| 699 | } |
| 700 | lines.push(line); |
| 701 | lines.join("\r\n") |
| 702 | } |
| 703 | |
| 704 | /// Append one token to the current quoted-printable line, folding with a soft break first |
| 705 | /// when it would otherwise pass the 76-character limit. |
| 706 | fn qp_push(lines: &mut Vec<String>, line: &mut String, tok: &str) { |
| 707 | if line.len() + tok.len() > 75 { |
| 708 | line.push('='); |
| 709 | lines.push(std::mem::take(line)); |
| 710 | } |
| 711 | line.push_str(tok); |
| 712 | } |
| 713 | |
| 714 | |
| 715 | #[cfg(test)] |
| 716 | mod tests { |
| 717 | use super::*; |
| 718 | |
| 719 | #[test] |
| 720 | fn parses_a_simple_message() { |
| 721 | let raw = "From: Ada <ada@x.example>\r\n\ |
| 722 | To: Bob <bob@y.example>\r\n\ |
| 723 | Subject: Hello there\r\n\ |
| 724 | Date: Mon, 1 Sep 2026 10:00:00 +0000\r\n\ |
| 725 | Message-ID: <abc@x.example>\r\n\ |
| 726 | \r\n\ |
| 727 | This is the body.\r\n"; |
| 728 | let m = ParsedMessage::parse(raw.as_bytes()); |
| 729 | assert_eq!(m.from, "Ada <ada@x.example>"); |
| 730 | assert_eq!(m.subject, "Hello there"); |
| 731 | assert_eq!(m.message_id, "<abc@x.example>"); |
| 732 | assert_eq!(m.body, "This is the body."); |
| 733 | } |
| 734 | |
| 735 | #[test] |
| 736 | fn unfolds_a_continued_header() { |
| 737 | let raw = "Subject: one two\r\n three four\r\nFrom: a@b\r\n\r\nbody\r\n"; |
| 738 | let hs = headers(raw); |
| 739 | assert_eq!(hget(&hs, "subject"), "one two three four"); |
| 740 | } |
| 741 | |
| 742 | #[test] |
| 743 | fn decodes_a_2047_subject() { |
| 744 | // "Grüße" base64 in UTF-8. |
| 745 | let enc = base64::encode("Grüße".as_bytes()); |
| 746 | let s = fmt!("=?utf-8?B?{}?=", enc); |
| 747 | assert_eq!(decode_words(&s), "Grüße"); |
| 748 | } |
| 749 | |
| 750 | #[test] |
| 751 | fn decodes_a_q_word() { |
| 752 | // "_" is a space in Q, so "=5F" is a literal underscore. |
| 753 | let s = "=?utf-8?Q?a=5Fb?="; |
| 754 | assert_eq!(decode_words(s), "a_b"); |
| 755 | } |
| 756 | |
| 757 | #[test] |
| 758 | fn joins_two_adjacent_words() { |
| 759 | let a = base64::encode("Ma".as_bytes()); |
| 760 | let b = base64::encode("ría".as_bytes()); |
| 761 | let s = fmt!("=?utf-8?B?{}?= =?utf-8?B?{}?=", a, b); |
| 762 | assert_eq!(decode_words(&s), "María"); |
| 763 | } |
| 764 | |
| 765 | #[test] |
| 766 | fn reads_the_plain_part_of_a_multipart() { |
| 767 | let raw = "Content-Type: multipart/alternative; boundary=\"BB\"\r\n\r\n\ |
| 768 | --BB\r\n\ |
| 769 | Content-Type: text/plain; charset=utf-8\r\n\r\n\ |
| 770 | the plain text\r\n\ |
| 771 | --BB\r\n\ |
| 772 | Content-Type: text/html; charset=utf-8\r\n\r\n\ |
| 773 | <p>the html</p>\r\n\ |
| 774 | --BB--\r\n"; |
| 775 | assert_eq!(readable_text(raw), "the plain text"); |
| 776 | } |
| 777 | |
| 778 | #[test] |
| 779 | fn falls_back_to_html_when_no_plain_part() { |
| 780 | let raw = "Content-Type: multipart/alternative; boundary=\"BB\"\r\n\r\n\ |
| 781 | --BB\r\n\ |
| 782 | Content-Type: text/html; charset=utf-8\r\n\r\n\ |
| 783 | <p>first</p><p>second</p>\r\n\ |
| 784 | --BB--\r\n"; |
| 785 | assert_eq!(readable_text(raw), "first\nsecond"); |
| 786 | } |
| 787 | |
| 788 | #[test] |
| 789 | fn decodes_quoted_printable_body() { |
| 790 | let raw = "Content-Type: text/plain; charset=utf-8\r\n\ |
| 791 | Content-Transfer-Encoding: quoted-printable\r\n\r\n\ |
| 792 | caf=C3=A9\r\n"; |
| 793 | assert_eq!(readable_text(raw), "café"); |
| 794 | } |
| 795 | |
| 796 | #[test] |
| 797 | fn names_an_attachment() { |
| 798 | let raw = "Content-Type: multipart/mixed; boundary=\"BB\"\r\n\r\n\ |
| 799 | --BB\r\n\ |
| 800 | Content-Type: text/plain\r\n\r\n\ |
| 801 | see attached\r\n\ |
| 802 | --BB\r\n\ |
| 803 | Content-Type: application/pdf; name=\"report.pdf\"\r\n\ |
| 804 | Content-Disposition: attachment; filename=\"report.pdf\"\r\n\r\n\ |
| 805 | JVBERi0=\r\n\ |
| 806 | --BB--\r\n"; |
| 807 | let m = ParsedMessage::parse(raw.as_bytes()); |
| 808 | assert_eq!(m.attachments, vec!["report.pdf".to_string()]); |
| 809 | assert_eq!(m.body, "see attached"); |
| 810 | } |
| 811 | |
| 812 | #[test] |
| 813 | fn builds_a_draft_and_reads_it_back() { |
| 814 | let d = DraftMessage { |
| 815 | from: fmt!("me@x.example"), |
| 816 | from_name: fmt!("Me Myself"), |
| 817 | to: vec![fmt!("You <you@y.example>")], |
| 818 | cc: Vec::new(), |
| 819 | subject: fmt!("Re: café"), |
| 820 | body: fmt!("Hello,\nThis is a draft.\n"), |
| 821 | in_reply_to: fmt!("<orig@y.example>"), |
| 822 | references: String::new(), |
| 823 | date: fmt!("Mon, 1 Sep 2026 12:00:00 +0000"), |
| 824 | message_id: fmt!("<new@x.example>"), |
| 825 | }; |
| 826 | let bytes = d.build().expect("the draft builds"); |
| 827 | let m = ParsedMessage::parse(&bytes); |
| 828 | assert_eq!(m.from, "Me Myself <me@x.example>"); |
| 829 | assert_eq!(m.to, "You <you@y.example>"); |
| 830 | assert_eq!(m.subject, "Re: café"); |
| 831 | assert_eq!(m.in_reply_to, "<orig@y.example>"); |
| 832 | assert_eq!(m.references, "<orig@y.example>"); |
| 833 | assert_eq!(m.body, "Hello,\nThis is a draft."); |
| 834 | } |
| 835 | |
| 836 | #[test] |
| 837 | fn a_draft_with_no_recipient_is_refused() { |
| 838 | let d = DraftMessage { |
| 839 | from: fmt!("me@x.example"), |
| 840 | to: Vec::new(), |
| 841 | date: fmt!("Mon, 1 Sep 2026 12:00:00 +0000"), |
| 842 | message_id: fmt!("<new@x.example>"), |
| 843 | ..Default::default() |
| 844 | }; |
| 845 | assert!(d.build().is_err()); |
| 846 | } |
| 847 | |
| 848 | #[test] |
| 849 | fn quoted_printable_encodes_trailing_space_and_from() { |
| 850 | let out = encode_qp("From here \nnext"); |
| 851 | // The trailing space before the newline is encoded, and a line beginning "From " is |
| 852 | // escaped at its F. |
| 853 | assert!(out.starts_with("=46rom here=20"), "got {:?}", out); |
| 854 | } |
| 855 | |
| 856 | #[test] |
| 857 | fn encode_word_round_trips_through_decode() { |
| 858 | let s = "Grüße aus München — a longer non-ASCII subject line that must be chunked"; |
| 859 | let encoded = encode_word(s); |
| 860 | // Joining the folded words back and decoding yields the original. |
| 861 | assert_eq!(decode_words(&encoded.replace("\r\n ", "")), s); |
| 862 | } |
| 863 | } |