SMTP send: fix recipient parsing and non-UTF-8 body handling (#10)

* mail: do not split a quoted display name on its comma

parse_address_list tracked angle-bracket depth but not quotes, so a
recipient like `"Doe, John" <john@x.com>` was split on the comma inside
the quoted name, yielding a bogus recipient (`"Doe`) next to the real
one. With a real contact named "Last, First" that either gets the whole
send rejected by Tuta or delivers to a garbage address.

Track the quote state too: inside `"..."`, commas and angle brackets are
literal. Tested with a quoted-comma recipient and a plain comma list.

* mail: decode non-UTF-8 bodies instead of echoing base64 or QP source

When a base64 or quoted-printable body decoded to bytes that were not
valid UTF-8 (e.g. a Latin-1 message), the parser fell back to returning
the still-encoded source: the recipient saw a wall of base64, or raw
=XX sequences. Decode the bytes lossily instead, so the text is readable
(non-UTF-8 bytes become the replacement char rather than garbage).

Full charset-aware decoding (Content-Type charset via encoding_rs) is a
follow-up; this fixes the worst symptom with no new dependency. Tested
with a non-UTF-8 base64 body and a non-UTF-8 quoted-printable byte.
This commit is contained in:
Anthony M
2026-06-14 21:12:59 +02:00
committed by GitHub
parent 1863144627
commit 7da9339146
+60 -7
View File
@@ -134,19 +134,28 @@ fn parse_address_single(raw: &str) -> (String, String) {
fn parse_address_list(raw: &str) -> Vec<(String, String)> { fn parse_address_list(raw: &str) -> Vec<(String, String)> {
let mut result = Vec::new(); let mut result = Vec::new();
let mut depth = 0i32; let mut depth = 0i32;
let mut in_quotes = false;
let mut current = String::new(); let mut current = String::new();
for ch in raw.chars() { for ch in raw.chars() {
match ch { match ch {
'<' => { // A quoted display name may contain commas and angle brackets that
// are NOT list separators, e.g. `"Doe, John" <j@x.com>`. Track the
// quote state so those stay part of the same entry instead of
// splitting it into bogus recipients.
'"' => {
in_quotes = !in_quotes;
current.push(ch);
}
'<' if !in_quotes => {
depth += 1; depth += 1;
current.push(ch); current.push(ch);
} }
'>' => { '>' if !in_quotes => {
depth -= 1; depth -= 1;
current.push(ch); current.push(ch);
} }
',' if depth == 0 => { ',' if depth == 0 && !in_quotes => {
let trimmed = current.trim().to_string(); let trimmed = current.trim().to_string();
if !trimmed.is_empty() { if !trimmed.is_empty() {
result.push(parse_address_single(&trimmed)); result.push(parse_address_single(&trimmed));
@@ -397,9 +406,11 @@ fn decode_body(body: &str, transfer_encoding: &str, content_type: &str) -> Strin
let clean: String = body.chars().filter(|c| !c.is_whitespace()).collect(); let clean: String = body.chars().filter(|c| !c.is_whitespace()).collect();
base64::engine::general_purpose::STANDARD base64::engine::general_purpose::STANDARD
.decode(&clean) .decode(&clean)
.ok() // Decode the bytes lossily rather than echoing the raw base64 when
.and_then(|bytes| String::from_utf8(bytes).ok()) // the payload is not valid UTF-8 (e.g. a Latin-1 body). Returning
.unwrap_or_else(|| body.to_string()) // the base64 blob as "the body" was the worst possible fallback.
.map(|bytes| String::from_utf8_lossy(&bytes).into_owned())
.unwrap_or_else(|_| body.to_string())
} else if transfer_encoding.contains("quoted-printable") { } else if transfer_encoding.contains("quoted-printable") {
decode_quoted_printable(body) decode_quoted_printable(body)
} else { } else {
@@ -438,7 +449,9 @@ fn decode_quoted_printable(s: &str) -> String {
i += 1; i += 1;
} }
} }
String::from_utf8(result).unwrap_or_else(|_| s.to_string()) // Use the decoded bytes (lossily) rather than echoing the raw `=XX`
// source when the result is not valid UTF-8.
String::from_utf8_lossy(&result).into_owned()
} }
fn html_escape(s: &str) -> String { fn html_escape(s: &str) -> String {
@@ -463,6 +476,46 @@ mod tests {
assert_eq!(msg.body_html, "<p>Hi Bob</p>"); assert_eq!(msg.body_html, "<p>Hi Bob</p>");
} }
#[test]
fn quoted_comma_in_display_name_is_one_recipient() {
let raw = "From: me@tuta.io\r\nTo: \"Doe, John\" <john@x.com>, bob@y.com\r\nSubject: t\r\n\r\nbody\r\n";
let msg = parse_rfc2822(raw);
let addrs: Vec<&str> = msg.to.iter().map(|(_, a)| a.as_str()).collect();
assert_eq!(
addrs,
vec!["john@x.com", "bob@y.com"],
"a comma inside a quoted display name must not split the recipient"
);
assert_eq!(msg.to[0].0, "Doe, John", "display name should be preserved");
}
#[test]
fn plain_address_list_still_splits_on_commas() {
let raw = "From: me@tuta.io\r\nTo: a@x.com, b@y.com, c@z.com\r\nSubject: t\r\n\r\nbody\r\n";
let msg = parse_rfc2822(raw);
let addrs: Vec<&str> = msg.to.iter().map(|(_, a)| a.as_str()).collect();
assert_eq!(addrs, vec!["a@x.com", "b@y.com", "c@z.com"]);
}
#[test]
fn base64_body_non_utf8_is_not_echoed_as_base64() {
// "caf" + 0xE9 (Latin-1 'é'): valid base64, not valid UTF-8.
let b64 = base64::engine::general_purpose::STANDARD.encode(b"caf\xe9");
let out = decode_body(&b64, "base64", "text/html");
assert!(!out.contains(&b64), "must not emit the raw base64 blob");
assert!(
out.starts_with("caf"),
"decoded text should be readable, got {out:?}"
);
}
#[test]
fn qp_non_utf8_byte_is_decoded_not_echoed() {
let out = decode_quoted_printable("caf=E9");
assert!(!out.contains("=E9"), "QP source must be decoded, not echoed");
assert!(out.starts_with("caf"), "got {out:?}");
}
#[test] #[test]
fn test_parse_multiple_recipients() { fn test_parse_multiple_recipients() {
let raw = "From: a@b.com\r\nTo: Bob <bob@x.com>, Charlie <charlie@x.com>\r\nCc: Dave <dave@x.com>\r\nSubject: Test\r\n\r\nbody"; let raw = "From: a@b.com\r\nTo: Bob <bob@x.com>, Charlie <charlie@x.com>\r\nCc: Dave <dave@x.com>\r\nSubject: Test\r\n\r\nbody";