Encode folder names as modified UTF-7 over IMAP

Add an imap::utf7 module (RFC 3501 §5.1.3) and apply it at the protocol
boundary: encode mailbox names in LIST, decode them in SELECT/STATUS.
Folder names internally stay UTF-8; only the IMAP wire form is UTF-7.

Also fix STATUS argument parsing to handle quoted mailbox names with
spaces (it split on the first space, truncating names like
"Not Important" or nested paths).

Live-tested: "Café" lists as "Caf&AOk-", a nested child lists as
"Caf&AOk-/Test dossier avec espace", and SELECT/STATUS resolve both.
This commit is contained in:
Anthony
2026-05-27 16:05:11 +02:00
committed by Anthony M
parent f37ff8bca6
commit 8f2bc4197d
3 changed files with 149 additions and 6 deletions
+1
View File
@@ -1,4 +1,5 @@
mod session;
mod utf7;
use std::sync::Arc;
use log::{info, error, debug};
+9 -6
View File
@@ -220,7 +220,8 @@ impl ImapSession {
let flags = folder.special_use.as_deref().unwrap_or("");
responses.push(format!(
"* LIST ({}) \"/\" \"{}\"\r\n",
flags, folder.imap_path
flags,
super::utf7::encode(&folder.imap_path)
));
}
responses.push(format!("{} OK LIST completed\r\n", tag));
@@ -233,8 +234,9 @@ impl ImapSession {
return vec![format!("{} NO Not authenticated\r\n", tag)];
}
let folder_name = args.trim().trim_matches('"');
let folder = match self.store.folder_by_imap_path(folder_name).await {
let raw_name = args.trim().trim_matches('"');
let folder_name = super::utf7::decode(raw_name).unwrap_or_else(|| raw_name.to_string());
let folder = match self.store.folder_by_imap_path(&folder_name).await {
Some(f) => f,
None => {
return vec![format!("{} NO [NONEXISTENT] Mailbox does not exist\r\n", tag)];
@@ -283,8 +285,9 @@ impl ImapSession {
if self.state == State::NotAuthenticated {
return vec![format!("{} NO Not authenticated\r\n", tag)];
}
let folder_name = args.split_whitespace().next().unwrap_or("").trim_matches('"');
let stored = match self.store.folder_by_imap_path(folder_name).await {
let (raw_name, _) = parse_imap_token(args.trim());
let folder_name = super::utf7::decode(&raw_name).unwrap_or_else(|| raw_name.clone());
let stored = match self.store.folder_by_imap_path(&folder_name).await {
Some(folder) => self.store.get_folder(&folder.id).await,
None => Vec::new(),
};
@@ -293,7 +296,7 @@ impl ImapSession {
vec![
format!(
"* STATUS \"{}\" (MESSAGES {} UNSEEN {} RECENT 0 UIDNEXT {} UIDVALIDITY 1)\r\n",
folder_name, count, unseen, self.uid_next
raw_name, count, unseen, self.uid_next
),
format!("{} OK STATUS completed\r\n", tag),
]
+139
View File
@@ -0,0 +1,139 @@
//! Modified UTF-7 for IMAP mailbox names (RFC 3501 §5.1.3).
//!
//! Printable ASCII (0x200x7E) represents itself, except `&` which is encoded
//! as `&-`. Any other run of characters is encoded as `&<b64>-` where `<b64>`
//! is the modified BASE64 (alphabet `+,` instead of `+/`, no padding) of the
//! run encoded as UTF-16BE.
use base64::alphabet::Alphabet;
use base64::engine::general_purpose::{GeneralPurpose, GeneralPurposeConfig};
use base64::engine::DecodePaddingMode;
use base64::Engine;
use std::sync::LazyLock;
/// Modified BASE64 alphabet: standard, but `/` becomes `,`.
static MODIFIED_B64: LazyLock<GeneralPurpose> = LazyLock::new(|| {
let alphabet =
Alphabet::new("ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+,")
.expect("valid alphabet");
let config = GeneralPurposeConfig::new()
.with_encode_padding(false)
.with_decode_padding_mode(DecodePaddingMode::RequireNone);
GeneralPurpose::new(&alphabet, config)
});
fn is_direct(c: char) -> bool {
// Printable ASCII except `&`.
('\u{20}'..='\u{7e}').contains(&c) && c != '&'
}
/// Encode a UTF-8 mailbox name into modified UTF-7.
pub fn encode(input: &str) -> String {
let mut out = String::with_capacity(input.len());
let mut shifted: Vec<u16> = Vec::new();
let flush = |shifted: &mut Vec<u16>, out: &mut String| {
if shifted.is_empty() {
return;
}
let mut bytes = Vec::with_capacity(shifted.len() * 2);
for unit in shifted.drain(..) {
bytes.extend_from_slice(&unit.to_be_bytes());
}
out.push('&');
out.push_str(&MODIFIED_B64.encode(&bytes));
out.push('-');
};
for c in input.chars() {
if c == '&' {
flush(&mut shifted, &mut out);
out.push_str("&-");
} else if is_direct(c) {
flush(&mut shifted, &mut out);
out.push(c);
} else {
let mut buf = [0u16; 2];
shifted.extend_from_slice(c.encode_utf16(&mut buf));
}
}
flush(&mut shifted, &mut out);
out
}
/// Decode a modified UTF-7 mailbox name back to UTF-8. Returns `None` if the
/// input is not valid modified UTF-7.
pub fn decode(input: &str) -> Option<String> {
let mut out = String::with_capacity(input.len());
let bytes = input.as_bytes();
let mut i = 0;
while i < bytes.len() {
let b = bytes[i];
if b == b'&' {
// Find the closing '-'.
let end = bytes[i + 1..].iter().position(|&x| x == b'-')? + i + 1;
let chunk = &input[i + 1..end];
if chunk.is_empty() {
// "&-" => literal '&'
out.push('&');
} else {
let decoded = MODIFIED_B64.decode(chunk.as_bytes()).ok()?;
if decoded.len() % 2 != 0 {
return None;
}
let units: Vec<u16> = decoded
.chunks_exact(2)
.map(|c| u16::from_be_bytes([c[0], c[1]]))
.collect();
out.push_str(&String::from_utf16(&units).ok()?);
}
i = end + 1;
} else {
// Direct ASCII byte.
out.push(b as char);
i += 1;
}
}
Some(out)
}
#[cfg(test)]
mod tests {
use super::*;
fn roundtrip(plain: &str, wire: &str) {
assert_eq!(encode(plain), wire, "encode {plain:?}");
assert_eq!(decode(wire).as_deref(), Some(plain), "decode {wire:?}");
}
#[test]
fn ascii_is_identity() {
roundtrip("INBOX", "INBOX");
roundtrip("Work/Projects", "Work/Projects");
roundtrip("Not Important", "Not Important");
}
#[test]
fn ampersand_is_escaped() {
roundtrip("R&D", "R&-D");
roundtrip("&", "&-");
}
#[test]
fn non_ascii_is_shifted() {
// Examples from RFC 3501.
roundtrip("Café", "Caf&AOk-");
roundtrip("~peter/mail/台北/日本語", "~peter/mail/&U,BTFw-/&ZeVnLIqe-");
}
#[test]
fn mixed_runs() {
roundtrip("Dossier éàü test", "Dossier &AOkA4AD8- test");
}
#[test]
fn decode_rejects_unterminated_shift() {
assert_eq!(decode("&AOk"), None);
}
}