mirror of
https://github.com/rustmailer/bichon.git
synced 2026-08-03 07:48:34 +02:00
fix: inline attachment detection and account-scoped export
- Treat MIME parts with Content-ID but no Content-Disposition as inline - Add account_ids filter to CLI export search to avoid pulling all accounts - Skip failed emails during export instead of aborting the entire batch
This commit is contained in:
Generated
+5
-5
@@ -293,7 +293,7 @@ checksum = "72b3254f16251a8381aa12e40e3c4d2f0199f8c6508fbecb9d91f575e0fbb8c6"
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "bichon-admin"
|
name = "bichon-admin"
|
||||||
version = "1.4.1"
|
version = "1.4.2"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"bichon-core",
|
"bichon-core",
|
||||||
"console",
|
"console",
|
||||||
@@ -311,7 +311,7 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "bichon-cli"
|
name = "bichon-cli"
|
||||||
version = "1.4.1"
|
version = "1.4.2"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"base64 0.22.1",
|
"base64 0.22.1",
|
||||||
"bichon-core",
|
"bichon-core",
|
||||||
@@ -337,7 +337,7 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "bichon-core"
|
name = "bichon-core"
|
||||||
version = "1.4.1"
|
version = "1.4.2"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"async-imap",
|
"async-imap",
|
||||||
"base64 0.22.1",
|
"base64 0.22.1",
|
||||||
@@ -396,7 +396,7 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "bichon-server"
|
name = "bichon-server"
|
||||||
version = "1.4.1"
|
version = "1.4.2"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"bichon-core",
|
"bichon-core",
|
||||||
"bichon-smtp",
|
"bichon-smtp",
|
||||||
@@ -420,7 +420,7 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "bichon-smtp"
|
name = "bichon-smtp"
|
||||||
version = "1.4.1"
|
version = "1.4.2"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"base64 0.22.1",
|
"base64 0.22.1",
|
||||||
"bichon-core",
|
"bichon-core",
|
||||||
|
|||||||
+1
-1
@@ -12,7 +12,7 @@ members = [
|
|||||||
resolver = "2"
|
resolver = "2"
|
||||||
|
|
||||||
[workspace.package]
|
[workspace.package]
|
||||||
version = "1.4.1"
|
version = "1.4.2"
|
||||||
edition = "2021"
|
edition = "2021"
|
||||||
|
|
||||||
[workspace.dependencies]
|
[workspace.dependencies]
|
||||||
|
|||||||
+1
-1
@@ -1,2 +1,2 @@
|
|||||||
base_url = "http://localhost:15630"
|
base_url = "http://localhost:15630"
|
||||||
api_token = "WuqNC0g8yNle7CVnxcvjUwjN"
|
api_token = "eErI7WN3PtKeLwWAbIfSXCP6"
|
||||||
|
|||||||
@@ -10,13 +10,17 @@ use crate::BichonCliConfig;
|
|||||||
pub async fn search_messages(
|
pub async fn search_messages(
|
||||||
client: &Client,
|
client: &Client,
|
||||||
config: &BichonCliConfig,
|
config: &BichonCliConfig,
|
||||||
|
account_ids: Option<std::collections::HashSet<u64>>,
|
||||||
page: u64,
|
page: u64,
|
||||||
page_size: u64,
|
page_size: u64,
|
||||||
) -> Option<DataPage<Envelope>> {
|
) -> Option<DataPage<Envelope>> {
|
||||||
let url = format!("{}/api/v1/search-messages", config.base_url);
|
let url = format!("{}/api/v1/search-messages", config.base_url);
|
||||||
|
|
||||||
let payload = EmailSearchRequest {
|
let payload = EmailSearchRequest {
|
||||||
filter: EmailSearchFilter::default(),
|
filter: EmailSearchFilter {
|
||||||
|
account_ids,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
page,
|
page,
|
||||||
page_size,
|
page_size,
|
||||||
sort_by: Some(SortBy::DATE),
|
sort_by: Some(SortBy::DATE),
|
||||||
|
|||||||
@@ -146,20 +146,23 @@ pub async fn handle_account_export(
|
|||||||
let mut total_pages;
|
let mut total_pages;
|
||||||
|
|
||||||
loop {
|
loop {
|
||||||
if let Some(batch) = search_messages(&client, config, current_page, page_size).await {
|
let account_ids = Some(std::collections::HashSet::from([account.id]));
|
||||||
|
if let Some(batch) = search_messages(&client, config, account_ids, current_page, page_size).await {
|
||||||
total_pages = batch.total_pages.unwrap();
|
total_pages = batch.total_pages.unwrap();
|
||||||
|
|
||||||
pb.set_message(format!("Page {}/{}", current_page, total_pages));
|
pb.set_message(format!("Page {}/{}", current_page, total_pages));
|
||||||
|
|
||||||
for envelope in batch.items {
|
for envelope in batch.items {
|
||||||
let success =
|
let success =
|
||||||
download_and_export_with_json_header(&client, config, envelope, &mut file)
|
download_and_export_with_json_header(&client, config, envelope.clone(), &mut file)
|
||||||
.await;
|
.await;
|
||||||
|
|
||||||
if !success {
|
if !success {
|
||||||
pb.finish_with_message("Failed");
|
eprintln!(
|
||||||
eprintln!(" ✘ Failed to export an email. Aborting process...");
|
" ✘ Failed to export email {}, skipping...",
|
||||||
return;
|
envelope.id
|
||||||
|
);
|
||||||
|
continue;
|
||||||
}
|
}
|
||||||
pb.inc(1);
|
pb.inc(1);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -412,23 +412,36 @@ pub async fn detach_and_store_attachments(
|
|||||||
let mut text_candidates: Vec<TextCandidate> = Vec::new();
|
let mut text_candidates: Vec<TextCandidate> = Vec::new();
|
||||||
|
|
||||||
for (raw_start, raw_end, att) in ranges {
|
for (raw_start, raw_end, att) in ranges {
|
||||||
// Step 2: Extract raw bytes and store them as standalone documents
|
// mail-parser may report attachment offsets past the body end for
|
||||||
let raw_bytes = &original_body[raw_start..raw_end];
|
// malformed messages; clamp the range to avoid a slice panic.
|
||||||
//This is the content hash of the decoded attachment, not the undecoded one.
|
let body_len = original_body.len();
|
||||||
|
let raw_start = raw_start.min(body_len);
|
||||||
|
let raw_end = raw_end.min(body_len);
|
||||||
|
let range_valid = raw_start < raw_end;
|
||||||
|
|
||||||
|
// content hash is computed from the decoded attachment contents,
|
||||||
|
// which is always available regardless of raw offset validity.
|
||||||
let content_hash = compute_content_hash(att.contents());
|
let content_hash = compute_content_hash(att.contents());
|
||||||
|
|
||||||
//"The actual content stored in the blob is the raw undecoded data, to avoid the reconstructed EML differing from the original due to decoding and re-encoding.
|
if range_valid {
|
||||||
attachments.push((content_hash.clone(), Bytes::copy_from_slice(raw_bytes)));//
|
let raw_bytes = &original_body[raw_start..raw_end];
|
||||||
|
// The actual content stored in the blob is the raw undecoded data.
|
||||||
|
attachments.push((content_hash.clone(), Bytes::copy_from_slice(raw_bytes)));
|
||||||
|
|
||||||
// Step 3: Replace raw attachment content with a hash-based placeholder
|
// Replace raw attachment content with a hash-based placeholder
|
||||||
let placeholder = format!("<<BICHON_DETACH_HASH:{}>>", &content_hash);
|
let placeholder = format!("<<BICHON_DETACH_HASH:{}>>", &content_hash);
|
||||||
let p_bytes = placeholder.as_bytes();
|
stripped_eml.splice(raw_start..raw_end, placeholder.as_bytes().iter().cloned());
|
||||||
stripped_eml.splice(raw_start..raw_end, p_bytes.iter().cloned());
|
} else {
|
||||||
|
// Invalid range: store a zero-length blob so the consistency
|
||||||
|
// check passes; reattachment will log a warning for the missing
|
||||||
|
// blob data but won't panic.
|
||||||
|
attachments.push((content_hash.clone(), Bytes::new()));
|
||||||
|
}
|
||||||
|
|
||||||
let inline = att
|
let inline = att
|
||||||
.content_disposition()
|
.content_disposition()
|
||||||
.map(|d| d.is_inline())
|
.map(|d| d.is_inline())
|
||||||
.unwrap_or(false);
|
.unwrap_or_else(|| att.content_id().is_some());
|
||||||
let file_type = att
|
let file_type = att
|
||||||
.content_type()
|
.content_type()
|
||||||
.map(|ct| {
|
.map(|ct| {
|
||||||
@@ -755,4 +768,56 @@ mod test {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Verifies that [`super::detach_and_store_attachments`] does not panic
|
||||||
|
/// when mail-parser reports attachment offsets past the raw body length.
|
||||||
|
///
|
||||||
|
/// Regression test for: "range end index X out of range for slice of
|
||||||
|
/// length Y" panic caused by a malformed email whose attachment
|
||||||
|
/// `raw_end_offset` exceeded the actual body size.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn detach_attachments_bounds_check() {
|
||||||
|
let raw = concat!(
|
||||||
|
"From: sender@example.com\r\n",
|
||||||
|
"To: recipient@example.com\r\n",
|
||||||
|
"Subject: Test\r\n",
|
||||||
|
"MIME-Version: 1.0\r\n",
|
||||||
|
"Content-Type: multipart/mixed; boundary=\"bnd\"\r\n",
|
||||||
|
"\r\n",
|
||||||
|
"--bnd\r\n",
|
||||||
|
"Content-Type: text/plain\r\n",
|
||||||
|
"\r\n",
|
||||||
|
"Hello\r\n",
|
||||||
|
"--bnd\r\n",
|
||||||
|
"Content-Type: application/octet-stream\r\n",
|
||||||
|
"Content-Disposition: attachment; filename=\"test.bin\"\r\n",
|
||||||
|
"\r\n",
|
||||||
|
"AAAAABBBBBCCCCCDDDDDEEEEEAAAAABBBBBCCCCCDDDDDEEEEE\r\n",
|
||||||
|
"--bnd--\r\n",
|
||||||
|
)
|
||||||
|
.as_bytes()
|
||||||
|
.to_vec();
|
||||||
|
|
||||||
|
let message = mail_parser::MessageParser::new()
|
||||||
|
.parse(&raw)
|
||||||
|
.expect("parse valid MIME message");
|
||||||
|
assert_eq!(message.attachment_count(), 1);
|
||||||
|
|
||||||
|
// Truncate the raw body so the attachment's raw_end_offset lies
|
||||||
|
// past the body end — exactly the scenario reported by users.
|
||||||
|
let truncated = &raw[..raw.len() - 20];
|
||||||
|
assert!(truncated.len() < raw.len());
|
||||||
|
|
||||||
|
// Must not panic.
|
||||||
|
let infos = super::detach_and_store_attachments(
|
||||||
|
truncated,
|
||||||
|
&message,
|
||||||
|
"test_content_hash",
|
||||||
|
)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
// The attachment count must still match so the consistency check
|
||||||
|
// in reattach_eml_content doesn't fail later.
|
||||||
|
assert_eq!(infos.len(), 1);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -206,7 +206,9 @@ pub fn retrieve_email_content(
|
|||||||
content_type.c_subtype.as_deref().unwrap_or("")
|
content_type.c_subtype.as_deref().unwrap_or("")
|
||||||
);
|
);
|
||||||
|
|
||||||
let inline = disposition.map(|d| d.is_inline()).unwrap_or(false);
|
let inline = disposition
|
||||||
|
.map(|d| d.is_inline())
|
||||||
|
.unwrap_or_else(|| attachment.content_id().is_some());
|
||||||
|
|
||||||
if inline {
|
if inline {
|
||||||
if let Some(html1) = html.as_deref() {
|
if let Some(html1) = html.as_deref() {
|
||||||
@@ -302,7 +304,9 @@ pub fn retrieve_nested_eml_content(
|
|||||||
for attachment in nested_message.attachments() {
|
for attachment in nested_message.attachments() {
|
||||||
let cid = attachment.content_id();
|
let cid = attachment.content_id();
|
||||||
let disposition = attachment.content_disposition();
|
let disposition = attachment.content_disposition();
|
||||||
let is_inline = disposition.map(|d| d.is_inline()).unwrap_or(false);
|
let is_inline = disposition
|
||||||
|
.map(|d| d.is_inline())
|
||||||
|
.unwrap_or_else(|| cid.is_some());
|
||||||
|
|
||||||
if has_html && is_inline && cid.is_some() {
|
if has_html && is_inline && cid.is_some() {
|
||||||
let content_id = cid.unwrap();
|
let content_id = cid.unwrap();
|
||||||
|
|||||||
@@ -86,13 +86,22 @@ pub fn detach_attachments_standalone(
|
|||||||
|
|
||||||
for (raw_start, raw_end, att) in ranges {
|
for (raw_start, raw_end, att) in ranges {
|
||||||
let content_hash = compute_content_hash(att.contents());
|
let content_hash = compute_content_hash(att.contents());
|
||||||
blobs.push((
|
let body_len = original_body.len();
|
||||||
content_hash.clone(),
|
let raw_start = raw_start.min(body_len);
|
||||||
Bytes::copy_from_slice(&original_body[raw_start..raw_end]),
|
let raw_end = raw_end.min(body_len);
|
||||||
));
|
let range_valid = raw_start < raw_end;
|
||||||
|
|
||||||
let placeholder = format!("<<BICHON_DETACH_HASH:{}>>", &content_hash);
|
if range_valid {
|
||||||
stripped_eml.splice(raw_start..raw_end, placeholder.as_bytes().iter().cloned());
|
blobs.push((
|
||||||
|
content_hash.clone(),
|
||||||
|
Bytes::copy_from_slice(&original_body[raw_start..raw_end]),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
if range_valid {
|
||||||
|
let placeholder = format!("<<BICHON_DETACH_HASH:{}>>", &content_hash);
|
||||||
|
stripped_eml.splice(raw_start..raw_end, placeholder.as_bytes().iter().cloned());
|
||||||
|
}
|
||||||
|
|
||||||
infos.push(AttachmentInfo {
|
infos.push(AttachmentInfo {
|
||||||
filename: att.attachment_name().map(|n| n.to_string()),
|
filename: att.attachment_name().map(|n| n.to_string()),
|
||||||
@@ -100,7 +109,7 @@ pub fn detach_attachments_standalone(
|
|||||||
inline: att
|
inline: att
|
||||||
.content_disposition()
|
.content_disposition()
|
||||||
.map(|d| d.is_inline())
|
.map(|d| d.is_inline())
|
||||||
.unwrap_or(false),
|
.unwrap_or_else(|| att.content_id().is_some()),
|
||||||
file_type: att
|
file_type: att
|
||||||
.content_type()
|
.content_type()
|
||||||
.map(|ct| {
|
.map(|ct| {
|
||||||
|
|||||||
Reference in New Issue
Block a user