import: bound the memory an IMAP or EWS import holds
ci / test (pull_request) Skipped
github/ci (branch) GitHub Actions
ci / github (pull_request) Successful in 2m50s
ci / announce (pull_request) Skipped

Two things let a large mailbox fill memory. IMAP workers handed every
fetched message to the archive writer through an unbounded queue, so fast
workers could hold whole folders' worth of bodies while the single writer
caught up. And fetch batches were sized by count only: 20 items per EWS
GetItem, 8 connections at a time, is a few megabytes of ordinary mail and
several gigabytes of large attachments.

- The IMAP event queue is bounded at two events per worker, so a worker
  waits for the writer instead of running ahead of it.
- Fetch batches are bounded by bytes as well as by count, with a shared
  helper, `sync::batch::by_count_and_bytes`: 32 MiB by default, and a
  single larger message goes alone.
  - IMAP learns each new message's RFC822.SIZE on the control connection,
    in 1000-UID metadata fetches, before the body fetch. A server that won't
    say leaves the chunks sized by count. `--fetch-batch-mib` sets the cap.
  - EWS asks FindItem for `item:Size` and packs GetItem batches by it,
    fetched `--ews-connections` batches at a time. `--ews-getitem-batch-mib`
    sets the cap. Items from an incremental SyncFolderItems run carry no
    size and stay batched by count.
- docs/usage.md describes both options.
This commit is contained in:
2026-09-30 12:32:29 -07:00
parent 2f33cd76a1
commit 3cae9464f0
18 changed files with 551 additions and 128 deletions
+1
View File
@@ -38,6 +38,7 @@ fn imap_config(account: &Account, imap: &Endpoint) -> ImapImportConfig {
automap: true,
include_deleted: false,
fetch_batch: 64,
fetch_batch_bytes: inbuxa_migrate::sync::batch::DEFAULT_BATCH_BYTES,
imap_connections: 2,
allow_source_change: false,
}
+1
View File
@@ -43,6 +43,7 @@ fn imap_config(account: &Account, imap: &integration::Endpoint) -> ImapImportCon
automap: true,
include_deleted: false,
fetch_batch: 64,
fetch_batch_bytes: inbuxa_migrate::sync::batch::DEFAULT_BATCH_BYTES,
imap_connections: 2,
allow_source_change: false,
}
+75 -6
View File
@@ -864,6 +864,7 @@ fn for_each_fetched_item_streams_every_id_across_windows() {
url: &url,
source_id: 1,
batch_size: 1,
batch_bytes: inbuxa_migrate::sync::batch::DEFAULT_BATCH_BYTES,
attachment_batch: 1,
connections: 2,
use_syncfolderitems: false,
@@ -873,18 +874,86 @@ fn for_each_fetched_item_streams_every_id_across_windows() {
let ids: Vec<ItemId> = (0..5).map(|i| ItemId::new(format!("I{i}"), "K")).collect();
let mut delivered = 0usize;
let failed = for_each_fetched_item(&ctx, ItemShape::Message, &ids, |msg| {
assert!(msg.success);
delivered += 1;
Ok(())
})
.expect("streaming fetch should succeed");
let failed =
for_each_fetched_item(&ctx, ItemShape::Message, &ids, &Default::default(), |msg| {
assert!(msg.success);
delivered += 1;
Ok(())
})
.expect("streaming fetch should succeed");
assert_eq!(delivered, 5, "every id must be delivered exactly once");
assert_eq!(failed, 0);
_m.assert();
}
#[test]
fn getitem_batches_are_split_by_bytes_when_sizes_are_known() {
use inbuxa_migrate::logging::Logger;
use inbuxa_migrate::sync::import_exchange_ews::items::{ItemRunCtx, for_each_fetched_item};
use std::collections::HashMap;
let mut server = mockito::Server::new();
let url = format!("{}/EWS/Exchange.asmx", server.url());
let one_message = envelope(&format!(
"<m:GetItemResponse{NS}><m:ResponseMessages><m:GetItemResponseMessage ResponseClass=\"Success\">\
<m:ResponseCode>NoError</m:ResponseCode><m:Items><t:Message><t:ItemId Id=\"X\" ChangeKey=\"K\"/></t:Message></m:Items>\
</m:GetItemResponseMessage></m:ResponseMessages></m:GetItemResponse>"
));
// Ten items fit one batch by count, but at 20 bytes each and a 30-byte
// cap every item goes alone: four GetItem calls, not one.
let m = server
.mock("POST", "/EWS/Exchange.asmx")
.with_status(200)
.with_header("content-type", TXT_XML)
.with_body(&one_message)
.expect(4)
.create();
let c = client(0);
let ctx = ItemRunCtx {
client: &c,
url: &url,
source_id: 1,
batch_size: 10,
batch_bytes: 30,
attachment_batch: 1,
connections: 2,
use_syncfolderitems: false,
sync_batch: 512,
logger: Logger::new(0),
};
let ids: Vec<ItemId> = (0..4).map(|i| ItemId::new(format!("I{i}"), "K")).collect();
let sizes: HashMap<String, u64> = ids.iter().map(|id| (id.id.clone(), 20)).collect();
let mut delivered = 0usize;
for_each_fetched_item(&ctx, ItemShape::Message, &ids, &sizes, |_| {
delivered += 1;
Ok(())
})
.expect("fetch");
assert_eq!(delivered, 4);
m.assert();
}
#[test]
fn find_item_reports_each_items_size() {
use inbuxa_migrate::exchange_ews::parse::parse_find_item_response;
let body = envelope(&format!(
"<m:FindItemResponse{NS}><m:ResponseMessages><m:FindItemResponseMessage ResponseClass=\"Success\">\
<m:ResponseCode>NoError</m:ResponseCode>\
<m:RootFolder TotalItemsInView=\"2\" IncludesLastItemInRange=\"true\"><t:Items>\
<t:Message><t:ItemId Id=\"A\" ChangeKey=\"K\"/><t:Size>1234</t:Size></t:Message>\
<t:Message><t:ItemId Id=\"B\" ChangeKey=\"K\"/></t:Message>\
</t:Items></m:RootFolder></m:FindItemResponseMessage></m:ResponseMessages></m:FindItemResponse>"
));
let r = parse_find_item_response(body.as_bytes()).unwrap();
assert_eq!(r.items.len(), 2);
assert_eq!(
(r.items[0].id.id.as_str(), r.items[0].size),
("A", Some(1234))
);
assert_eq!((r.items[1].id.id.as_str(), r.items[1].size), ("B", None));
}
#[test]
fn warning_response_class_is_treated_as_success_in_mock() {
let body = envelope(&format!(
+86
View File
@@ -169,6 +169,7 @@ fn run_import(
automap: true,
include_deleted: false,
fetch_batch: 256,
fetch_batch_bytes: inbuxa_migrate::sync::batch::DEFAULT_BATCH_BYTES,
imap_connections: 1,
allow_source_change: false,
};
@@ -2135,3 +2136,88 @@ fn a_failed_chunk_keeps_the_ones_before_it_and_a_rerun_fetches_only_what_is_miss
db::init::apply_schema(&conn).unwrap();
assert_eq!(count(&conn, "emails"), 2);
}
#[test]
fn body_fetches_are_split_by_message_size() {
// Three 20-byte messages under a 30-byte cap: the control connection
// learns the sizes first, and each body is then fetched on its own.
let control: Script = Box::new(|conn: &mut MockConn| -> std::io::Result<()> {
auth_preamble(conn, "IMAP4rev2 LITERAL+ AUTH=PLAIN")?;
let (tag, _) = conn.read_command()?;
conn.write_line("* LIST () \"/\" \"INBOX\"")?;
conn.write_line(&format!("{tag} OK"))?;
let (tag, _) = conn.read_command()?;
conn.write_line(&format!("{tag} OK"))?;
let (tag, _) = conn.read_command()?;
write_select(conn, &tag, 555, 4, 3)?;
let (tag, _) = conn.read_command()?;
conn.write_line("* SEARCH 1 2 3")?;
conn.write_line(&format!("{tag} OK"))?;
let (tag, cmd) = conn.read_command()?;
assert_eq!(cmd, "UID FETCH 1:3 (UID RFC822.SIZE)");
for uid in 1..=3 {
conn.write_line(&format!("* {uid} FETCH (UID {uid} RFC822.SIZE 20)"))?;
}
conn.write_line(&format!("{tag} OK"))?;
drain_until_close(conn);
Ok(())
});
let fetches = std::sync::Arc::new(Mutex::new(Vec::<String>::new()));
let worker: Script = {
let fetches = fetches.clone();
Box::new(move |conn: &mut MockConn| -> std::io::Result<()> {
auth_preamble(conn, "IMAP4rev2 LITERAL+ AUTH=PLAIN")?;
let (tag, _) = conn.read_command()?;
write_select(conn, &tag, 555, 4, 3)?;
let bodies: [&[u8]; 3] = [BODY_A, BODY_B, BODY_C];
for _ in 0..3 {
let (tag, cmd) = conn.read_command()?;
let uid: u32 = cmd
.strip_prefix("UID FETCH ")
.and_then(|r| r.split(' ').next())
.and_then(|n| n.parse().ok())
.unwrap_or_else(|| panic!("expected a single-UID fetch, got {cmd}"));
fetches.lock().unwrap().push(cmd.clone());
write_fetch_message(conn, uid, uid, bodies[(uid - 1) as usize])?;
conn.write_line(&format!("{tag} OK"))?;
}
drain_until_close(conn);
Ok(())
})
};
let server = MockImap::start_scripts(vec![control, worker]);
let archive = tempfile("byte-chunks");
let summary =
run_import(&server, "alice", archive, |c| c.fetch_batch_bytes = 30).expect("import");
assert_eq!(email_counts(&summary).created, 3, "summary={summary:?}");
assert_eq!(
fetches.lock().unwrap().len(),
3,
"one body fetch per message"
);
}
#[test]
fn a_chunk_larger_than_the_event_queue_completes() {
// One worker holds at most two events in flight; a ten-message chunk
// has to wait on the archive writer, not deadlock on it.
const UIDS: &[u32] = &[1, 2, 3, 4, 5, 6, 7, 8, 9, 10];
let worker: Script = Box::new(|conn: &mut MockConn| -> std::io::Result<()> {
auth_preamble(conn, "IMAP4rev2 LITERAL+ AUTH=PLAIN")?;
let (tag, _) = conn.read_command()?;
write_select(conn, &tag, 999, 11, 10)?;
let (tag, cmd) = conn.read_command()?;
assert!(cmd.starts_with("UID FETCH 1:10 "), "got {cmd}");
for uid in UIDS {
let body = format!("From: a@b\r\nMessage-ID: <{uid}@h>\r\n\r\nbody {uid}");
write_fetch_message(conn, *uid, *uid, body.as_bytes())?;
}
conn.write_line(&format!("{tag} OK"))?;
drain_until_close(conn);
Ok(())
});
let server = MockImap::start_scripts(vec![control_script_one_folder(999, 11, UIDS), worker]);
let archive = tempfile("backpressure");
let summary = run_import(&server, "alice", archive, |_| {}).expect("import");
assert_eq!(email_counts(&summary).created, 10, "summary={summary:?}");
}
+1
View File
@@ -71,6 +71,7 @@ fn imap_basic_config(localpart: &str) -> ImapImportConfig {
automap: true,
include_deleted: false,
fetch_batch: 256,
fetch_batch_bytes: inbuxa_migrate::sync::batch::DEFAULT_BATCH_BYTES,
imap_connections: 4,
allow_source_change: false,
}