import: bound the memory an IMAP or EWS import holds
ci / test (pull_request) Skipped
github/ci (branch) GitHub Actions
ci / github (pull_request) Successful in 2m50s
ci / announce (pull_request) Skipped

Two things let a large mailbox fill memory. IMAP workers handed every
fetched message to the archive writer through an unbounded queue, so fast
workers could hold whole folders' worth of bodies while the single writer
caught up. And fetch batches were sized by count only: 20 items per EWS
GetItem, 8 connections at a time, is a few megabytes of ordinary mail and
several gigabytes of large attachments.

- The IMAP event queue is bounded at two events per worker, so a worker
  waits for the writer instead of running ahead of it.
- Fetch batches are bounded by bytes as well as by count, with a shared
  helper, `sync::batch::by_count_and_bytes`: 32 MiB by default, and a
  single larger message goes alone.
  - IMAP learns each new message's RFC822.SIZE on the control connection,
    in 1000-UID metadata fetches, before the body fetch. A server that won't
    say leaves the chunks sized by count. `--fetch-batch-mib` sets the cap.
  - EWS asks FindItem for `item:Size` and packs GetItem batches by it,
    fetched `--ews-connections` batches at a time. `--ews-getitem-batch-mib`
    sets the cap. Items from an incremental SyncFolderItems run carry no
    size and stay batched by count.
- docs/usage.md describes both options.
This commit is contained in:
2026-09-30 12:32:29 -07:00
parent 2f33cd76a1
commit 3cae9464f0
18 changed files with 551 additions and 128 deletions
+44 -37
View File
@@ -1,5 +1,6 @@
/*
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <[email protected]>
* SPDX-FileCopyrightText: 2026 John Coffey <[email protected]>
*
* SPDX-License-Identifier: Apache-2.0 OR MIT
*/
@@ -104,47 +105,53 @@ fn reconcile_one(
to_fetch.push(id.clone());
}
if !to_fetch.is_empty() {
let failed_items = for_each_fetched_item(ctx, ItemShape::CalendarItem, &to_fetch, |msg| {
if !msg.success {
let failed_items = for_each_fetched_item(
ctx,
ItemShape::CalendarItem,
&to_fetch,
&outcome.sizes,
|msg| {
if !msg.success {
if matches!(
msg.response_code,
crate::exchange_ews::types::ResponseCode::ItemNotFound
) {
counts.skipped += 1;
} else {
counts.failed += 1;
ctx.logger
.warn(&format!("GetItem (calendar) error: {}", msg.response_code));
}
return Ok(());
}
let parsed = parse_calendar_item(&msg.inner_xml).map_err(Error::from)?;
if parsed.id.id.is_empty() {
counts.failed += 1;
return Ok(());
}
if matches!(
msg.response_code,
crate::exchange_ews::types::ResponseCode::ItemNotFound
parsed.calendar_item_type,
Some(CalendarItemType::Occurrence) | Some(CalendarItemType::Exception)
) {
counts.skipped += 1;
} else {
counts.failed += 1;
ctx.logger
.warn(&format!("GetItem (calendar) error: {}", msg.response_code));
return Ok(());
}
return Ok(());
}
let parsed = parse_calendar_item(&msg.inner_xml).map_err(Error::from)?;
if parsed.id.id.is_empty() {
counts.failed += 1;
return Ok(());
}
if matches!(
parsed.calendar_item_type,
Some(CalendarItemType::Occurrence) | Some(CalendarItemType::Exception)
) {
counts.skipped += 1;
return Ok(());
}
let existing = plan
.present_changed
.iter()
.find(|(id, _)| id.id == parsed.id.id)
.map(|(_, local)| *local);
apply_event(
conn,
ctx,
&parsed,
local_folder_id,
&folder.id,
existing,
counts,
)
})?;
let existing = plan
.present_changed
.iter()
.find(|(id, _)| id.id == parsed.id.id)
.map(|(_, local)| *local);
apply_event(
conn,
ctx,
&parsed,
local_folder_id,
&folder.id,
existing,
counts,
)
},
)?;
counts.failed += failed_items;
}
delete_vanished(