Author SHA1 Message Date
jcoffey-dev 22d8ad8572 Publish amd64 first, then arm64, on a builder that keeps its cache
ci / fork-checks (pull_request) Successful in 49s
ci / build (pull_request) Successful in 4m30s
Two release builds side by side on one machine each take twice as long,
and production only needs amd64. publish-amd64 now pushes :<version> as
soon as the amd64 build is done; publish-arm64 builds arm64 afterwards,
then replaces :<version> with the two-platform index and moves :latest.

Both jobs use one named BuildKit builder whose container outlives the
job, so the dependency layer (cargo chef cook) is reused until the
dependencies change. The release is created after amd64; the binaries
are attached once arm64 is in.
2026-09-24 08:04:53 -07:00
21 changed files with 219 additions and 846 deletions
+69 -15
View File
@@ -3,11 +3,28 @@
# whether a person pushed it or weekly-release.yml created it through the
# releases API.
#
# The image is multi-arch (linux/amd64, linux/arm64) as before, but built in
# one buildx run on host1 instead of one native runner per architecture: the
# Dockerfile's builder stage runs on the build platform and cross-compiles
# with an aarch64 linker, so only the small final stage (apt, setcap) goes
# through QEMU for arm64. No digest-joining job is needed.
# The image is multi-arch (linux/amd64, linux/arm64), built by two jobs on
# the image-build runner rather than one buildx run for both. The Dockerfile's
# builder stage runs on the build platform and cross-compiles with an aarch64
# linker, so only the small final stage (apt, setcap) goes through QEMU for
# arm64 -- but two release builds (LTO, one codegen unit) side by side on one
# machine each take twice as long. Production runs amd64, so amd64 goes first
# and on its own:
# * publish-amd64 pushes :<version>-amd64 and :<version>, a plain amd64
# image, as soon as its build is done. A deploy can start from it.
# * publish-arm64 then builds arm64, pushes :<version>-arm64, and replaces
# :<version> with the two-platform index. :latest moves only here, so it
# never names an image without arm64.
#
# Both jobs use one BuildKit builder, `gitea-builder`, whose container
# (buildx_buildkit_gitea-builder0) and state volume stay on the runner's host
# between jobs: a job container's `buildx create` finds the existing container
# and reuses it and its cache. The dependency build (`cargo chef cook`) is
# keyed on the recipe, which only a dependency change alters, so a release
# normally compiles just the workspace. Removing that container or its volume
# costs the next release a cold build, nothing more. The planner and dependency
# layers for the build platform are shared, so arm64 also reuses what amd64
# just did where it can.
#
# Two guards before anything is pushed:
# * the tag must be v<brand_version!>. The version is a string in
@@ -62,7 +79,7 @@ jobs:
echo "version=$V" >> "$GITHUB_OUTPUT"
echo "version $V"
publish:
publish-amd64:
needs: [version]
runs-on: docker
container:
@@ -81,16 +98,15 @@ jobs:
test -n "$REGISTRY" && test -n "$VERSION"
test -n "$PACKAGE_TOKEN" || { echo "PACKAGE_TOKEN secret is not set on this repository" >&2; exit 1; }
echo "$PACKAGE_TOKEN" | docker login -u jcoffey-dev --password-stdin "$REGISTRY"
docker run --privileged --rm tonistiigi/binfmt --install arm64
docker buildx create --use --name gitea-builder --driver docker-container || docker buildx use gitea-builder
# Attestations off, as before: they add manifests of their own to the
# index, and the index should hold the two images and nothing else.
# Attestations off, as before: they add manifests of their own, and the
# index should hold the two images and nothing else.
- run: |
docker buildx build \
--platform linux/amd64,linux/arm64 \
--platform linux/amd64 \
--provenance=false --sbom=false \
--tag "$IMAGE:$VERSION-amd64" \
--tag "$IMAGE:$VERSION" \
--tag "$IMAGE:latest" \
--push .
docker buildx imagetools inspect "$IMAGE:$VERSION"
# Gitea keeps a container package on its owner; linking it shows it on
@@ -103,11 +119,47 @@ jobs:
- if: always()
run: docker logout "$REGISTRY" || true
publish-arm64:
needs: [version, publish-amd64]
runs-on: docker
container:
image: docker:28-cli@sha256:625d9431a9f54c5a2bc90f24f0e1c3d55b1349fd857dd85035f98c2c9acbdd4d # 28-cli
volumes:
- /var/run/docker.sock:/var/run/docker.sock
env:
DOCKER_BUILDKIT: "1"
REGISTRY: ${{ vars.REGISTRY }}
IMAGE: ${{ vars.REGISTRY }}/${{ github.repository }}
VERSION: ${{ needs.version.outputs.version }}
PACKAGE_TOKEN: ${{ secrets.PACKAGE_TOKEN }}
steps:
- uses: coffey-labs/actions/checkout@fab0c4d45e0162963965f1555df27b7bed5e20ec
- run: |
echo "$PACKAGE_TOKEN" | docker login -u jcoffey-dev --password-stdin "$REGISTRY"
docker run --privileged --rm tonistiigi/binfmt --install arm64
docker buildx create --use --name gitea-builder --driver docker-container || docker buildx use gitea-builder
# The index is built from the two per-architecture tags rather than from
# :<version>, which by now is the amd64 image and would be read as such.
- run: |
docker buildx build \
--platform linux/arm64 \
--provenance=false --sbom=false \
--tag "$IMAGE:$VERSION-arm64" \
--push .
docker buildx imagetools create \
--tag "$IMAGE:$VERSION" \
--tag "$IMAGE:latest" \
"$IMAGE:$VERSION-amd64" "$IMAGE:$VERSION-arm64"
docker buildx imagetools inspect "$IMAGE:$VERSION"
- if: always()
run: docker logout "$REGISTRY" || true
# The weekly release creates its Release (and so the tag) first; a tag
# pushed by hand has none. Either way the tag ends up with exactly one
# Release, created after the image exists so its pull instructions work.
# Release, created once the amd64 image exists so its pull instructions
# work; arm64 and the binaries follow.
release:
needs: [version, publish]
needs: [version, publish-amd64]
runs-on: light
container:
image: python:3.13-slim@sha256:8d9d0b8bcf6506481eae4907c18f5e3e7902e629f5f6d684f9e7c32e85e3ddf0 # 3.13-slim
@@ -131,7 +183,9 @@ jobs:
except urllib.error.HTTPError as e:
if e.code != 404: raise
image = f"{os.environ['REGISTRY']}/{os.environ['REPO']}:{version}"
body = (f"Container image: `{image}` (linux/amd64, linux/arm64); also `:latest`.\n\n"
body = (f"Container image: `{image}` (linux/amd64, linux/arm64); also `:latest`. "
"amd64 is published first; arm64 is added to the same tag when its build "
"finishes, and `:latest` moves then.\n\n"
"Binaries for a host install are attached: `inbuxa-linux-amd64.tar.gz` and "
"`inbuxa-linux-arm64.tar.gz`, with `SHA256SUMS`. Each is the binary out of this "
"release's image for that architecture, so it is the same build. The image "
@@ -154,7 +208,7 @@ jobs:
# `docker create` does not start anything, so pulling an arm64 image on an
# amd64 runner and copying a file out of it needs no emulation.
binaries:
needs: [version, publish, release]
needs: [version, publish-arm64, release]
runs-on: docker
container:
image: docker:28-cli@sha256:625d9431a9f54c5a2bc90f24f0e1c3d55b1349fd857dd85035f98c2c9acbdd4d # 28-cli
-54
View File
@@ -335,57 +335,3 @@ impl EmailPush {
}
}
}
/// inbuxa: the task locks this node holds, so a graceful stop can hand them
/// back instead of leaving the tasks blocked until the locks expire.
pub struct TaskLocks {
held: parking_lot::Mutex<ahash::AHashSet<u64>>,
stopping: AtomicBool,
expiry: std::sync::atomic::AtomicU64,
}
impl TaskLocks {
/// How long a task lock lasts, in seconds, unless it is released first.
pub const DEFAULT_EXPIRY: u64 = 60 * 60;
pub fn is_stopping(&self) -> bool {
self.stopping.load(Ordering::Acquire)
}
/// Stops new claims and returns the ids of every lock still held.
pub fn stop(&self) -> Vec<u64> {
self.stopping.store(true, Ordering::Release);
self.held.lock().drain().collect()
}
pub fn insert(&self, id: u64) {
self.held.lock().insert(id);
}
pub fn remove(&self, id: u64) {
self.held.lock().remove(&id);
}
pub fn held(&self) -> usize {
self.held.lock().len()
}
pub fn expiry(&self) -> u64 {
self.expiry.load(Ordering::Relaxed)
}
/// Changes the lock lifetime; the tests shorten it.
pub fn set_expiry(&self, seconds: u64) {
self.expiry.store(seconds.max(1), Ordering::Relaxed);
}
}
impl Default for TaskLocks {
fn default() -> Self {
Self {
held: Default::default(),
stopping: AtomicBool::new(false),
expiry: std::sync::atomic::AtomicU64::new(Self::DEFAULT_EXPIRY),
}
}
}
-2
View File
@@ -279,8 +279,6 @@ pub struct HttpAuthCache {
pub struct Ipc {
pub push_tx: mpsc::Sender<PushEvent>,
pub task_tx: Arc<Notify>,
// inbuxa: task locks held by this node, released on a graceful stop
pub task_locks: Arc<crate::ipc::TaskLocks>,
pub queue_tx: mpsc::Sender<QueueEvent>,
pub report_tx: mpsc::Sender<ReportingEvent>,
pub broadcast_tx: Option<mpsc::Sender<BroadcastEvent>>,
-1
View File
@@ -297,7 +297,6 @@ pub fn build_ipc(has_pubsub: bool) -> (Ipc, IpcReceivers) {
report_tx,
broadcast_tx: has_pubsub.then_some(broadcast_tx),
task_tx: Arc::new(Notify::new()),
task_locks: Arc::new(crate::ipc::TaskLocks::default()),
train_task_controller: Arc::new(TrainTaskController::default()),
},
IpcReceivers {
+1 -1
View File
@@ -8,7 +8,7 @@ store = { path = "../store" }
registry = { path = "../registry" }
trc = { path = "../trc" }
futures = { version = "0.3", optional = true }
tokio = { version = "1.53", features = ["sync", "fs", "io-util", "rt", "time"] }
tokio = { version = "1.53", features = ["sync", "fs", "io-util"] }
async-nats = { version = "0.50", default-features = false, features = ["server_2_10", "server_2_11", "aws-lc-rs"], optional = true }
zenoh = { version = "1.10.0", default-features = false, features = ["auth_pubkey", "transport_multilink", "transport_compression", "transport_quic", "transport_tcp", "transport_tls", "transport_udp"], optional = true }
rdkafka = { version = "0.39", features = ["cmake-build"], optional = true }
+2 -118
View File
@@ -2,22 +2,13 @@
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <[email protected]>
*
* SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL
*
* Modified by Coffey Labs in 2026 for INBUXA.
*/
use std::{
sync::{
Arc,
atomic::{AtomicBool, Ordering},
},
time::Duration,
};
use std::sync::Arc;
use crate::Coordinator;
use async_nats::Client;
use registry::schema::structs::NatsCoordinator;
use trc::ClusterEvent;
pub mod pubsub;
@@ -56,116 +47,9 @@ impl NatsPubSub {
opts = opts.token(credentials);
}
// inbuxa: connect in the background and keep trying, so a node that
// starts while NATS is down still joins the cluster once NATS is
// back, instead of running without a coordinator until restarted;
// and report the connection going and coming back
let reporter = Arc::new(Reporter::default());
opts = opts.retry_on_initial_connect().event_callback({
let reporter = reporter.clone();
move |event| {
let reporter = reporter.clone();
async move { reporter.report(event) }
}
});
let connection_timeout = config.timeout_connection.into_inner();
async_nats::connect_with_options(config.addresses.into_inner(), opts)
.await
.map(|client| {
reporter.watch_first_connection(client.clone(), connection_timeout);
Coordinator::Nats(Arc::new(NatsPubSub { client }))
})
.map(|client| Coordinator::Nats(Arc::new(NatsPubSub { client })))
.map_err(|err| format!("Failed to connect to Nats: {}", err))
}
/// inbuxa: whether the client is connected to a NATS server right now.
pub fn is_connected(&self) -> bool {
matches!(
self.client.connection_state(),
async_nats::connection::State::Connected
)
}
}
/// inbuxa: reports the client's connection events as the server's own.
#[derive(Default)]
struct Reporter {
connected_once: AtomicBool,
// A failed attempt raises an error each time the client retries, every
// few seconds while NATS is down: report the first after each change
error_reported: AtomicBool,
}
impl Reporter {
fn report(&self, event: async_nats::Event) {
match event {
async_nats::Event::Connected => {
self.connected_once.store(true, Ordering::Relaxed);
self.error_reported.store(false, Ordering::Relaxed);
trc::event!(Cluster(ClusterEvent::CoordinatorConnected), Type = "nats");
}
async_nats::Event::Disconnected => {
self.error_reported.store(false, Ordering::Relaxed);
trc::event!(
Cluster(ClusterEvent::CoordinatorDisconnected),
Type = "nats",
Details = "Connection lost; reconnecting in the background",
);
}
async_nats::Event::Closed => {
trc::event!(
Cluster(ClusterEvent::CoordinatorDisconnected),
Type = "nats",
Details = "Connection closed; no further attempts will be made",
);
}
async_nats::Event::ClientError(async_nats::ClientError::MaxReconnects) => {
trc::event!(
Cluster(ClusterEvent::CoordinatorDisconnected),
Type = "nats",
Details = "Gave up reconnecting (maxReconnects reached)",
);
}
async_nats::Event::ClientError(err) => {
if !self.error_reported.swap(true, Ordering::Relaxed) {
trc::event!(
Cluster(ClusterEvent::CoordinatorError),
Type = "nats",
Details = "Connection attempt failed; retrying",
Reason = err.to_string(),
);
}
}
event => {
trc::event!(
Cluster(ClusterEvent::CoordinatorError),
Type = "nats",
Details = event.to_string(),
);
}
}
}
/// The first connection is made in the background, so say so when it
/// hasn't been made within the connection timeout. The client keeps
/// trying, and reports the connection when it comes.
fn watch_first_connection(self: &Arc<Self>, client: Client, timeout: Duration) {
let reporter = self.clone();
tokio::spawn(async move {
tokio::time::sleep(timeout).await;
if !reporter.connected_once.load(Ordering::Relaxed)
&& !matches!(
client.connection_state(),
async_nats::connection::State::Connected
)
{
trc::event!(
Cluster(ClusterEvent::CoordinatorDisconnected),
Type = "nats",
Details = "Not connected at startup; retrying in the background",
);
}
});
}
}
-13
View File
@@ -2,8 +2,6 @@
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <[email protected]>
*
* SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL
*
* Modified by Coffey Labs in 2026 for INBUXA.
*/
use crate::{Coordinator, Msg, PubSubStream};
@@ -45,17 +43,6 @@ impl Coordinator {
pub fn is_none(&self) -> bool {
matches!(self, Coordinator::None)
}
/// inbuxa: whether the coordinator is connected right now, for the
/// backends that track it (NATS); `None` for the others and when no
/// coordinator is configured.
pub fn is_connected(&self) -> Option<bool> {
match self {
#[cfg(feature = "nats")]
Coordinator::Nats(store) => Some(store.is_connected()),
_ => None,
}
}
}
impl PubSubStream {
-21
View File
@@ -562,27 +562,6 @@ impl ParseHttp for Server {
})
.into_http_response());
}
// inbuxa: the cluster coordinator's connection, for
// monitoring. It stays out of live and ready on purpose:
// a node without its coordinator still serves mail, and
// failing those would have an orchestrator restart, or
// take out of service, every node at once when the
// coordinator goes down
"cluster" => {
let coordinator = &self.core.storage.coordinator;
let (status, state) = match coordinator.is_connected() {
Some(true) => (StatusCode::OK, "connected"),
Some(false) => (StatusCode::SERVICE_UNAVAILABLE, "disconnected"),
None if coordinator.is_none() => (StatusCode::OK, "none"),
None => (StatusCode::OK, "unknown"),
};
return Ok(http_proto::JsonResponse::with_status(
status,
serde_json::json!({ "coordinator": state }),
)
.no_cache()
.into_http_response());
}
_ => (),
}
}
-4
View File
@@ -109,10 +109,6 @@ async fn main() -> std::io::Result<()> {
// Wait for shutdown signal
wait_for_shutdown().await;
// inbuxa: hand back the task locks this node holds, so other nodes can
// run those tasks now rather than when the locks expire
services::task_manager::lock::release_task_locks(&inner.build_server()).await;
// Shutdown collector
Collector::shutdown();
+1 -9
View File
@@ -91,15 +91,7 @@ impl SearchIndexTask for Server {
build_contact_document(self, account_id, document_id).await
}
IndexDocumentType::File => {
// File indexing not implemented yet. inbuxa: still
// one result per task: update_tasks pairs them by
// position, and a missing one shifts every result
// after it onto the wrong task
results.push(IndexTaskResult {
task_type: TaskType::Insert,
index: task.document_type,
result: TaskResult::Ignored,
});
// File indexing not implemented yet
continue;
}
};
+2 -34
View File
@@ -2,8 +2,6 @@
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <[email protected]>
*
* SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL
*
* Modified by Coffey Labs in 2026 for INBUXA.
*/
use crate::task_manager::*;
@@ -15,21 +13,13 @@ pub trait TaskLockManager: Sync + Send {
impl TaskLockManager for Server {
async fn try_lock_task(&self, id: u64) -> bool {
// inbuxa: a node that is stopping claims nothing new
let locks = &self.inner.ipc.task_locks;
if locks.is_stopping() {
return false;
}
match self
.in_memory_store()
.try_lock(KV_LOCK_TASK, &id.to_be_bytes(), locks.expiry())
.try_lock(KV_LOCK_TASK, &id.to_be_bytes(), DEFAULT_LOCK_EXPIRY)
.await
{
Ok(result) => {
if result {
locks.insert(id);
} else {
if !result {
trc::event!(
TaskManager(TaskManagerEvent::TaskLocked),
Id = id,
@@ -58,27 +48,5 @@ impl TaskLockManager for Server {
.caused_by(trc::location!())
);
}
self.inner.ipc.task_locks.remove(id);
}
}
/// inbuxa: on a graceful stop, stops claiming tasks and releases every task
/// lock this node holds, so the rest of the cluster can pick the tasks up at
/// once instead of after the lock expires. Returns how many were released.
pub async fn release_task_locks(server: &Server) -> usize {
let ids = server.inner.ipc.task_locks.stop();
for id in &ids {
if let Err(err) = server
.in_memory_store()
.remove_lock(KV_LOCK_TASK, &id.to_be_bytes())
.await
{
trc::error!(
err.details("Failed to release task lock on shutdown")
.ctx(trc::Key::Id, *id)
.caused_by(trc::location!())
);
}
}
ids.len()
}
+140 -197
View File
@@ -2,8 +2,6 @@
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <[email protected]>
*
* SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL
*
* Modified by Coffey Labs in 2026 for INBUXA.
*/
use crate::task_manager::acme::AcmeTask;
@@ -20,7 +18,7 @@ use crate::task_manager::report::{self, SubmitReportTask};
use crate::task_manager::restore_item::RestoreItemTask;
use crate::task_manager::spam_classifier::SpamFilterMaintenanceTask;
use crate::task_manager::{
CLAIM_RECHECK_INTERVAL, Locked, QUEUE_REFRESH_INTERVAL, TaskDetails, TaskFailureType, TaskInfo,
DEFAULT_LOCK_EXPIRY, Locked, QUEUE_REFRESH_INTERVAL, TaskDetails, TaskFailureType, TaskInfo,
TaskJob, TaskManagerIpc, TaskResult,
};
use common::BuildServer;
@@ -126,47 +124,72 @@ pub fn spawn_task_manager(inner: Arc<Inner>) {
let server = inner.build_server();
let batch_size = server.core.email.index_batch_size;
let mut batch = Vec::with_capacity(batch_size);
if let Some(task) = fetch_task(&server, job).await {
batch.push(task);
match server
.store()
.get_value::<Task>(ValueKey::from(ValueClass::TaskQueue(
TaskQueueClass::Task { id: job.id },
)))
.await
{
Ok(Some(task)) => {
batch.push(TaskDetails { task, info: job });
}
Ok(None) => {
trc::event!(
TaskManager(TaskManagerEvent::TaskIgnored),
Id = job.id,
Reason = "Task not found in store, likely already processed.",
);
}
Err(err) => {
trc::error!(
err.id(job.id)
.details("Failed to retrieve task details.")
.caused_by(trc::location!())
);
}
}
while batch.len() < batch_size {
match rx.try_recv() {
Ok(job) => {
if let Some(task) = fetch_task(&server, job).await {
batch.push(task);
match server
.store()
.get_value::<Task>(ValueKey::from(ValueClass::TaskQueue(
TaskQueueClass::Task { id: job.id },
)))
.await
{
Ok(Some(task)) => {
batch.push(TaskDetails { task, info: job });
}
Ok(None) => {
trc::event!(
TaskManager(TaskManagerEvent::TaskIgnored),
Id = job.id,
Reason = "Task not found in store, likely already processed.",
);
}
Err(err) => {
trc::error!(
err.id(job.id)
.details("Failed to retrieve task details.")
.caused_by(trc::location!())
);
}
}
}
Err(_) => break,
}
}
// Dispatch. inbuxa: on a task of its own, so a panic
// releases the batch's locks and leaves this worker
// running; a dead worker would keep claiming tasks it
// can never run
// Dispatch
let mut refresh_queue = false;
let ids = batch.iter().map(|task| task.info.id).collect::<Vec<_>>();
let run = {
let server = server.clone();
tokio::spawn(async move {
let results = server.index(&batch).await;
(batch, results)
})
};
match run.await {
Ok((mut batch, results)) => {
let results = results.into_iter().map(|r| {
refresh_queue |= r.result.is_retry();
r.result
});
update_tasks(&server, &mut batch, results).await;
}
Err(err) => {
worker_failed(&server, &ids, err).await;
refresh_queue = true;
}
}
let results = server.index(&batch).await.into_iter().map(|r| {
refresh_queue |= r.result.is_retry();
r.result
});
update_tasks(&server, &mut batch, results).await;
if refresh_queue || rx.is_empty() {
server.notify_task_queue();
@@ -180,31 +203,83 @@ pub fn spawn_task_manager(inner: Arc<Inner>) {
let server = inner.build_server();
let mut refresh_queue = false;
if let Some(TaskDetails { task, info }) = fetch_task(&server, job).await {
// inbuxa: on a task of its own, as above
let run = {
let server = server.clone();
let server_instance = server_instance.clone();
tokio::spawn(async move {
let result = run_task(&server, &task, server_instance).await;
(task, result)
})
};
match run.await {
Ok((task, result)) => {
refresh_queue = result.is_retry();
match server
.store()
.get_value::<Task>(ValueKey::from(ValueClass::TaskQueue(
TaskQueueClass::Task { id: job.id },
)))
.await
{
Ok(Some(task)) => {
let result = match &task {
Task::CalendarAlarmEmail(task) => {
server.send_email_alarm(task, server_instance.clone()).await
}
Task::CalendarAlarmNotification(task) => {
server.send_display_alarm(task).await
}
Task::CalendarItipMessage(task) => {
server.send_imip(task, server_instance.clone()).await
}
Task::MergeThreads(task) => server.merge_threads(task).await,
Task::DmarcReport(task) => {
server
.submit_report(report::ReportId::Dmarc(task.report_id.id()))
.await
}
Task::TlsReport(task) => {
server
.submit_report(report::ReportId::Tls(task.report_id.id()))
.await
}
Task::RestoreArchivedItem(task) => server.restore_item(task).await,
Task::DestroyAccount(task) => server.destroy_account(task).await,
Task::AccountMaintenance(task) => {
server.account_maintenance(task).await
}
Task::TenantMaintenance(task) => {
server.tenant_maintenance(task).await
}
Task::StoreMaintenance(task) => {
server.store_maintenance(task).await
}
Task::SpamFilterMaintenance(task) => {
Box::pin(server.spam_filter_maintenance(task)).await
}
Task::AcmeRenewal(task) => server.acme_management(task).await,
Task::DkimManagement(task_dkim_rotation) => {
server.dkim_management(task_dkim_rotation).await
}
Task::DnsManagement(task_dns_management) => {
server.dns_management(task_dns_management).await
}
Task::IndexDocument(_)
| Task::UnindexDocument(_)
| Task::IndexTrace(_) => unreachable!(),
};
update_tasks(
&server,
&mut [TaskDetails { task, info }],
vec![result],
)
.await;
}
Err(err) => {
worker_failed(&server, &[info.id], err).await;
refresh_queue = true;
}
refresh_queue = result.is_retry();
update_tasks(
&server,
&mut [TaskDetails { task, info: job }],
vec![result],
)
.await;
}
Ok(None) => {
trc::event!(
TaskManager(TaskManagerEvent::TaskIgnored),
Id = job.id,
Reason = "Task not found in store, likely already processed.",
);
}
Err(err) => {
trc::error!(
err.id(job.id)
.details("Failed to retrieve task details.")
.caused_by(trc::location!())
);
}
}
@@ -243,13 +318,6 @@ pub(crate) trait TaskQueueManager: Sync + Send {
impl TaskQueueManager for Server {
async fn process_tasks(&self, ipc: &mut TaskManagerIpc) -> Duration {
// inbuxa: a node that is stopping has released its locks and claims
// nothing new
let task_locks = &self.inner.ipc.task_locks;
if task_locks.is_stopping() {
return Duration::from_secs(QUEUE_REFRESH_INTERVAL);
}
let lock_expiry = task_locks.expiry();
let now_timestamp = now();
let from_key = ValueKey::<ValueClass> {
account_id: 0,
@@ -325,7 +393,9 @@ impl TaskQueueManager for Server {
let locked = entry.get_mut();
if locked.expires <= now || locked.due < task_due {
locked.expires = Instant::now()
+ std::time::Duration::from_secs(lock_expiry + 1);
+ std::time::Duration::from_secs(
DEFAULT_LOCK_EXPIRY + 1,
);
locked.due = task_due;
tasks.push((
TaskJob {
@@ -341,7 +411,9 @@ impl TaskQueueManager for Server {
Entry::Vacant(entry) => {
entry.insert(Locked {
expires: Instant::now()
+ std::time::Duration::from_secs(lock_expiry + 1),
+ std::time::Duration::from_secs(
DEFAULT_LOCK_EXPIRY + 1,
),
due: task_due,
revision: ipc.revision,
});
@@ -392,26 +464,12 @@ impl TaskQueueManager for Server {
let tx = &ipc.txs[task_type_idx as usize];
if tx.capacity() > 0 {
let id = task_job.id;
if !self.try_lock_task(id).await {
// inbuxa: another node holds the task. Look again after a
// short while rather than a full lock lifetime from now:
// the holder may have claimed it after this scan began,
// or run on a clock ahead of this one, and waiting the
// whole lifetime again would leave the task stuck for
// another hour past its lock if that holder died
if let Some(locked) = ipc.locked.get_mut(&id) {
locked.expires =
Instant::now() + Duration::from_secs(claim_recheck_interval(lock_expiry));
}
} else if tx.send(task_job).await.is_err() {
if self.try_lock_task(task_job.id).await && tx.send(task_job).await.is_err() {
trc::event!(
Server(trc::ServerEvent::ThreadError),
Details = "Error sending task.",
CausedBy = trc::location!()
);
// inbuxa: nothing will run it here, so don't hold it
self.remove_index_lock(id).await;
}
} else {
// If the channel is full, release the lock so it can be picked up in the next iteration
@@ -423,117 +481,9 @@ impl TaskQueueManager for Server {
let now = Instant::now();
ipc.locked
.retain(|_, locked| locked.expires > now && locked.revision == ipc.revision);
let sleep_for = Duration::from_secs(next_event.map_or(QUEUE_REFRESH_INTERVAL, |timestamp| {
Duration::from_secs(next_event.map_or(QUEUE_REFRESH_INTERVAL, |timestamp| {
timestamp.saturating_sub(store::write::now())
}));
// inbuxa: wake up when a claim held elsewhere is due to be tried
// again, rather than only on the next task or refresh
ipc.locked
.values()
.map(|locked| locked.expires.saturating_duration_since(now))
.min()
.map_or(sleep_for, |recheck| sleep_for.min(recheck.max(Duration::from_secs(1))))
}
}
async fn run_task(
server: &Server,
task: &Task,
server_instance: Arc<ServerInstance>,
) -> TaskResult {
match task {
Task::CalendarAlarmEmail(task) => {
server.send_email_alarm(task, server_instance.clone()).await
}
Task::CalendarAlarmNotification(task) => {
server.send_display_alarm(task).await
}
Task::CalendarItipMessage(task) => {
server.send_imip(task, server_instance.clone()).await
}
Task::MergeThreads(task) => server.merge_threads(task).await,
Task::DmarcReport(task) => {
server
.submit_report(report::ReportId::Dmarc(task.report_id.id()))
.await
}
Task::TlsReport(task) => {
server
.submit_report(report::ReportId::Tls(task.report_id.id()))
.await
}
Task::RestoreArchivedItem(task) => server.restore_item(task).await,
Task::DestroyAccount(task) => server.destroy_account(task).await,
Task::AccountMaintenance(task) => {
server.account_maintenance(task).await
}
Task::TenantMaintenance(task) => {
server.tenant_maintenance(task).await
}
Task::StoreMaintenance(task) => {
server.store_maintenance(task).await
}
Task::SpamFilterMaintenance(task) => {
Box::pin(server.spam_filter_maintenance(task)).await
}
Task::AcmeRenewal(task) => server.acme_management(task).await,
Task::DkimManagement(task_dkim_rotation) => {
server.dkim_management(task_dkim_rotation).await
}
Task::DnsManagement(task_dns_management) => {
server.dns_management(task_dns_management).await
}
Task::IndexDocument(_)
| Task::UnindexDocument(_)
| Task::IndexTrace(_) => unreachable!(),
}
}
/// Reads a claimed task. When it is gone or can't be read, the claim is
/// released: inbuxa: holding it would block the task, everywhere, until
/// the lock expired.
async fn fetch_task(server: &Server, job: TaskJob) -> Option<TaskDetails> {
match server
.store()
.get_value::<Task>(ValueKey::from(ValueClass::TaskQueue(TaskQueueClass::Task {
id: job.id,
})))
.await
{
Ok(Some(task)) => Some(TaskDetails { task, info: job }),
Ok(None) => {
trc::event!(
TaskManager(TaskManagerEvent::TaskIgnored),
Id = job.id,
Reason = "Task not found in store, likely already processed.",
);
server.remove_index_lock(job.id).await;
None
}
Err(err) => {
trc::error!(
err.id(job.id)
.details("Failed to retrieve task details.")
.caused_by(trc::location!())
);
server.remove_index_lock(job.id).await;
None
}
}
}
/// inbuxa: a task panicked: its locks are released so it runs again, here or
/// on another node, and the worker carries on.
async fn worker_failed(server: &Server, ids: &[u64], err: tokio::task::JoinError) {
trc::event!(
Server(trc::ServerEvent::ThreadError),
Details = "Task worker failed",
Reason = err.to_string(),
CausedBy = trc::location!()
);
for id in ids {
server.remove_index_lock(*id).await;
}))
}
}
@@ -664,13 +614,6 @@ async fn update_tasks(
}
}
/// inbuxa: how long to wait before trying again to claim a task another node
/// holds: a twelfth of the lock lifetime, so five minutes for the one-hour
/// lock, never more than that and never under a second.
pub(crate) fn claim_recheck_interval(lock_expiry: u64) -> u64 {
(lock_expiry / 12).clamp(1, CLAIM_RECHECK_INTERVAL)
}
pub fn perpetual_retry_time(typ: TaskType, attempt: u64) -> Option<u64> {
matches!(
typ,
+1 -3
View File
@@ -35,9 +35,7 @@ pub mod scheduler;
pub mod spam_classifier;
const QUEUE_REFRESH_INTERVAL: u64 = 60 * 5; // 5 minutes
// inbuxa: the lock lifetime (one hour) lives in common::ipc::TaskLocks, per
// server, so a graceful stop can release the locks and the tests can shorten it
const CLAIM_RECHECK_INTERVAL: u64 = 60 * 5; // 5 minutes
const DEFAULT_LOCK_EXPIRY: u64 = 60 * 60; // 1 hour
pub(crate) struct TaskManagerIpc {
txs: [mpsc::Sender<TaskJob>; TaskType::COUNT],
+2 -7
View File
@@ -10,9 +10,8 @@
// inbuxa: 637 to 641 are the fork's SCIM events (SCIM-54); 642 is
// auth.legacy-protocol-refused (legacy-protocols LP-6); 643 is
// security.legacy-protocols-changed (LP-8); 644 to 646 are the cluster
// coordinator's connection events
pub const TOTAL_EVENT_COUNT: usize = 647;
// security.legacy-protocols-changed (LP-8)
pub const TOTAL_EVENT_COUNT: usize = 644;
pub const TOTAL_METRIC_COUNT: usize = 369;
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
@@ -151,10 +150,6 @@ pub enum ClusterEvent {
MessageSkipped = 47,
MessageInvalid = 49,
NodeIdRenewed = 275,
// inbuxa: the coordinator's connection
CoordinatorConnected = 644,
CoordinatorDisconnected = 645,
CoordinatorError = 646,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
-32
View File
@@ -81,10 +81,6 @@ impl EventType {
b"cluster.message-skipped" => EventType::Cluster(ClusterEvent::MessageSkipped),
b"cluster.message-invalid" => EventType::Cluster(ClusterEvent::MessageInvalid),
b"cluster.node-id-renewed" => EventType::Cluster(ClusterEvent::NodeIdRenewed),
// inbuxa: coordinator connection
b"cluster.coordinator-connected" => EventType::Cluster(ClusterEvent::CoordinatorConnected),
b"cluster.coordinator-disconnected" => EventType::Cluster(ClusterEvent::CoordinatorDisconnected),
b"cluster.coordinator-error" => EventType::Cluster(ClusterEvent::CoordinatorError),
b"dane.authentication-success" => EventType::Dane(DaneEvent::AuthenticationSuccess),
b"dane.authentication-failure" => EventType::Dane(DaneEvent::AuthenticationFailure),
b"dane.no-certificates-found" => EventType::Dane(DaneEvent::NoCertificatesFound),
@@ -746,14 +742,6 @@ impl EventType {
EventType::Cluster(ClusterEvent::MessageSkipped) => "cluster.message-skipped",
EventType::Cluster(ClusterEvent::MessageInvalid) => "cluster.message-invalid",
EventType::Cluster(ClusterEvent::NodeIdRenewed) => "cluster.node-id-renewed",
// inbuxa: coordinator connection
EventType::Cluster(ClusterEvent::CoordinatorConnected) => {
"cluster.coordinator-connected"
}
EventType::Cluster(ClusterEvent::CoordinatorDisconnected) => {
"cluster.coordinator-disconnected"
}
EventType::Cluster(ClusterEvent::CoordinatorError) => "cluster.coordinator-error",
EventType::Dane(DaneEvent::AuthenticationSuccess) => "dane.authentication-success",
EventType::Dane(DaneEvent::AuthenticationFailure) => "dane.authentication-failure",
EventType::Dane(DaneEvent::NoCertificatesFound) => "dane.no-certificates-found",
@@ -1536,10 +1524,6 @@ impl EventType {
EventType::Cluster(ClusterEvent::MessageSkipped) => 47,
EventType::Cluster(ClusterEvent::MessageInvalid) => 49,
EventType::Cluster(ClusterEvent::NodeIdRenewed) => 275,
// inbuxa: coordinator connection
EventType::Cluster(ClusterEvent::CoordinatorConnected) => 644,
EventType::Cluster(ClusterEvent::CoordinatorDisconnected) => 645,
EventType::Cluster(ClusterEvent::CoordinatorError) => 646,
EventType::Dane(DaneEvent::AuthenticationSuccess) => 67,
EventType::Dane(DaneEvent::AuthenticationFailure) => 66,
EventType::Dane(DaneEvent::NoCertificatesFound) => 69,
@@ -2192,10 +2176,6 @@ impl EventType {
47 => Some(EventType::Cluster(ClusterEvent::MessageSkipped)),
49 => Some(EventType::Cluster(ClusterEvent::MessageInvalid)),
275 => Some(EventType::Cluster(ClusterEvent::NodeIdRenewed)),
// inbuxa: coordinator connection
644 => Some(EventType::Cluster(ClusterEvent::CoordinatorConnected)),
645 => Some(EventType::Cluster(ClusterEvent::CoordinatorDisconnected)),
646 => Some(EventType::Cluster(ClusterEvent::CoordinatorError)),
67 => Some(EventType::Dane(DaneEvent::AuthenticationSuccess)),
66 => Some(EventType::Dane(DaneEvent::AuthenticationFailure)),
69 => Some(EventType::Dane(DaneEvent::NoCertificatesFound)),
@@ -3134,10 +3114,6 @@ impl EventType {
EventType::Auth(AuthEvent::TooManyAttempts) => Level::Warn,
EventType::Calendar(CalendarEvent::AlarmFailed) => Level::Warn,
EventType::Cluster(ClusterEvent::SubscriberDisconnected) => Level::Warn,
// inbuxa: coordinator connection
EventType::Cluster(ClusterEvent::CoordinatorConnected) => Level::Info,
EventType::Cluster(ClusterEvent::CoordinatorDisconnected) => Level::Warn,
EventType::Cluster(ClusterEvent::CoordinatorError) => Level::Warn,
EventType::Delivery(DeliveryEvent::MissingOutboundHostname) => Level::Warn,
EventType::Delivery(DeliveryEvent::ConcurrencyLimitExceeded) => Level::Warn,
EventType::Delivery(DeliveryEvent::RateLimitExceeded) => Level::Warn,
@@ -3268,10 +3244,6 @@ impl EventType {
EventType::Cluster(ClusterEvent::MessageSkipped) => "PubSub message skipped",
EventType::Cluster(ClusterEvent::MessageInvalid) => "Invalid PubSub message",
EventType::Cluster(ClusterEvent::NodeIdRenewed) => "Node ID renewed",
// inbuxa: coordinator connection
EventType::Cluster(ClusterEvent::CoordinatorConnected) => "Coordinator connected",
EventType::Cluster(ClusterEvent::CoordinatorDisconnected) => "Coordinator unavailable",
EventType::Cluster(ClusterEvent::CoordinatorError) => "Coordinator error",
EventType::Dane(DaneEvent::AuthenticationSuccess) => "DANE authentication successful",
EventType::Dane(DaneEvent::AuthenticationFailure) => "DANE authentication failed",
EventType::Dane(DaneEvent::NoCertificatesFound) => "No certificates found for DANE",
@@ -4350,10 +4322,6 @@ impl EventType {
EventType::Cluster(ClusterEvent::MessageSkipped),
EventType::Cluster(ClusterEvent::MessageInvalid),
EventType::Cluster(ClusterEvent::NodeIdRenewed),
// inbuxa: coordinator connection
EventType::Cluster(ClusterEvent::CoordinatorConnected),
EventType::Cluster(ClusterEvent::CoordinatorDisconnected),
EventType::Cluster(ClusterEvent::CoordinatorError),
EventType::Dane(DaneEvent::AuthenticationSuccess),
EventType::Dane(DaneEvent::AuthenticationFailure),
EventType::Dane(DaneEvent::NoCertificatesFound),
Binary file not shown.
+1 -1
View File
@@ -1 +1 @@
XFI3xuKC_rH1KZyaVBF0uTIiRDXRqyYboijquiGz2eg
VbnFuwCOTBh0s2T-NuRhb2JaJr8Jl5s3LgXv4Pv2sTg
-145
View File
@@ -1,145 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! A node that starts while its NATS coordinator is down joins the cluster
//! once NATS comes up, without a restart, and reports the coordinator's
//! connection on `/healthz/cluster` as it goes and comes back.
use crate::utils::server::TestServerBuilder;
use coordinator::Coordinator;
use registry::{
schema::{
enums::NetworkListenerProtocol,
structs::{Coordinator as CoordinatorSetting, NatsCoordinator},
},
types::map::Map,
};
use serde_json::{Value, json};
use std::time::{Duration, Instant};
use testcontainers::{
GenericImage, ImageExt, core::IntoContainerPort, core::WaitFor, runners::AsyncRunner,
};
const HTTP_PORT: u16 = 11_310;
const TOPIC: &str = "inbuxa-coordinator-test";
#[tokio::test(flavor = "multi_thread")]
pub async fn coordinator_reconnect_tests() {
println!("Running coordinator reconnect tests...");
// A port with no NATS server behind it, yet
let nats_port = std::net::TcpListener::bind("127.0.0.1:0")
.unwrap()
.local_addr()
.unwrap()
.port();
let config = NatsCoordinator {
addresses: Map::new(vec![format!("127.0.0.1:{nats_port}")]),
use_tls: false,
timeout_connection: 1_000u64.into(),
..Default::default()
};
// 1. The node starts, without a build error, while NATS is down, and
// says so
let test = TestServerBuilder::new("coordinator_reconnect_tests")
.await
.with_object(CoordinatorSetting::Nats(config.clone()))
.await
.with_listener(NetworkListenerProtocol::Http, "http", HTTP_PORT, true)
.await
.build()
.await;
let coordinator = test.server.core.storage.coordinator.clone();
assert!(
coordinator.is_enabled(),
"a coordinator, though not connected"
);
assert_eq!(coordinator.is_connected(), Some(false));
assert_eq!(
cluster_health().await,
(503, json!({"coordinator": "disconnected"}))
);
// A subscription made now, as the broadcast subscriber makes it at
// startup, has to work once NATS is up
let mut stream = coordinator.subscribe(TOPIC).await.unwrap();
// 2. NATS comes up: the node connects on its own
let nats = GenericImage::new("nats", "latest")
.with_wait_for(WaitFor::message_on_stderr("Server is ready"))
.with_mapped_port(nats_port, 4222.tcp())
.start()
.await
.expect("Failed to start NATS container");
wait_for_health(200, "connected").await;
let other_node = coordinator::backend::nats::NatsPubSub::open(config.clone())
.await
.unwrap();
wait_until_connected(&other_node).await;
round_trip(&other_node, &mut stream, b"after startup").await;
// 3. NATS goes away: the node reports it; and it comes back: the node
// reconnects and the same subscription carries on
nats.stop().await.unwrap();
wait_for_health(503, "disconnected").await;
nats.start().await.unwrap();
wait_for_health(200, "connected").await;
wait_until_connected(&other_node).await;
round_trip(&other_node, &mut stream, b"after reconnect").await;
drop(nats);
if test.is_reset() {
test.temp_dir.delete();
}
}
async fn cluster_health() -> (u16, Value) {
let response = reqwest::Client::builder()
.danger_accept_invalid_certs(true)
.timeout(Duration::from_secs(5))
.build()
.unwrap()
.get(format!("https://127.0.0.1:{HTTP_PORT}/healthz/cluster"))
.send()
.await
.unwrap();
let status = response.status().as_u16();
(status, response.json().await.unwrap())
}
async fn wait_for_health(status: u16, state: &str) {
let started = Instant::now();
loop {
let health = cluster_health().await;
if health == (status, json!({"coordinator": state})) {
return;
}
assert!(
started.elapsed() < Duration::from_secs(30),
"expected {status} {state}, still {health:?}"
);
tokio::time::sleep(Duration::from_millis(250)).await;
}
}
async fn wait_until_connected(coordinator: &Coordinator) {
let started = Instant::now();
while coordinator.is_connected() != Some(true) {
assert!(started.elapsed() < Duration::from_secs(30), "not connected");
tokio::time::sleep(Duration::from_millis(100)).await;
}
}
/// Another node publishes; this one's subscription receives it.
async fn round_trip(from: &Coordinator, stream: &mut coordinator::PubSubStream, payload: &[u8]) {
from.publish(TOPIC, payload.to_vec()).await.unwrap();
let message = tokio::time::timeout(Duration::from_secs(10), stream.next())
.await
.expect("no message within 10 seconds")
.expect("subscription ended");
assert_eq!(message.payload(), payload);
}
-4
View File
@@ -2,11 +2,7 @@
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <[email protected]>
*
* SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL
*
* Modified by Coffey Labs in 2026 for INBUXA.
*/
pub mod broadcast;
#[cfg(feature = "nats")]
pub mod coordinator; // inbuxa: coordinator reconnects
pub mod stress;
-1
View File
@@ -21,7 +21,6 @@ pub mod replica_cluster; // inbuxa: read replicas across nodes
pub mod scaleout; // inbuxa: scale-out storage
#[cfg(any(feature = "postgres", feature = "mysql"))]
pub mod sql_timeout;
pub mod task_locks; // inbuxa: task locks across nodes
use crate::utils::server::TestServerBuilder;
use std::io::Read;
-184
View File
@@ -1,184 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Task locks across nodes: tasks claimed by a node that then disappears
//! run elsewhere once its locks expire, and a node that stops gracefully
//! hands its locks back at once. The other node is played by writing its
//! locks straight into the shared in-memory store, as a node that claimed
//! the tasks and died leaves them.
use crate::utils::server::TestServerBuilder;
use common::{KV_LOCK_TASK, Server};
use registry::schema::{
enums::IndexDocumentType,
structs::{Task, TaskIndexDocument, TaskStatus},
};
use services::task_manager::lock::{TaskLockManager, release_task_locks};
use std::time::{Duration, Instant};
use store::{
ValueKey,
write::{BatchBuilder, TaskQueueClass, ValueClass},
};
use utils::snowflake::SnowflakeIdGenerator;
// Short enough for a test, long enough that the recheck interval (a twelfth
// of it) is well below it
const LOCK_EXPIRY: u64 = 12;
#[tokio::test(flavor = "multi_thread")]
pub async fn task_lock_tests() {
let test = TestServerBuilder::new("task_lock_tests")
.await
.build()
.await;
let server = test.server.clone();
println!(
"Running task lock tests on {}...",
std::env::var("STORE").unwrap_or_default()
);
server.inner.ipc.task_locks.set_expiry(LOCK_EXPIRY);
// 1. Another node claimed the tasks and died. Its locks block them until
// they expire; then this node runs them, without waiting for anything
// else to wake it
let ids = new_task_ids(4);
for id in &ids {
assert!(foreign_lock(&server, *id, LOCK_EXPIRY).await);
}
schedule(&server, &ids).await;
let started = Instant::now();
server.notify_task_queue();
tokio::time::sleep(Duration::from_secs(3)).await;
assert_eq!(pending(&server, &ids).await, ids.len(), "held by the other node");
wait_until_done(&server, &ids, Duration::from_secs(LOCK_EXPIRY + 10)).await;
let elapsed = started.elapsed();
assert!(
elapsed >= Duration::from_secs(LOCK_EXPIRY - 2),
"ran before the other node's locks expired: {elapsed:?}"
);
// 2. The other node's locks outlive what this node expects: claimed just
// after this node looked, or by a node whose clock runs ahead. This node
// keeps checking at the recheck interval, so the tasks run soon after
// those locks expire, not a whole lock lifetime later
let held_for = LOCK_EXPIRY + LOCK_EXPIRY / 2;
let ids = new_task_ids(4);
for id in &ids {
assert!(foreign_lock(&server, *id, held_for).await);
}
schedule(&server, &ids).await;
let started = Instant::now();
server.notify_task_queue();
wait_until_done(&server, &ids, Duration::from_secs(held_for + 8)).await;
let elapsed = started.elapsed();
assert!(
elapsed >= Duration::from_secs(held_for - 2),
"ran before the other node's locks expired: {elapsed:?}"
);
// 3. A graceful stop releases the locks this node holds: another node
// can claim those tasks at once, and this one claims nothing more
let ids = new_task_ids(3);
for id in &ids {
assert!(server.try_lock_task(*id).await, "claim {id}");
}
assert_eq!(server.inner.ipc.task_locks.held(), ids.len());
for id in &ids {
assert!(
!foreign_lock(&server, *id, LOCK_EXPIRY).await,
"held while this node runs"
);
}
assert_eq!(release_task_locks(&server).await, ids.len());
assert_eq!(server.inner.ipc.task_locks.held(), 0);
for id in &ids {
assert!(
foreign_lock(&server, *id, LOCK_EXPIRY).await,
"released on stop: {id}"
);
}
let [id] = new_task_ids(1)[..] else {
unreachable!()
};
assert!(!server.try_lock_task(id).await, "a stopping node claims nothing");
for id in ids {
let _ = server
.in_memory_store()
.remove_lock(KV_LOCK_TASK, &id.to_be_bytes())
.await;
}
if test.is_reset() {
test.temp_dir.delete();
}
}
fn new_task_ids(count: usize) -> Vec<u64> {
(0..count)
.map(|_| SnowflakeIdGenerator::global_id().unwrap())
.collect()
}
/// The other node's claim on a task, as its task manager takes it.
async fn foreign_lock(server: &Server, id: u64, seconds: u64) -> bool {
server
.in_memory_store()
.try_lock(KV_LOCK_TASK, &id.to_be_bytes(), seconds)
.await
.unwrap()
}
/// Unindex tasks for files that don't exist: files aren't search-indexed and
/// there is no undelete note, so running one only drops it from the queue.
async fn schedule(server: &Server, ids: &[u64]) {
let mut batch = BatchBuilder::new();
for (n, id) in ids.iter().enumerate() {
batch.schedule_task_with_id(
*id,
Task::UnindexDocument(TaskIndexDocument {
account_id: 0u32.into(),
document_id: (u32::MAX - n as u32).into(),
document_type: IndexDocumentType::File,
status: TaskStatus::now(),
}),
);
}
server.store().write(batch.build_all()).await.unwrap();
}
async fn pending(server: &Server, ids: &[u64]) -> usize {
let mut count = 0;
for id in ids {
if server
.store()
.get_value::<Task>(ValueKey::from(ValueClass::TaskQueue(
TaskQueueClass::Task { id: *id },
)))
.await
.unwrap()
.is_some()
{
count += 1;
}
}
count
}
async fn wait_until_done(server: &Server, ids: &[u64], within: Duration) {
let started = Instant::now();
loop {
let left = pending(server, ids).await;
if left == 0 {
return;
}
assert!(
started.elapsed() < within,
"{left} task(s) still pending after {:?}",
started.elapsed()
);
tokio::time::sleep(Duration::from_millis(250)).await;
}
}