Compare commits
5
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f398d95062 | ||
|
|
81ec917d74 | ||
|
|
e2ab26ad19 | ||
|
|
0b4aa9c084 | ||
|
|
4e6c8b916e |
@@ -14,8 +14,11 @@
|
|||||||
# crates/types/src/branding.rs, not Cargo.toml, and the image is tagged
|
# crates/types/src/branding.rs, not Cargo.toml, and the image is tagged
|
||||||
# with it, so a tag beside an unbumped macro would publish an image that
|
# with it, so a tag beside an unbumped macro would publish an image that
|
||||||
# reports a different version from its tag.
|
# reports a different version from its tag.
|
||||||
# * the tag must be on main, so an image never describes code that was never
|
# * the tag must be on main or on a release/* branch, so an image never
|
||||||
# reviewed onto the default branch.
|
# describes code that was never reviewed onto one of them. A release/*
|
||||||
|
# branch carries a hotfix: it starts at an earlier release tag, takes
|
||||||
|
# fixes through pull requests into it, and is tagged there, so production
|
||||||
|
# can get a fix without everything that has landed on main since.
|
||||||
#
|
#
|
||||||
# :latest moves with every published tag: tags are cut by the weekly release
|
# :latest moves with every published tag: tags are cut by the weekly release
|
||||||
# (or by hand for a real release); there are no prerelease tags here.
|
# (or by hand for a real release); there are no prerelease tags here.
|
||||||
@@ -57,8 +60,13 @@ jobs:
|
|||||||
echo "Refusing to publish an image that would report the wrong version." >&2
|
echo "Refusing to publish an image that would report the wrong version." >&2
|
||||||
exit 1
|
exit 1
|
||||||
fi
|
fi
|
||||||
git merge-base --is-ancestor "$(git rev-parse "${TAG}^{commit}")" origin/main \
|
commit="$(git rev-parse "${TAG}^{commit}")"
|
||||||
|| { echo "$TAG is not on main" >&2; exit 1; }
|
on=""
|
||||||
|
for ref in origin/main $(git for-each-ref --format='%(refname:short)' 'refs/remotes/origin/release/*'); do
|
||||||
|
if git merge-base --is-ancestor "$commit" "$ref"; then on="$ref"; break; fi
|
||||||
|
done
|
||||||
|
[ -n "$on" ] || { echo "$TAG is not on main or a release/* branch" >&2; exit 1; }
|
||||||
|
echo "$TAG is on $on"
|
||||||
echo "version=$V" >> "$GITHUB_OUTPUT"
|
echo "version=$V" >> "$GITHUB_OUTPUT"
|
||||||
echo "version $V"
|
echo "version $V"
|
||||||
|
|
||||||
|
|||||||
@@ -335,57 +335,3 @@ impl EmailPush {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// inbuxa: the task locks this node holds, so a graceful stop can hand them
|
|
||||||
/// back instead of leaving the tasks blocked until the locks expire.
|
|
||||||
pub struct TaskLocks {
|
|
||||||
held: parking_lot::Mutex<ahash::AHashSet<u64>>,
|
|
||||||
stopping: AtomicBool,
|
|
||||||
expiry: std::sync::atomic::AtomicU64,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl TaskLocks {
|
|
||||||
/// How long a task lock lasts, in seconds, unless it is released first.
|
|
||||||
pub const DEFAULT_EXPIRY: u64 = 60 * 60;
|
|
||||||
|
|
||||||
pub fn is_stopping(&self) -> bool {
|
|
||||||
self.stopping.load(Ordering::Acquire)
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Stops new claims and returns the ids of every lock still held.
|
|
||||||
pub fn stop(&self) -> Vec<u64> {
|
|
||||||
self.stopping.store(true, Ordering::Release);
|
|
||||||
self.held.lock().drain().collect()
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn insert(&self, id: u64) {
|
|
||||||
self.held.lock().insert(id);
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn remove(&self, id: u64) {
|
|
||||||
self.held.lock().remove(&id);
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn held(&self) -> usize {
|
|
||||||
self.held.lock().len()
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn expiry(&self) -> u64 {
|
|
||||||
self.expiry.load(Ordering::Relaxed)
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Changes the lock lifetime; the tests shorten it.
|
|
||||||
pub fn set_expiry(&self, seconds: u64) {
|
|
||||||
self.expiry.store(seconds.max(1), Ordering::Relaxed);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
impl Default for TaskLocks {
|
|
||||||
fn default() -> Self {
|
|
||||||
Self {
|
|
||||||
held: Default::default(),
|
|
||||||
stopping: AtomicBool::new(false),
|
|
||||||
expiry: std::sync::atomic::AtomicU64::new(Self::DEFAULT_EXPIRY),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -279,8 +279,6 @@ pub struct HttpAuthCache {
|
|||||||
pub struct Ipc {
|
pub struct Ipc {
|
||||||
pub push_tx: mpsc::Sender<PushEvent>,
|
pub push_tx: mpsc::Sender<PushEvent>,
|
||||||
pub task_tx: Arc<Notify>,
|
pub task_tx: Arc<Notify>,
|
||||||
// inbuxa: task locks held by this node, released on a graceful stop
|
|
||||||
pub task_locks: Arc<crate::ipc::TaskLocks>,
|
|
||||||
pub queue_tx: mpsc::Sender<QueueEvent>,
|
pub queue_tx: mpsc::Sender<QueueEvent>,
|
||||||
pub report_tx: mpsc::Sender<ReportingEvent>,
|
pub report_tx: mpsc::Sender<ReportingEvent>,
|
||||||
pub broadcast_tx: Option<mpsc::Sender<BroadcastEvent>>,
|
pub broadcast_tx: Option<mpsc::Sender<BroadcastEvent>>,
|
||||||
|
|||||||
@@ -23,13 +23,6 @@ use utils::{UnwrapFailure, codec::leb128::Leb128_};
|
|||||||
|
|
||||||
pub(super) const MAGIC_MARKER: u8 = 123;
|
pub(super) const MAGIC_MARKER: u8 = 123;
|
||||||
|
|
||||||
// inbuxa: blobs kept under a fixed name instead of a content hash. Nothing
|
|
||||||
// links to them, so the export names them outright.
|
|
||||||
const NAMED_BLOBS: &[&[u8]] = &[
|
|
||||||
crate::manager::SPAM_CLASSIFIER_KEY,
|
|
||||||
crate::manager::SPAM_TRAINER_KEY,
|
|
||||||
];
|
|
||||||
|
|
||||||
#[derive(Debug, Clone, Copy, Hash, PartialEq, Eq)]
|
#[derive(Debug, Clone, Copy, Hash, PartialEq, Eq)]
|
||||||
pub(super) enum Family {
|
pub(super) enum Family {
|
||||||
Data = 0,
|
Data = 0,
|
||||||
@@ -150,21 +143,15 @@ impl Core {
|
|||||||
.await
|
.await
|
||||||
.failed("Failed to iterate over data store");
|
.failed("Failed to iterate over data store");
|
||||||
|
|
||||||
// inbuxa: the trained spam classifier and its trainer state are
|
for hash in blobs {
|
||||||
// blobs stored under fixed names with no blob link, so the walk
|
|
||||||
// over links above never reaches them.
|
|
||||||
let named = NAMED_BLOBS.iter().map(|key| key.to_vec());
|
|
||||||
for key in blobs
|
|
||||||
.into_iter()
|
|
||||||
.map(|hash| hash.as_slice().to_vec())
|
|
||||||
.chain(named)
|
|
||||||
{
|
|
||||||
if let Some(blob) = blob_store
|
if let Some(blob) = blob_store
|
||||||
.get_blob(&key, 0..usize::MAX)
|
.get_blob(hash.as_slice(), 0..usize::MAX)
|
||||||
.await
|
.await
|
||||||
.failed("Failed to get blob")
|
.failed("Failed to get blob")
|
||||||
{
|
{
|
||||||
writer.send((key, blob)).failed("Failed to send key");
|
writer
|
||||||
|
.send((hash.as_slice().to_vec(), blob))
|
||||||
|
.failed("Failed to send key");
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}),
|
}),
|
||||||
@@ -336,13 +323,7 @@ impl Family {
|
|||||||
SUBSPACE_REGISTRY_IDX,
|
SUBSPACE_REGISTRY_IDX,
|
||||||
SUBSPACE_REGISTRY_PK,
|
SUBSPACE_REGISTRY_PK,
|
||||||
SUBSPACE_DIRECTORY,
|
SUBSPACE_DIRECTORY,
|
||||||
// inbuxa: registry objects the upstream list left out, so an
|
store::SUBSPACE_INBUXA, // inbuxa: masked email
|
||||||
// export dropped them: archived items (undelete) and spam
|
|
||||||
// training samples. Their indexes and id counters already
|
|
||||||
// travel in this family and in `data`, so they ride along.
|
|
||||||
SUBSPACE_DELETED_ITEMS,
|
|
||||||
SUBSPACE_SPAM_SAMPLES,
|
|
||||||
store::SUBSPACE_INBUXA, // inbuxa: the fork's own data (masked email, undelete, policies)
|
|
||||||
],
|
],
|
||||||
Family::Changelog => &[SUBSPACE_LOGS],
|
Family::Changelog => &[SUBSPACE_LOGS],
|
||||||
Family::Queue => &[SUBSPACE_QUEUE_MESSAGE, SUBSPACE_QUEUE_EVENT],
|
Family::Queue => &[SUBSPACE_QUEUE_MESSAGE, SUBSPACE_QUEUE_EVENT],
|
||||||
|
|||||||
@@ -54,13 +54,6 @@ Options:
|
|||||||
-o, --console Open the store console
|
-o, --console Open the store console
|
||||||
-h, --help Print help
|
-h, --help Print help
|
||||||
-V, --version Print version
|
-V, --version Print version
|
||||||
|
|
||||||
An export holds everything in the data and blob stores except short-lived
|
|
||||||
in-memory state (rate limits, locks, greylisting) and the full-text search
|
|
||||||
index, which belongs to one search backend. An import into an empty store
|
|
||||||
queues the index to be rebuilt when the server next starts. EXPORT_TYPES
|
|
||||||
limits an export to some of: data, registry, blob, changelog, queue, report,
|
|
||||||
telemetry, tasks.
|
|
||||||
"#
|
"#
|
||||||
);
|
);
|
||||||
|
|
||||||
@@ -263,10 +256,10 @@ impl BootManager {
|
|||||||
telemetry.enable();
|
telemetry.enable();
|
||||||
|
|
||||||
// Parse settings and restore
|
// Parse settings and restore
|
||||||
let core = Box::pin(Core::parse(&mut bootstrap, storage)).await;
|
Box::pin(Core::parse(&mut bootstrap, storage))
|
||||||
let imported = core.restore(path).await;
|
.await
|
||||||
// inbuxa: the search index isn't exported; rebuild it
|
.restore(path)
|
||||||
core.queue_reindex(&imported).await;
|
.await;
|
||||||
std::process::exit(0);
|
std::process::exit(0);
|
||||||
}
|
}
|
||||||
StoreOp::Console => {
|
StoreOp::Console => {
|
||||||
@@ -297,7 +290,6 @@ pub fn build_ipc(has_pubsub: bool) -> (Ipc, IpcReceivers) {
|
|||||||
report_tx,
|
report_tx,
|
||||||
broadcast_tx: has_pubsub.then_some(broadcast_tx),
|
broadcast_tx: has_pubsub.then_some(broadcast_tx),
|
||||||
task_tx: Arc::new(Notify::new()),
|
task_tx: Arc::new(Notify::new()),
|
||||||
task_locks: Arc::new(crate::ipc::TaskLocks::default()),
|
|
||||||
train_task_controller: Arc::new(TrainTaskController::default()),
|
train_task_controller: Arc::new(TrainTaskController::default()),
|
||||||
},
|
},
|
||||||
IpcReceivers {
|
IpcReceivers {
|
||||||
|
|||||||
@@ -9,22 +9,15 @@
|
|||||||
use super::backup::MAGIC_MARKER;
|
use super::backup::MAGIC_MARKER;
|
||||||
use crate::{Core, DATABASE_SCHEMA_VERSION};
|
use crate::{Core, DATABASE_SCHEMA_VERSION};
|
||||||
use lz4_flex::frame::FrameDecoder;
|
use lz4_flex::frame::FrameDecoder;
|
||||||
use registry::{
|
use registry::schema::enums::CompressionAlgo;
|
||||||
schema::{
|
|
||||||
enums::{CompressionAlgo, TaskStoreMaintenanceType},
|
|
||||||
structs::{Task, TaskStatus, TaskStoreMaintenance},
|
|
||||||
},
|
|
||||||
types::EnumImpl,
|
|
||||||
};
|
|
||||||
use std::{
|
use std::{
|
||||||
fs::File,
|
fs::File,
|
||||||
io::{BufReader, ErrorKind, Read},
|
io::{BufReader, ErrorKind, Read},
|
||||||
path::{Path, PathBuf},
|
path::{Path, PathBuf},
|
||||||
};
|
};
|
||||||
use store::{
|
use store::{
|
||||||
BlobStore, IterateParams, SUBSPACE_BLOBS, SUBSPACE_COUNTER, SUBSPACE_INDEXES,
|
BlobStore, IterateParams, SUBSPACE_BLOBS, SUBSPACE_COUNTER, SUBSPACE_INDEXES, SUBSPACE_QUOTA,
|
||||||
SUBSPACE_PROPERTY, SUBSPACE_QUOTA, SUBSPACE_REGISTRY_PK, SUBSPACE_TELEMETRY_SPAN, Store,
|
SUBSPACE_REGISTRY_PK, Store, U32_LEN,
|
||||||
U32_LEN,
|
|
||||||
write::{
|
write::{
|
||||||
AnyClass, AnyKey, BatchBuilder, ValueClass,
|
AnyClass, AnyKey, BatchBuilder, ValueClass,
|
||||||
key::{DeserializeBigEndian, is_node_id_key},
|
key::{DeserializeBigEndian, is_node_id_key},
|
||||||
@@ -34,9 +27,7 @@ use types::{collection::Collection, field::Field};
|
|||||||
use utils::{UnwrapFailure, failed};
|
use utils::{UnwrapFailure, failed};
|
||||||
|
|
||||||
impl Core {
|
impl Core {
|
||||||
/// Imports an export into an empty store and returns the subspaces it
|
pub async fn restore(&self, src: PathBuf) {
|
||||||
/// wrote. inbuxa: the caller hands them to [`Core::queue_reindex`].
|
|
||||||
pub async fn restore(&self, src: PathBuf) -> Vec<u8> {
|
|
||||||
// Backup the core
|
// Backup the core
|
||||||
let paths = if src.is_dir() {
|
let paths = if src.is_dir() {
|
||||||
let mut paths = Vec::new();
|
let mut paths = Vec::new();
|
||||||
@@ -73,13 +64,6 @@ impl Core {
|
|||||||
std::process::exit(1);
|
std::process::exit(1);
|
||||||
}
|
}
|
||||||
|
|
||||||
let mut imported = paths
|
|
||||||
.iter()
|
|
||||||
.map(|path| KeyValueReader::new(path).subspace)
|
|
||||||
.collect::<Vec<_>>();
|
|
||||||
imported.sort_unstable();
|
|
||||||
imported.dedup();
|
|
||||||
|
|
||||||
let mut tasks = Vec::new();
|
let mut tasks = Vec::new();
|
||||||
for path in paths {
|
for path in paths {
|
||||||
let storage = self.storage.clone();
|
let storage = self.storage.clone();
|
||||||
@@ -92,54 +76,6 @@ impl Core {
|
|||||||
for task in tasks {
|
for task in tasks {
|
||||||
task.await.failed("Failed to wait for task");
|
task.await.failed("Failed to wait for task");
|
||||||
}
|
}
|
||||||
|
|
||||||
imported
|
|
||||||
}
|
|
||||||
|
|
||||||
/// inbuxa: an export never carries the full-text index. It is built by
|
|
||||||
/// and for one search backend (the SQL stores index into their own
|
|
||||||
/// tables, the key-value stores into a subspace, external engines keep it
|
|
||||||
/// themselves), so it would be wrong or unreadable after a move to
|
|
||||||
/// another one. Instead, an import queues the same reindex tasks an
|
|
||||||
/// administrator can queue by hand (`reindexAccounts` and
|
|
||||||
/// `reindexTelemetry` store maintenance), and the server rebuilds the
|
|
||||||
/// index for whatever search store it is configured with once it starts.
|
|
||||||
pub async fn queue_reindex(&self, imported: &[u8]) -> Vec<TaskStoreMaintenanceType> {
|
|
||||||
let mut queued = Vec::new();
|
|
||||||
if imported.contains(&SUBSPACE_PROPERTY) {
|
|
||||||
queued.push(TaskStoreMaintenanceType::ReindexAccounts);
|
|
||||||
}
|
|
||||||
if imported.contains(&SUBSPACE_TELEMETRY_SPAN) {
|
|
||||||
queued.push(TaskStoreMaintenanceType::ReindexTelemetry);
|
|
||||||
}
|
|
||||||
if queued.is_empty() {
|
|
||||||
return queued;
|
|
||||||
}
|
|
||||||
|
|
||||||
let mut batch = BatchBuilder::new();
|
|
||||||
for maintenance_type in &queued {
|
|
||||||
batch.schedule_task(Task::StoreMaintenance(TaskStoreMaintenance {
|
|
||||||
maintenance_type: *maintenance_type,
|
|
||||||
status: TaskStatus::now(),
|
|
||||||
shard_index: None,
|
|
||||||
}));
|
|
||||||
}
|
|
||||||
self.storage
|
|
||||||
.data
|
|
||||||
.write(batch.build_all())
|
|
||||||
.await
|
|
||||||
.failed("Failed to queue the reindex tasks");
|
|
||||||
|
|
||||||
println!(
|
|
||||||
"Queued {} to rebuild the search index; it runs when the server starts.",
|
|
||||||
queued
|
|
||||||
.iter()
|
|
||||||
.map(|t| t.as_str())
|
|
||||||
.collect::<Vec<_>>()
|
|
||||||
.join(" and ")
|
|
||||||
);
|
|
||||||
|
|
||||||
queued
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -189,22 +125,17 @@ async fn restore_file(store: Store, blob_store: BlobStore, path: &Path) {
|
|||||||
}
|
}
|
||||||
SUBSPACE_COUNTER | SUBSPACE_QUOTA => {
|
SUBSPACE_COUNTER | SUBSPACE_QUOTA => {
|
||||||
while let Some((key, value)) = reader.next() {
|
while let Some((key, value)) = reader.next() {
|
||||||
let class = ValueClass::Any(AnyClass {
|
batch.add(
|
||||||
subspace: reader.subspace,
|
ValueClass::Any(AnyClass {
|
||||||
key,
|
subspace: reader.subspace,
|
||||||
});
|
key,
|
||||||
let value = u64::from_le_bytes(
|
}),
|
||||||
value
|
u64::from_le_bytes(
|
||||||
.try_into()
|
value
|
||||||
.expect("Failed to deserialize counter/quota"),
|
.try_into()
|
||||||
) as i64;
|
.expect("Failed to deserialize counter/quota"),
|
||||||
// inbuxa: the SQL stores add a negative amount with an UPDATE,
|
) as i64,
|
||||||
// which does nothing to a row that isn't there yet, so a
|
);
|
||||||
// negative counter vanished on import. Create the row first.
|
|
||||||
if value < 0 {
|
|
||||||
batch.add(class.clone(), 0);
|
|
||||||
}
|
|
||||||
batch.add(class, value);
|
|
||||||
if batch.is_large_batch() {
|
if batch.is_large_batch() {
|
||||||
store
|
store
|
||||||
.write(batch.build_all())
|
.write(batch.build_all())
|
||||||
|
|||||||
@@ -8,7 +8,7 @@ store = { path = "../store" }
|
|||||||
registry = { path = "../registry" }
|
registry = { path = "../registry" }
|
||||||
trc = { path = "../trc" }
|
trc = { path = "../trc" }
|
||||||
futures = { version = "0.3", optional = true }
|
futures = { version = "0.3", optional = true }
|
||||||
tokio = { version = "1.53", features = ["sync", "fs", "io-util", "rt", "time"] }
|
tokio = { version = "1.53", features = ["sync", "fs", "io-util"] }
|
||||||
async-nats = { version = "0.50", default-features = false, features = ["server_2_10", "server_2_11", "aws-lc-rs"], optional = true }
|
async-nats = { version = "0.50", default-features = false, features = ["server_2_10", "server_2_11", "aws-lc-rs"], optional = true }
|
||||||
zenoh = { version = "1.10.0", default-features = false, features = ["auth_pubkey", "transport_multilink", "transport_compression", "transport_quic", "transport_tcp", "transport_tls", "transport_udp"], optional = true }
|
zenoh = { version = "1.10.0", default-features = false, features = ["auth_pubkey", "transport_multilink", "transport_compression", "transport_quic", "transport_tcp", "transport_tls", "transport_udp"], optional = true }
|
||||||
rdkafka = { version = "0.39", features = ["cmake-build"], optional = true }
|
rdkafka = { version = "0.39", features = ["cmake-build"], optional = true }
|
||||||
|
|||||||
@@ -2,22 +2,13 @@
|
|||||||
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <[email protected]>
|
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <[email protected]>
|
||||||
*
|
*
|
||||||
* SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL
|
* SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL
|
||||||
*
|
|
||||||
* Modified by Coffey Labs in 2026 for INBUXA.
|
|
||||||
*/
|
*/
|
||||||
|
|
||||||
use std::{
|
use std::sync::Arc;
|
||||||
sync::{
|
|
||||||
Arc,
|
|
||||||
atomic::{AtomicBool, Ordering},
|
|
||||||
},
|
|
||||||
time::Duration,
|
|
||||||
};
|
|
||||||
|
|
||||||
use crate::Coordinator;
|
use crate::Coordinator;
|
||||||
use async_nats::Client;
|
use async_nats::Client;
|
||||||
use registry::schema::structs::NatsCoordinator;
|
use registry::schema::structs::NatsCoordinator;
|
||||||
use trc::ClusterEvent;
|
|
||||||
|
|
||||||
pub mod pubsub;
|
pub mod pubsub;
|
||||||
|
|
||||||
@@ -56,116 +47,9 @@ impl NatsPubSub {
|
|||||||
opts = opts.token(credentials);
|
opts = opts.token(credentials);
|
||||||
}
|
}
|
||||||
|
|
||||||
// inbuxa: connect in the background and keep trying, so a node that
|
|
||||||
// starts while NATS is down still joins the cluster once NATS is
|
|
||||||
// back, instead of running without a coordinator until restarted;
|
|
||||||
// and report the connection going and coming back
|
|
||||||
let reporter = Arc::new(Reporter::default());
|
|
||||||
opts = opts.retry_on_initial_connect().event_callback({
|
|
||||||
let reporter = reporter.clone();
|
|
||||||
move |event| {
|
|
||||||
let reporter = reporter.clone();
|
|
||||||
async move { reporter.report(event) }
|
|
||||||
}
|
|
||||||
});
|
|
||||||
let connection_timeout = config.timeout_connection.into_inner();
|
|
||||||
|
|
||||||
async_nats::connect_with_options(config.addresses.into_inner(), opts)
|
async_nats::connect_with_options(config.addresses.into_inner(), opts)
|
||||||
.await
|
.await
|
||||||
.map(|client| {
|
.map(|client| Coordinator::Nats(Arc::new(NatsPubSub { client })))
|
||||||
reporter.watch_first_connection(client.clone(), connection_timeout);
|
|
||||||
Coordinator::Nats(Arc::new(NatsPubSub { client }))
|
|
||||||
})
|
|
||||||
.map_err(|err| format!("Failed to connect to Nats: {}", err))
|
.map_err(|err| format!("Failed to connect to Nats: {}", err))
|
||||||
}
|
}
|
||||||
|
|
||||||
/// inbuxa: whether the client is connected to a NATS server right now.
|
|
||||||
pub fn is_connected(&self) -> bool {
|
|
||||||
matches!(
|
|
||||||
self.client.connection_state(),
|
|
||||||
async_nats::connection::State::Connected
|
|
||||||
)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// inbuxa: reports the client's connection events as the server's own.
|
|
||||||
#[derive(Default)]
|
|
||||||
struct Reporter {
|
|
||||||
connected_once: AtomicBool,
|
|
||||||
// A failed attempt raises an error each time the client retries, every
|
|
||||||
// few seconds while NATS is down: report the first after each change
|
|
||||||
error_reported: AtomicBool,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl Reporter {
|
|
||||||
fn report(&self, event: async_nats::Event) {
|
|
||||||
match event {
|
|
||||||
async_nats::Event::Connected => {
|
|
||||||
self.connected_once.store(true, Ordering::Relaxed);
|
|
||||||
self.error_reported.store(false, Ordering::Relaxed);
|
|
||||||
trc::event!(Cluster(ClusterEvent::CoordinatorConnected), Type = "nats");
|
|
||||||
}
|
|
||||||
async_nats::Event::Disconnected => {
|
|
||||||
self.error_reported.store(false, Ordering::Relaxed);
|
|
||||||
trc::event!(
|
|
||||||
Cluster(ClusterEvent::CoordinatorDisconnected),
|
|
||||||
Type = "nats",
|
|
||||||
Details = "Connection lost; reconnecting in the background",
|
|
||||||
);
|
|
||||||
}
|
|
||||||
async_nats::Event::Closed => {
|
|
||||||
trc::event!(
|
|
||||||
Cluster(ClusterEvent::CoordinatorDisconnected),
|
|
||||||
Type = "nats",
|
|
||||||
Details = "Connection closed; no further attempts will be made",
|
|
||||||
);
|
|
||||||
}
|
|
||||||
async_nats::Event::ClientError(async_nats::ClientError::MaxReconnects) => {
|
|
||||||
trc::event!(
|
|
||||||
Cluster(ClusterEvent::CoordinatorDisconnected),
|
|
||||||
Type = "nats",
|
|
||||||
Details = "Gave up reconnecting (maxReconnects reached)",
|
|
||||||
);
|
|
||||||
}
|
|
||||||
async_nats::Event::ClientError(err) => {
|
|
||||||
if !self.error_reported.swap(true, Ordering::Relaxed) {
|
|
||||||
trc::event!(
|
|
||||||
Cluster(ClusterEvent::CoordinatorError),
|
|
||||||
Type = "nats",
|
|
||||||
Details = "Connection attempt failed; retrying",
|
|
||||||
Reason = err.to_string(),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
event => {
|
|
||||||
trc::event!(
|
|
||||||
Cluster(ClusterEvent::CoordinatorError),
|
|
||||||
Type = "nats",
|
|
||||||
Details = event.to_string(),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The first connection is made in the background, so say so when it
|
|
||||||
/// hasn't been made within the connection timeout. The client keeps
|
|
||||||
/// trying, and reports the connection when it comes.
|
|
||||||
fn watch_first_connection(self: &Arc<Self>, client: Client, timeout: Duration) {
|
|
||||||
let reporter = self.clone();
|
|
||||||
tokio::spawn(async move {
|
|
||||||
tokio::time::sleep(timeout).await;
|
|
||||||
if !reporter.connected_once.load(Ordering::Relaxed)
|
|
||||||
&& !matches!(
|
|
||||||
client.connection_state(),
|
|
||||||
async_nats::connection::State::Connected
|
|
||||||
)
|
|
||||||
{
|
|
||||||
trc::event!(
|
|
||||||
Cluster(ClusterEvent::CoordinatorDisconnected),
|
|
||||||
Type = "nats",
|
|
||||||
Details = "Not connected at startup; retrying in the background",
|
|
||||||
);
|
|
||||||
}
|
|
||||||
});
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -2,8 +2,6 @@
|
|||||||
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <[email protected]>
|
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <[email protected]>
|
||||||
*
|
*
|
||||||
* SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL
|
* SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL
|
||||||
*
|
|
||||||
* Modified by Coffey Labs in 2026 for INBUXA.
|
|
||||||
*/
|
*/
|
||||||
|
|
||||||
use crate::{Coordinator, Msg, PubSubStream};
|
use crate::{Coordinator, Msg, PubSubStream};
|
||||||
@@ -45,17 +43,6 @@ impl Coordinator {
|
|||||||
pub fn is_none(&self) -> bool {
|
pub fn is_none(&self) -> bool {
|
||||||
matches!(self, Coordinator::None)
|
matches!(self, Coordinator::None)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// inbuxa: whether the coordinator is connected right now, for the
|
|
||||||
/// backends that track it (NATS); `None` for the others and when no
|
|
||||||
/// coordinator is configured.
|
|
||||||
pub fn is_connected(&self) -> Option<bool> {
|
|
||||||
match self {
|
|
||||||
#[cfg(feature = "nats")]
|
|
||||||
Coordinator::Nats(store) => Some(store.is_connected()),
|
|
||||||
_ => None,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
impl PubSubStream {
|
impl PubSubStream {
|
||||||
|
|||||||
@@ -562,27 +562,6 @@ impl ParseHttp for Server {
|
|||||||
})
|
})
|
||||||
.into_http_response());
|
.into_http_response());
|
||||||
}
|
}
|
||||||
// inbuxa: the cluster coordinator's connection, for
|
|
||||||
// monitoring. It stays out of live and ready on purpose:
|
|
||||||
// a node without its coordinator still serves mail, and
|
|
||||||
// failing those would have an orchestrator restart, or
|
|
||||||
// take out of service, every node at once when the
|
|
||||||
// coordinator goes down
|
|
||||||
"cluster" => {
|
|
||||||
let coordinator = &self.core.storage.coordinator;
|
|
||||||
let (status, state) = match coordinator.is_connected() {
|
|
||||||
Some(true) => (StatusCode::OK, "connected"),
|
|
||||||
Some(false) => (StatusCode::SERVICE_UNAVAILABLE, "disconnected"),
|
|
||||||
None if coordinator.is_none() => (StatusCode::OK, "none"),
|
|
||||||
None => (StatusCode::OK, "unknown"),
|
|
||||||
};
|
|
||||||
return Ok(http_proto::JsonResponse::with_status(
|
|
||||||
status,
|
|
||||||
serde_json::json!({ "coordinator": state }),
|
|
||||||
)
|
|
||||||
.no_cache()
|
|
||||||
.into_http_response());
|
|
||||||
}
|
|
||||||
_ => (),
|
_ => (),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -427,24 +427,9 @@ pub(crate) async fn trace_query(
|
|||||||
}
|
}
|
||||||
None => false,
|
None => false,
|
||||||
},
|
},
|
||||||
// The queue id column is an integer on every search backend, and
|
Property::QueueId => match value.as_str() {
|
||||||
// holds a trace's first queue id; the keywords carry all of them
|
|
||||||
Property::QueueId => match value
|
|
||||||
.as_str()
|
|
||||||
.and_then(|v| v.trim().parse::<u64>().ok())
|
|
||||||
.or_else(|| value.as_u64())
|
|
||||||
{
|
|
||||||
Some(queue_id) => {
|
Some(queue_id) => {
|
||||||
search.extend([
|
search.push(SearchFilter::eq(TracingSearchField::QueueId, queue_id.to_string()));
|
||||||
SearchFilter::Or,
|
|
||||||
SearchFilter::eq(TracingSearchField::QueueId, queue_id),
|
|
||||||
SearchFilter::has_text(
|
|
||||||
TracingSearchField::Keywords,
|
|
||||||
queue_id.to_string(),
|
|
||||||
nlp::language::Language::None,
|
|
||||||
),
|
|
||||||
SearchFilter::End,
|
|
||||||
]);
|
|
||||||
true
|
true
|
||||||
}
|
}
|
||||||
None => false,
|
None => false,
|
||||||
|
|||||||
@@ -2,6 +2,8 @@
|
|||||||
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <[email protected]>
|
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <[email protected]>
|
||||||
*
|
*
|
||||||
* SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL
|
* SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL
|
||||||
|
*
|
||||||
|
* Modified by Coffey Labs in 2026 for INBUXA.
|
||||||
*/
|
*/
|
||||||
|
|
||||||
use crate::{
|
use crate::{
|
||||||
@@ -15,22 +17,42 @@ use jmap_proto::{error::set::SetError, types::state::State};
|
|||||||
use jmap_tools::{Key, Value};
|
use jmap_tools::{Key, Value};
|
||||||
use registry::{
|
use registry::{
|
||||||
jmap::IntoValue,
|
jmap::IntoValue,
|
||||||
schema::prelude::{Object, ObjectInner, ObjectType, Property},
|
schema::{
|
||||||
|
prelude::{Object, ObjectInner, ObjectType, Property},
|
||||||
|
structs::Task,
|
||||||
|
},
|
||||||
types::{EnumImpl, datetime::UTCDateTime},
|
types::{EnumImpl, datetime::UTCDateTime},
|
||||||
};
|
};
|
||||||
|
use services::task_manager::lock::TaskLockManager;
|
||||||
use smtp::reporting::index::{ExternalReportIndex, InternalReportIndex};
|
use smtp::reporting::index::{ExternalReportIndex, InternalReportIndex};
|
||||||
use std::str::FromStr;
|
use std::str::FromStr;
|
||||||
use store::{
|
use store::{
|
||||||
U64_LEN, ValueKey,
|
U64_LEN, ValueKey,
|
||||||
registry::{RegistryFilter, RegistryFilterValue, RegistryQuery},
|
registry::{RegistryFilter, RegistryFilterValue, RegistryQuery},
|
||||||
write::{BatchBuilder, RegistryClass, ValueClass, key::KeySerializer},
|
write::{BatchBuilder, RegistryClass, TaskQueueClass, ValueClass, key::KeySerializer},
|
||||||
};
|
};
|
||||||
use trc::AddContext;
|
use trc::AddContext;
|
||||||
use types::id::Id;
|
use types::id::Id;
|
||||||
|
|
||||||
pub(crate) async fn report_set(
|
pub(crate) async fn report_set(
|
||||||
mut set: RegistrySetResponse<'_>,
|
set: RegistrySetResponse<'_>,
|
||||||
) -> trc::Result<RegistrySetResponse<'_>> {
|
) -> trc::Result<RegistrySetResponse<'_>> {
|
||||||
|
// inbuxa: task locks taken to reschedule reports are released however
|
||||||
|
// the request ends; a held lock is renewed, so a leaked one would keep
|
||||||
|
// the report's task from ever running
|
||||||
|
let server = set.server;
|
||||||
|
let mut locked_tasks = Vec::new();
|
||||||
|
let result = report_set_locked(set, &mut locked_tasks).await;
|
||||||
|
for task_id in locked_tasks {
|
||||||
|
server.remove_index_lock(task_id).await;
|
||||||
|
}
|
||||||
|
result
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn report_set_locked<'x>(
|
||||||
|
mut set: RegistrySetResponse<'x>,
|
||||||
|
locked_tasks: &mut Vec<u64>,
|
||||||
|
) -> trc::Result<RegistrySetResponse<'x>> {
|
||||||
let object_id = set.object_type.to_id();
|
let object_id = set.object_type.to_id();
|
||||||
|
|
||||||
// Reports cannot be created
|
// Reports cannot be created
|
||||||
@@ -89,12 +111,45 @@ pub(crate) async fn report_set(
|
|||||||
.get_value::<Object>(ValueKey::from(key.clone()))
|
.get_value::<Object>(ValueKey::from(key.clone()))
|
||||||
.await?
|
.await?
|
||||||
{
|
{
|
||||||
|
// inbuxa: the report's task shares its id. Hold the task
|
||||||
|
// while its queue rows move, as x:Task/set does, and move the
|
||||||
|
// row the task is actually queued under
|
||||||
|
if !set.server.try_lock_task(item_id).await {
|
||||||
|
set.response.not_updated.append(
|
||||||
|
id,
|
||||||
|
SetError::forbidden().with_description(
|
||||||
|
"The report is being sent and cannot be rescheduled".to_string(),
|
||||||
|
),
|
||||||
|
);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
locked_tasks.push(item_id);
|
||||||
|
let queued = set
|
||||||
|
.server
|
||||||
|
.store()
|
||||||
|
.get_value::<Task>(ValueKey::from(ValueClass::TaskQueue(
|
||||||
|
TaskQueueClass::Task { id: item_id },
|
||||||
|
)))
|
||||||
|
.await?;
|
||||||
|
|
||||||
match &mut report_obj.inner {
|
match &mut report_obj.inner {
|
||||||
ObjectInner::DmarcInternalReport(report) => {
|
ObjectInner::DmarcInternalReport(report) => {
|
||||||
report.reschedule_ops(&mut batch, item_id, report_obj.revision, deliver_at);
|
report.reschedule_ops(
|
||||||
|
&mut batch,
|
||||||
|
item_id,
|
||||||
|
report_obj.revision,
|
||||||
|
deliver_at,
|
||||||
|
queued.as_ref(),
|
||||||
|
);
|
||||||
}
|
}
|
||||||
ObjectInner::TlsInternalReport(report) => {
|
ObjectInner::TlsInternalReport(report) => {
|
||||||
report.reschedule_ops(&mut batch, item_id, report_obj.revision, deliver_at);
|
report.reschedule_ops(
|
||||||
|
&mut batch,
|
||||||
|
item_id,
|
||||||
|
report_obj.revision,
|
||||||
|
deliver_at,
|
||||||
|
queued.as_ref(),
|
||||||
|
);
|
||||||
}
|
}
|
||||||
_ => {}
|
_ => {}
|
||||||
}
|
}
|
||||||
@@ -156,6 +211,9 @@ pub(crate) async fn report_set(
|
|||||||
.write(batch.build_all())
|
.write(batch.build_all())
|
||||||
.await
|
.await
|
||||||
.caused_by(trc::location!())?;
|
.caused_by(trc::location!())?;
|
||||||
|
// inbuxa: a rescheduled report may now be due sooner than the task
|
||||||
|
// manager's next scan
|
||||||
|
set.server.notify_task_queue();
|
||||||
}
|
}
|
||||||
|
|
||||||
Ok(set)
|
Ok(set)
|
||||||
|
|||||||
@@ -463,15 +463,10 @@ pub(crate) async fn task_query(
|
|||||||
.set_values(typ.is_some()),
|
.set_values(typ.is_some()),
|
||||||
|key, value| {
|
|key, value| {
|
||||||
if let Some(typ) = typ {
|
if let Some(typ) = typ {
|
||||||
let task_type =
|
// inbuxa: a row whose type can't be read matches no type
|
||||||
TaskType::from_id(value.deserialize_be_u16(0)?).ok_or_else(|| {
|
// filter; the task manager logs and repairs it
|
||||||
trc::StoreEvent::DataCorruption
|
let task_type = value.deserialize_be_u16(0).ok().and_then(TaskType::from_id);
|
||||||
.into_err()
|
if task_type != Some(typ) {
|
||||||
.ctx(trc::Key::Key, key.to_vec())
|
|
||||||
.ctx(trc::Key::Value, value.to_vec())
|
|
||||||
.caused_by(trc::location!())
|
|
||||||
})?;
|
|
||||||
if task_type != typ {
|
|
||||||
return Ok(true);
|
return Ok(true);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -109,10 +109,6 @@ async fn main() -> std::io::Result<()> {
|
|||||||
// Wait for shutdown signal
|
// Wait for shutdown signal
|
||||||
wait_for_shutdown().await;
|
wait_for_shutdown().await;
|
||||||
|
|
||||||
// inbuxa: hand back the task locks this node holds, so other nodes can
|
|
||||||
// run those tasks now rather than when the locks expire
|
|
||||||
services::task_manager::lock::release_task_locks(&inner.build_server()).await;
|
|
||||||
|
|
||||||
// Shutdown collector
|
// Shutdown collector
|
||||||
Collector::shutdown();
|
Collector::shutdown();
|
||||||
|
|
||||||
|
|||||||
@@ -26,7 +26,7 @@ pub fn spawn_broadcast_subscriber(inner: Arc<Inner>, mut shutdown_rx: watch::Rec
|
|||||||
};
|
};
|
||||||
|
|
||||||
tokio::spawn(async move {
|
tokio::spawn(async move {
|
||||||
let mut retry_count: u32 = 0;
|
let mut retry_count = 0;
|
||||||
|
|
||||||
trc::event!(Cluster(ClusterEvent::SubscriberStart));
|
trc::event!(Cluster(ClusterEvent::SubscriberStart));
|
||||||
|
|
||||||
@@ -53,7 +53,7 @@ pub fn spawn_broadcast_subscriber(inner: Arc<Inner>, mut shutdown_rx: watch::Rec
|
|||||||
);
|
);
|
||||||
|
|
||||||
match tokio::time::timeout(
|
match tokio::time::timeout(
|
||||||
subscribe_retry_delay(retry_count),
|
Duration::from_secs(1 << retry_count.max(6)),
|
||||||
shutdown_rx.changed(),
|
shutdown_rx.changed(),
|
||||||
)
|
)
|
||||||
.await
|
.await
|
||||||
@@ -62,7 +62,7 @@ pub fn spawn_broadcast_subscriber(inner: Arc<Inner>, mut shutdown_rx: watch::Rec
|
|||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
Err(_) => {
|
Err(_) => {
|
||||||
retry_count = retry_count.saturating_add(1);
|
retry_count += 1;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -234,11 +234,6 @@ pub fn spawn_broadcast_subscriber(inner: Arc<Inner>, mut shutdown_rx: watch::Rec
|
|||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Delay before the next subscribe attempt: 1 s, 2 s, 4 s ... capped at 64 s.
|
|
||||||
fn subscribe_retry_delay(retry_count: u32) -> Duration {
|
|
||||||
Duration::from_secs(1u64 << retry_count.min(6))
|
|
||||||
}
|
|
||||||
|
|
||||||
fn log_event(event: &BroadcastEvent) -> trc::Value {
|
fn log_event(event: &BroadcastEvent) -> trc::Value {
|
||||||
match event {
|
match event {
|
||||||
BroadcastEvent::PushNotification(notification) => match notification {
|
BroadcastEvent::PushNotification(notification) => match notification {
|
||||||
@@ -301,19 +296,3 @@ fn log_event(event: &BroadcastEvent) -> trc::Value {
|
|||||||
BroadcastEvent::QueueRefresh => "QueueRefresh".into(),
|
BroadcastEvent::QueueRefresh => "QueueRefresh".into(),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
|
||||||
mod tests {
|
|
||||||
use super::subscribe_retry_delay;
|
|
||||||
use std::time::Duration;
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn subscribe_retry_backoff_grows_then_caps() {
|
|
||||||
let schedule: Vec<u64> = (0..10)
|
|
||||||
.map(|n| subscribe_retry_delay(n).as_secs())
|
|
||||||
.collect();
|
|
||||||
assert_eq!(schedule, vec![1, 2, 4, 8, 16, 32, 64, 64, 64, 64]);
|
|
||||||
// No shift overflow at the top of the range.
|
|
||||||
assert_eq!(subscribe_retry_delay(u32::MAX), Duration::from_secs(64));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -91,15 +91,7 @@ impl SearchIndexTask for Server {
|
|||||||
build_contact_document(self, account_id, document_id).await
|
build_contact_document(self, account_id, document_id).await
|
||||||
}
|
}
|
||||||
IndexDocumentType::File => {
|
IndexDocumentType::File => {
|
||||||
// File indexing not implemented yet. inbuxa: still
|
// File indexing not implemented yet
|
||||||
// one result per task: update_tasks pairs them by
|
|
||||||
// position, and a missing one shifts every result
|
|
||||||
// after it onto the wrong task
|
|
||||||
results.push(IndexTaskResult {
|
|
||||||
task_type: TaskType::Insert,
|
|
||||||
index: task.document_type,
|
|
||||||
result: TaskResult::Ignored,
|
|
||||||
});
|
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
@@ -575,14 +567,20 @@ async fn build_contact_document(
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
// inbuxa: MON-16: a trace's search document, when trace search is on
|
// inbuxa: MON-16: a trace's search document, when trace search is on:
|
||||||
|
// its event types, queue ids, and addresses, their domains, hosts, IPs,
|
||||||
|
// message ids and account names as keywords
|
||||||
async fn build_tracing_span_document(
|
async fn build_tracing_span_document(
|
||||||
server: &Server,
|
server: &Server,
|
||||||
span_id: u64,
|
span_id: u64,
|
||||||
) -> trc::Result<Option<IndexDocument>> {
|
) -> trc::Result<Option<IndexDocument>> {
|
||||||
use common::telemetry::tracers::store::MaybeTrace;
|
use common::telemetry::tracers::store::MaybeTrace;
|
||||||
use registry::schema::structs::Search;
|
use registry::schema::{enums::SearchTracingField, structs::Search};
|
||||||
use store::write::{TelemetryClass, ValueClass};
|
use store::{
|
||||||
|
search::TracingSearchField,
|
||||||
|
write::{TelemetryClass, ValueClass},
|
||||||
|
};
|
||||||
|
use trc::Key;
|
||||||
|
|
||||||
let settings = server
|
let settings = server
|
||||||
.registry()
|
.registry()
|
||||||
@@ -592,6 +590,7 @@ async fn build_tracing_span_document(
|
|||||||
if !settings.index_telemetry {
|
if !settings.index_telemetry {
|
||||||
return Ok(None);
|
return Ok(None);
|
||||||
}
|
}
|
||||||
|
let wants = |field: SearchTracingField| settings.index_tracing_fields.iter().any(|f| *f == field);
|
||||||
let Some(MaybeTrace(Some(trace))) = server
|
let Some(MaybeTrace(Some(trace))) = server
|
||||||
.tracing_store()
|
.tracing_store()
|
||||||
.get_value::<MaybeTrace>(ValueKey::from(ValueClass::Telemetry(TelemetryClass::Span(
|
.get_value::<MaybeTrace>(ValueKey::from(ValueClass::Telemetry(TelemetryClass::Span(
|
||||||
@@ -602,67 +601,23 @@ async fn build_tracing_span_document(
|
|||||||
return Ok(None);
|
return Ok(None);
|
||||||
};
|
};
|
||||||
|
|
||||||
Ok(Some(trace_search_document(
|
|
||||||
span_id,
|
|
||||||
&trace,
|
|
||||||
&settings
|
|
||||||
.index_tracing_fields
|
|
||||||
.iter()
|
|
||||||
.copied()
|
|
||||||
.collect::<Vec<_>>(),
|
|
||||||
)))
|
|
||||||
}
|
|
||||||
|
|
||||||
/// inbuxa: MON-16: the search document for a stored trace.
|
|
||||||
///
|
|
||||||
/// The event type and queue id columns are integers on every search backend
|
|
||||||
/// (BIGINT on PostgreSQL and MySQL, long on Elasticsearch), and each holds a
|
|
||||||
/// single value per trace: the event type is the trace's opening event, the
|
|
||||||
/// one `x:Trace/query` filters on, and the queue id is the first queue id the
|
|
||||||
/// trace mentions. Every queue id also goes into the keywords, so a session
|
|
||||||
/// that queued several messages is found by any of them.
|
|
||||||
pub fn trace_search_document(
|
|
||||||
span_id: u64,
|
|
||||||
trace: ®istry::schema::structs::Trace,
|
|
||||||
fields: &[registry::schema::enums::SearchTracingField],
|
|
||||||
) -> IndexDocument {
|
|
||||||
use registry::schema::{enums::SearchTracingField, structs::TraceValue};
|
|
||||||
use store::search::TracingSearchField;
|
|
||||||
use trc::Key;
|
|
||||||
|
|
||||||
let wants = |field: SearchTracingField| fields.contains(&field);
|
|
||||||
let mut document = IndexDocument::new(SearchIndex::Tracing).with_id(span_id);
|
let mut document = IndexDocument::new(SearchIndex::Tracing).with_id(span_id);
|
||||||
if wants(SearchTracingField::EventType)
|
|
||||||
&& let Some(first) = trace.events.iter().next()
|
|
||||||
{
|
|
||||||
document.index_unsigned(TracingSearchField::EventType, first.event.to_id() as u64);
|
|
||||||
}
|
|
||||||
|
|
||||||
let mut seen = store::ahash::AHashSet::new();
|
let mut seen = store::ahash::AHashSet::new();
|
||||||
let mut queue_id_indexed = false;
|
|
||||||
for event in trace.events.iter() {
|
for event in trace.events.iter() {
|
||||||
|
if wants(SearchTracingField::EventType) && seen.insert(event.event.as_str().to_string()) {
|
||||||
|
document.index_keyword(TracingSearchField::EventType, event.event.as_str());
|
||||||
|
}
|
||||||
for kv in event.key_values.iter() {
|
for kv in event.key_values.iter() {
|
||||||
let text = match &kv.value {
|
let text = match &kv.value {
|
||||||
TraceValue::String(v) => v.value.clone(),
|
registry::schema::structs::TraceValue::String(v) => v.value.clone(),
|
||||||
TraceValue::UnsignedInt(v) => v.value.to_string(),
|
registry::schema::structs::TraceValue::UnsignedInt(v) => v.value.to_string(),
|
||||||
TraceValue::IpAddr(v) => v.value.to_string(),
|
registry::schema::structs::TraceValue::IpAddr(v) => v.value.to_string(),
|
||||||
_ => continue,
|
_ => continue,
|
||||||
};
|
};
|
||||||
match kv.key {
|
match kv.key {
|
||||||
Key::QueueId => {
|
Key::QueueId if wants(SearchTracingField::QueueId) => {
|
||||||
let Ok(queue_id) = text.parse::<u64>() else {
|
if seen.insert(format!("q:{text}")) {
|
||||||
continue;
|
document.index_keyword(TracingSearchField::QueueId, &text);
|
||||||
};
|
|
||||||
if wants(SearchTracingField::QueueId) && !queue_id_indexed {
|
|
||||||
document.index_unsigned(TracingSearchField::QueueId, queue_id);
|
|
||||||
queue_id_indexed = true;
|
|
||||||
}
|
|
||||||
if wants(SearchTracingField::Keywords) && seen.insert(format!("k:{text}")) {
|
|
||||||
document.index_text(
|
|
||||||
TracingSearchField::Keywords,
|
|
||||||
&text,
|
|
||||||
nlp::language::Language::None,
|
|
||||||
);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
Key::From
|
Key::From
|
||||||
@@ -693,7 +648,7 @@ pub fn trace_search_document(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
document
|
Ok(Some(document))
|
||||||
}
|
}
|
||||||
|
|
||||||
// inbuxa: UD-1, UD-4: archives a deleted file, event or contact noted at
|
// inbuxa: UD-1, UD-4: archives a deleted file, event or contact noted at
|
||||||
|
|||||||
@@ -2,8 +2,6 @@
|
|||||||
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <[email protected]>
|
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <[email protected]>
|
||||||
*
|
*
|
||||||
* SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL
|
* SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL
|
||||||
*
|
|
||||||
* Modified by Coffey Labs in 2026 for INBUXA.
|
|
||||||
*/
|
*/
|
||||||
|
|
||||||
use crate::task_manager::*;
|
use crate::task_manager::*;
|
||||||
@@ -15,21 +13,13 @@ pub trait TaskLockManager: Sync + Send {
|
|||||||
|
|
||||||
impl TaskLockManager for Server {
|
impl TaskLockManager for Server {
|
||||||
async fn try_lock_task(&self, id: u64) -> bool {
|
async fn try_lock_task(&self, id: u64) -> bool {
|
||||||
// inbuxa: a node that is stopping claims nothing new
|
|
||||||
let locks = &self.inner.ipc.task_locks;
|
|
||||||
if locks.is_stopping() {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
match self
|
match self
|
||||||
.in_memory_store()
|
.in_memory_store()
|
||||||
.try_lock(KV_LOCK_TASK, &id.to_be_bytes(), locks.expiry())
|
.try_lock(KV_LOCK_TASK, &id.to_be_bytes(), DEFAULT_LOCK_EXPIRY)
|
||||||
.await
|
.await
|
||||||
{
|
{
|
||||||
Ok(result) => {
|
Ok(result) => {
|
||||||
if result {
|
if !result {
|
||||||
locks.insert(id);
|
|
||||||
} else {
|
|
||||||
trc::event!(
|
trc::event!(
|
||||||
TaskManager(TaskManagerEvent::TaskLocked),
|
TaskManager(TaskManagerEvent::TaskLocked),
|
||||||
Id = id,
|
Id = id,
|
||||||
@@ -58,27 +48,5 @@ impl TaskLockManager for Server {
|
|||||||
.caused_by(trc::location!())
|
.caused_by(trc::location!())
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
self.inner.ipc.task_locks.remove(id);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// inbuxa: on a graceful stop, stops claiming tasks and releases every task
|
|
||||||
/// lock this node holds, so the rest of the cluster can pick the tasks up at
|
|
||||||
/// once instead of after the lock expires. Returns how many were released.
|
|
||||||
pub async fn release_task_locks(server: &Server) -> usize {
|
|
||||||
let ids = server.inner.ipc.task_locks.stop();
|
|
||||||
for id in &ids {
|
|
||||||
if let Err(err) = server
|
|
||||||
.in_memory_store()
|
|
||||||
.remove_lock(KV_LOCK_TASK, &id.to_be_bytes())
|
|
||||||
.await
|
|
||||||
{
|
|
||||||
trc::error!(
|
|
||||||
err.details("Failed to release task lock on shutdown")
|
|
||||||
.ctx(trc::Key::Id, *id)
|
|
||||||
.caused_by(trc::location!())
|
|
||||||
);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
ids.len()
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -20,7 +20,7 @@ use crate::task_manager::report::{self, SubmitReportTask};
|
|||||||
use crate::task_manager::restore_item::RestoreItemTask;
|
use crate::task_manager::restore_item::RestoreItemTask;
|
||||||
use crate::task_manager::spam_classifier::SpamFilterMaintenanceTask;
|
use crate::task_manager::spam_classifier::SpamFilterMaintenanceTask;
|
||||||
use crate::task_manager::{
|
use crate::task_manager::{
|
||||||
CLAIM_RECHECK_INTERVAL, Locked, QUEUE_REFRESH_INTERVAL, TaskDetails, TaskFailureType, TaskInfo,
|
DEFAULT_LOCK_EXPIRY, Locked, QUEUE_REFRESH_INTERVAL, TaskDetails, TaskFailureType, TaskInfo,
|
||||||
TaskJob, TaskManagerIpc, TaskResult,
|
TaskJob, TaskManagerIpc, TaskResult,
|
||||||
};
|
};
|
||||||
use common::BuildServer;
|
use common::BuildServer;
|
||||||
@@ -29,6 +29,7 @@ use common::network::limiter::ConcurrencyLimiter;
|
|||||||
use common::network::{ServerInstance, TcpAcceptor};
|
use common::network::{ServerInstance, TcpAcceptor};
|
||||||
use common::{Inner, Server};
|
use common::{Inner, Server};
|
||||||
use registry::schema::enums::TaskType;
|
use registry::schema::enums::TaskType;
|
||||||
|
use registry::schema::prelude::ObjectType;
|
||||||
use registry::schema::structs::{
|
use registry::schema::structs::{
|
||||||
Task, TaskManager, TaskRetryStrategy, TaskStatus, TaskStatusFailed, TaskStatusRetry,
|
Task, TaskManager, TaskRetryStrategy, TaskStatus, TaskStatusFailed, TaskStatusRetry,
|
||||||
};
|
};
|
||||||
@@ -126,47 +127,72 @@ pub fn spawn_task_manager(inner: Arc<Inner>) {
|
|||||||
let server = inner.build_server();
|
let server = inner.build_server();
|
||||||
let batch_size = server.core.email.index_batch_size;
|
let batch_size = server.core.email.index_batch_size;
|
||||||
let mut batch = Vec::with_capacity(batch_size);
|
let mut batch = Vec::with_capacity(batch_size);
|
||||||
if let Some(task) = fetch_task(&server, job).await {
|
match server
|
||||||
batch.push(task);
|
.store()
|
||||||
|
.get_value::<Task>(ValueKey::from(ValueClass::TaskQueue(
|
||||||
|
TaskQueueClass::Task { id: job.id },
|
||||||
|
)))
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
Ok(Some(task)) => {
|
||||||
|
batch.push(TaskDetails { task, info: job });
|
||||||
|
}
|
||||||
|
Ok(None) => {
|
||||||
|
trc::event!(
|
||||||
|
TaskManager(TaskManagerEvent::TaskIgnored),
|
||||||
|
Id = job.id,
|
||||||
|
Reason = "Task not found in store, likely already processed.",
|
||||||
|
);
|
||||||
|
}
|
||||||
|
Err(err) => {
|
||||||
|
trc::error!(
|
||||||
|
err.id(job.id)
|
||||||
|
.details("Failed to retrieve task details.")
|
||||||
|
.caused_by(trc::location!())
|
||||||
|
);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
while batch.len() < batch_size {
|
while batch.len() < batch_size {
|
||||||
match rx.try_recv() {
|
match rx.try_recv() {
|
||||||
Ok(job) => {
|
Ok(job) => {
|
||||||
if let Some(task) = fetch_task(&server, job).await {
|
match server
|
||||||
batch.push(task);
|
.store()
|
||||||
|
.get_value::<Task>(ValueKey::from(ValueClass::TaskQueue(
|
||||||
|
TaskQueueClass::Task { id: job.id },
|
||||||
|
)))
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
Ok(Some(task)) => {
|
||||||
|
batch.push(TaskDetails { task, info: job });
|
||||||
|
}
|
||||||
|
Ok(None) => {
|
||||||
|
trc::event!(
|
||||||
|
TaskManager(TaskManagerEvent::TaskIgnored),
|
||||||
|
Id = job.id,
|
||||||
|
Reason = "Task not found in store, likely already processed.",
|
||||||
|
);
|
||||||
|
}
|
||||||
|
Err(err) => {
|
||||||
|
trc::error!(
|
||||||
|
err.id(job.id)
|
||||||
|
.details("Failed to retrieve task details.")
|
||||||
|
.caused_by(trc::location!())
|
||||||
|
);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
Err(_) => break,
|
Err(_) => break,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Dispatch. inbuxa: on a task of its own, so a panic
|
// Dispatch
|
||||||
// releases the batch's locks and leaves this worker
|
|
||||||
// running; a dead worker would keep claiming tasks it
|
|
||||||
// can never run
|
|
||||||
let mut refresh_queue = false;
|
let mut refresh_queue = false;
|
||||||
let ids = batch.iter().map(|task| task.info.id).collect::<Vec<_>>();
|
let results = server.index(&batch).await.into_iter().map(|r| {
|
||||||
let run = {
|
refresh_queue |= r.result.is_retry();
|
||||||
let server = server.clone();
|
r.result
|
||||||
tokio::spawn(async move {
|
});
|
||||||
let results = server.index(&batch).await;
|
update_tasks(&server, &mut batch, results).await;
|
||||||
(batch, results)
|
|
||||||
})
|
|
||||||
};
|
|
||||||
match run.await {
|
|
||||||
Ok((mut batch, results)) => {
|
|
||||||
let results = results.into_iter().map(|r| {
|
|
||||||
refresh_queue |= r.result.is_retry();
|
|
||||||
r.result
|
|
||||||
});
|
|
||||||
update_tasks(&server, &mut batch, results).await;
|
|
||||||
}
|
|
||||||
Err(err) => {
|
|
||||||
worker_failed(&server, &ids, err).await;
|
|
||||||
refresh_queue = true;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if refresh_queue || rx.is_empty() {
|
if refresh_queue || rx.is_empty() {
|
||||||
server.notify_task_queue();
|
server.notify_task_queue();
|
||||||
@@ -180,31 +206,83 @@ pub fn spawn_task_manager(inner: Arc<Inner>) {
|
|||||||
let server = inner.build_server();
|
let server = inner.build_server();
|
||||||
let mut refresh_queue = false;
|
let mut refresh_queue = false;
|
||||||
|
|
||||||
if let Some(TaskDetails { task, info }) = fetch_task(&server, job).await {
|
match server
|
||||||
// inbuxa: on a task of its own, as above
|
.store()
|
||||||
let run = {
|
.get_value::<Task>(ValueKey::from(ValueClass::TaskQueue(
|
||||||
let server = server.clone();
|
TaskQueueClass::Task { id: job.id },
|
||||||
let server_instance = server_instance.clone();
|
)))
|
||||||
tokio::spawn(async move {
|
.await
|
||||||
let result = run_task(&server, &task, server_instance).await;
|
{
|
||||||
(task, result)
|
Ok(Some(task)) => {
|
||||||
})
|
let result = match &task {
|
||||||
};
|
Task::CalendarAlarmEmail(task) => {
|
||||||
match run.await {
|
server.send_email_alarm(task, server_instance.clone()).await
|
||||||
Ok((task, result)) => {
|
}
|
||||||
refresh_queue = result.is_retry();
|
Task::CalendarAlarmNotification(task) => {
|
||||||
|
server.send_display_alarm(task).await
|
||||||
|
}
|
||||||
|
Task::CalendarItipMessage(task) => {
|
||||||
|
server.send_imip(task, server_instance.clone()).await
|
||||||
|
}
|
||||||
|
Task::MergeThreads(task) => server.merge_threads(task).await,
|
||||||
|
Task::DmarcReport(task) => {
|
||||||
|
server
|
||||||
|
.submit_report(report::ReportId::Dmarc(task.report_id.id()))
|
||||||
|
.await
|
||||||
|
}
|
||||||
|
Task::TlsReport(task) => {
|
||||||
|
server
|
||||||
|
.submit_report(report::ReportId::Tls(task.report_id.id()))
|
||||||
|
.await
|
||||||
|
}
|
||||||
|
Task::RestoreArchivedItem(task) => server.restore_item(task).await,
|
||||||
|
Task::DestroyAccount(task) => server.destroy_account(task).await,
|
||||||
|
Task::AccountMaintenance(task) => {
|
||||||
|
server.account_maintenance(task).await
|
||||||
|
}
|
||||||
|
Task::TenantMaintenance(task) => {
|
||||||
|
server.tenant_maintenance(task).await
|
||||||
|
}
|
||||||
|
Task::StoreMaintenance(task) => {
|
||||||
|
server.store_maintenance(task).await
|
||||||
|
}
|
||||||
|
Task::SpamFilterMaintenance(task) => {
|
||||||
|
Box::pin(server.spam_filter_maintenance(task)).await
|
||||||
|
}
|
||||||
|
Task::AcmeRenewal(task) => server.acme_management(task).await,
|
||||||
|
Task::DkimManagement(task_dkim_rotation) => {
|
||||||
|
server.dkim_management(task_dkim_rotation).await
|
||||||
|
}
|
||||||
|
Task::DnsManagement(task_dns_management) => {
|
||||||
|
server.dns_management(task_dns_management).await
|
||||||
|
}
|
||||||
|
Task::IndexDocument(_)
|
||||||
|
| Task::UnindexDocument(_)
|
||||||
|
| Task::IndexTrace(_) => unreachable!(),
|
||||||
|
};
|
||||||
|
|
||||||
update_tasks(
|
refresh_queue = result.is_retry();
|
||||||
&server,
|
|
||||||
&mut [TaskDetails { task, info }],
|
update_tasks(
|
||||||
vec![result],
|
&server,
|
||||||
)
|
&mut [TaskDetails { task, info: job }],
|
||||||
.await;
|
vec![result],
|
||||||
}
|
)
|
||||||
Err(err) => {
|
.await;
|
||||||
worker_failed(&server, &[info.id], err).await;
|
}
|
||||||
refresh_queue = true;
|
Ok(None) => {
|
||||||
}
|
trc::event!(
|
||||||
|
TaskManager(TaskManagerEvent::TaskIgnored),
|
||||||
|
Id = job.id,
|
||||||
|
Reason = "Task not found in store, likely already processed.",
|
||||||
|
);
|
||||||
|
}
|
||||||
|
Err(err) => {
|
||||||
|
trc::error!(
|
||||||
|
err.id(job.id)
|
||||||
|
.details("Failed to retrieve task details.")
|
||||||
|
.caused_by(trc::location!())
|
||||||
|
);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -243,13 +321,6 @@ pub(crate) trait TaskQueueManager: Sync + Send {
|
|||||||
|
|
||||||
impl TaskQueueManager for Server {
|
impl TaskQueueManager for Server {
|
||||||
async fn process_tasks(&self, ipc: &mut TaskManagerIpc) -> Duration {
|
async fn process_tasks(&self, ipc: &mut TaskManagerIpc) -> Duration {
|
||||||
// inbuxa: a node that is stopping has released its locks and claims
|
|
||||||
// nothing new
|
|
||||||
let task_locks = &self.inner.ipc.task_locks;
|
|
||||||
if task_locks.is_stopping() {
|
|
||||||
return Duration::from_secs(QUEUE_REFRESH_INTERVAL);
|
|
||||||
}
|
|
||||||
let lock_expiry = task_locks.expiry();
|
|
||||||
let now_timestamp = now();
|
let now_timestamp = now();
|
||||||
let from_key = ValueKey::<ValueClass> {
|
let from_key = ValueKey::<ValueClass> {
|
||||||
account_id: 0,
|
account_id: 0,
|
||||||
@@ -269,6 +340,7 @@ impl TaskQueueManager for Server {
|
|||||||
|
|
||||||
// Retrieve tasks pending to be processed
|
// Retrieve tasks pending to be processed
|
||||||
let mut tasks = Vec::new();
|
let mut tasks = Vec::new();
|
||||||
|
let mut unreadable = Vec::new();
|
||||||
let now = Instant::now();
|
let now = Instant::now();
|
||||||
let mut next_event = None;
|
let mut next_event = None;
|
||||||
let roles = &self.core.network.roles;
|
let roles = &self.core.network.roles;
|
||||||
@@ -283,12 +355,21 @@ impl TaskQueueManager for Server {
|
|||||||
let task_id = key.deserialize_be_u64(U64_LEN)?;
|
let task_id = key.deserialize_be_u64(U64_LEN)?;
|
||||||
|
|
||||||
if task_due <= now_timestamp {
|
if task_due <= now_timestamp {
|
||||||
let task_type_idx = value.deserialize_be_u16(0)?;
|
// inbuxa: a row whose task type can't be read is
|
||||||
let task_type = TaskType::from_id(task_type_idx).ok_or_else(|| {
|
// set aside, not allowed to end the scan: every
|
||||||
trc::StoreEvent::DataCorruption
|
// task due after it would wait behind it
|
||||||
.caused_by(trc::location!())
|
let Some((task_type_idx, task_type)) = value
|
||||||
.ctx(trc::Key::Value, value)
|
.deserialize_be_u16(0)
|
||||||
})?;
|
.ok()
|
||||||
|
.and_then(|idx| TaskType::from_id(idx).map(|typ| (idx, typ)))
|
||||||
|
else {
|
||||||
|
unreadable.push(UnreadableDueRow {
|
||||||
|
due: task_due,
|
||||||
|
id: task_id,
|
||||||
|
value: value.to_vec(),
|
||||||
|
});
|
||||||
|
return Ok(true);
|
||||||
|
};
|
||||||
let enabled = match task_type {
|
let enabled = match task_type {
|
||||||
TaskType::IndexDocument
|
TaskType::IndexDocument
|
||||||
| TaskType::UnindexDocument
|
| TaskType::UnindexDocument
|
||||||
@@ -325,7 +406,9 @@ impl TaskQueueManager for Server {
|
|||||||
let locked = entry.get_mut();
|
let locked = entry.get_mut();
|
||||||
if locked.expires <= now || locked.due < task_due {
|
if locked.expires <= now || locked.due < task_due {
|
||||||
locked.expires = Instant::now()
|
locked.expires = Instant::now()
|
||||||
+ std::time::Duration::from_secs(lock_expiry + 1);
|
+ std::time::Duration::from_secs(
|
||||||
|
DEFAULT_LOCK_EXPIRY + 1,
|
||||||
|
);
|
||||||
locked.due = task_due;
|
locked.due = task_due;
|
||||||
tasks.push((
|
tasks.push((
|
||||||
TaskJob {
|
TaskJob {
|
||||||
@@ -341,7 +424,9 @@ impl TaskQueueManager for Server {
|
|||||||
Entry::Vacant(entry) => {
|
Entry::Vacant(entry) => {
|
||||||
entry.insert(Locked {
|
entry.insert(Locked {
|
||||||
expires: Instant::now()
|
expires: Instant::now()
|
||||||
+ std::time::Duration::from_secs(lock_expiry + 1),
|
+ std::time::Duration::from_secs(
|
||||||
|
DEFAULT_LOCK_EXPIRY + 1,
|
||||||
|
),
|
||||||
due: task_due,
|
due: task_due,
|
||||||
revision: ipc.revision,
|
revision: ipc.revision,
|
||||||
});
|
});
|
||||||
@@ -374,6 +459,11 @@ impl TaskQueueManager for Server {
|
|||||||
);
|
);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
if !unreadable.is_empty() && repair_due_rows(self, unreadable).await {
|
||||||
|
// Look again at once for the rows that were rewritten
|
||||||
|
self.notify_task_queue();
|
||||||
|
}
|
||||||
|
|
||||||
if !tasks.is_empty() {
|
if !tasks.is_empty() {
|
||||||
trc::event!(
|
trc::event!(
|
||||||
TaskManager(TaskManagerEvent::TaskAcquired),
|
TaskManager(TaskManagerEvent::TaskAcquired),
|
||||||
@@ -392,26 +482,12 @@ impl TaskQueueManager for Server {
|
|||||||
let tx = &ipc.txs[task_type_idx as usize];
|
let tx = &ipc.txs[task_type_idx as usize];
|
||||||
|
|
||||||
if tx.capacity() > 0 {
|
if tx.capacity() > 0 {
|
||||||
let id = task_job.id;
|
if self.try_lock_task(task_job.id).await && tx.send(task_job).await.is_err() {
|
||||||
if !self.try_lock_task(id).await {
|
|
||||||
// inbuxa: another node holds the task. Look again after a
|
|
||||||
// short while rather than a full lock lifetime from now:
|
|
||||||
// the holder may have claimed it after this scan began,
|
|
||||||
// or run on a clock ahead of this one, and waiting the
|
|
||||||
// whole lifetime again would leave the task stuck for
|
|
||||||
// another hour past its lock if that holder died
|
|
||||||
if let Some(locked) = ipc.locked.get_mut(&id) {
|
|
||||||
locked.expires =
|
|
||||||
Instant::now() + Duration::from_secs(claim_recheck_interval(lock_expiry));
|
|
||||||
}
|
|
||||||
} else if tx.send(task_job).await.is_err() {
|
|
||||||
trc::event!(
|
trc::event!(
|
||||||
Server(trc::ServerEvent::ThreadError),
|
Server(trc::ServerEvent::ThreadError),
|
||||||
Details = "Error sending task.",
|
Details = "Error sending task.",
|
||||||
CausedBy = trc::location!()
|
CausedBy = trc::location!()
|
||||||
);
|
);
|
||||||
// inbuxa: nothing will run it here, so don't hold it
|
|
||||||
self.remove_index_lock(id).await;
|
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
// If the channel is full, release the lock so it can be picked up in the next iteration
|
// If the channel is full, release the lock so it can be picked up in the next iteration
|
||||||
@@ -423,117 +499,9 @@ impl TaskQueueManager for Server {
|
|||||||
let now = Instant::now();
|
let now = Instant::now();
|
||||||
ipc.locked
|
ipc.locked
|
||||||
.retain(|_, locked| locked.expires > now && locked.revision == ipc.revision);
|
.retain(|_, locked| locked.expires > now && locked.revision == ipc.revision);
|
||||||
let sleep_for = Duration::from_secs(next_event.map_or(QUEUE_REFRESH_INTERVAL, |timestamp| {
|
Duration::from_secs(next_event.map_or(QUEUE_REFRESH_INTERVAL, |timestamp| {
|
||||||
timestamp.saturating_sub(store::write::now())
|
timestamp.saturating_sub(store::write::now())
|
||||||
}));
|
}))
|
||||||
|
|
||||||
// inbuxa: wake up when a claim held elsewhere is due to be tried
|
|
||||||
// again, rather than only on the next task or refresh
|
|
||||||
ipc.locked
|
|
||||||
.values()
|
|
||||||
.map(|locked| locked.expires.saturating_duration_since(now))
|
|
||||||
.min()
|
|
||||||
.map_or(sleep_for, |recheck| sleep_for.min(recheck.max(Duration::from_secs(1))))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn run_task(
|
|
||||||
server: &Server,
|
|
||||||
task: &Task,
|
|
||||||
server_instance: Arc<ServerInstance>,
|
|
||||||
) -> TaskResult {
|
|
||||||
match task {
|
|
||||||
Task::CalendarAlarmEmail(task) => {
|
|
||||||
server.send_email_alarm(task, server_instance.clone()).await
|
|
||||||
}
|
|
||||||
Task::CalendarAlarmNotification(task) => {
|
|
||||||
server.send_display_alarm(task).await
|
|
||||||
}
|
|
||||||
Task::CalendarItipMessage(task) => {
|
|
||||||
server.send_imip(task, server_instance.clone()).await
|
|
||||||
}
|
|
||||||
Task::MergeThreads(task) => server.merge_threads(task).await,
|
|
||||||
Task::DmarcReport(task) => {
|
|
||||||
server
|
|
||||||
.submit_report(report::ReportId::Dmarc(task.report_id.id()))
|
|
||||||
.await
|
|
||||||
}
|
|
||||||
Task::TlsReport(task) => {
|
|
||||||
server
|
|
||||||
.submit_report(report::ReportId::Tls(task.report_id.id()))
|
|
||||||
.await
|
|
||||||
}
|
|
||||||
Task::RestoreArchivedItem(task) => server.restore_item(task).await,
|
|
||||||
Task::DestroyAccount(task) => server.destroy_account(task).await,
|
|
||||||
Task::AccountMaintenance(task) => {
|
|
||||||
server.account_maintenance(task).await
|
|
||||||
}
|
|
||||||
Task::TenantMaintenance(task) => {
|
|
||||||
server.tenant_maintenance(task).await
|
|
||||||
}
|
|
||||||
Task::StoreMaintenance(task) => {
|
|
||||||
server.store_maintenance(task).await
|
|
||||||
}
|
|
||||||
Task::SpamFilterMaintenance(task) => {
|
|
||||||
Box::pin(server.spam_filter_maintenance(task)).await
|
|
||||||
}
|
|
||||||
Task::AcmeRenewal(task) => server.acme_management(task).await,
|
|
||||||
Task::DkimManagement(task_dkim_rotation) => {
|
|
||||||
server.dkim_management(task_dkim_rotation).await
|
|
||||||
}
|
|
||||||
Task::DnsManagement(task_dns_management) => {
|
|
||||||
server.dns_management(task_dns_management).await
|
|
||||||
}
|
|
||||||
Task::IndexDocument(_)
|
|
||||||
| Task::UnindexDocument(_)
|
|
||||||
| Task::IndexTrace(_) => unreachable!(),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Reads a claimed task. When it is gone or can't be read, the claim is
|
|
||||||
/// released: inbuxa: holding it would block the task, everywhere, until
|
|
||||||
/// the lock expired.
|
|
||||||
async fn fetch_task(server: &Server, job: TaskJob) -> Option<TaskDetails> {
|
|
||||||
match server
|
|
||||||
.store()
|
|
||||||
.get_value::<Task>(ValueKey::from(ValueClass::TaskQueue(TaskQueueClass::Task {
|
|
||||||
id: job.id,
|
|
||||||
})))
|
|
||||||
.await
|
|
||||||
{
|
|
||||||
Ok(Some(task)) => Some(TaskDetails { task, info: job }),
|
|
||||||
Ok(None) => {
|
|
||||||
trc::event!(
|
|
||||||
TaskManager(TaskManagerEvent::TaskIgnored),
|
|
||||||
Id = job.id,
|
|
||||||
Reason = "Task not found in store, likely already processed.",
|
|
||||||
);
|
|
||||||
server.remove_index_lock(job.id).await;
|
|
||||||
None
|
|
||||||
}
|
|
||||||
Err(err) => {
|
|
||||||
trc::error!(
|
|
||||||
err.id(job.id)
|
|
||||||
.details("Failed to retrieve task details.")
|
|
||||||
.caused_by(trc::location!())
|
|
||||||
);
|
|
||||||
server.remove_index_lock(job.id).await;
|
|
||||||
None
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// inbuxa: a task panicked: its locks are released so it runs again, here or
|
|
||||||
/// on another node, and the worker carries on.
|
|
||||||
async fn worker_failed(server: &Server, ids: &[u64], err: tokio::task::JoinError) {
|
|
||||||
trc::event!(
|
|
||||||
Server(trc::ServerEvent::ThreadError),
|
|
||||||
Details = "Task worker failed",
|
|
||||||
Reason = err.to_string(),
|
|
||||||
CausedBy = trc::location!()
|
|
||||||
);
|
|
||||||
for id in ids {
|
|
||||||
server.remove_index_lock(*id).await;
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -664,13 +632,6 @@ async fn update_tasks(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// inbuxa: how long to wait before trying again to claim a task another node
|
|
||||||
/// holds: a twelfth of the lock lifetime, so five minutes for the one-hour
|
|
||||||
/// lock, never more than that and never under a second.
|
|
||||||
pub(crate) fn claim_recheck_interval(lock_expiry: u64) -> u64 {
|
|
||||||
(lock_expiry / 12).clamp(1, CLAIM_RECHECK_INTERVAL)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn perpetual_retry_time(typ: TaskType, attempt: u64) -> Option<u64> {
|
pub fn perpetual_retry_time(typ: TaskType, attempt: u64) -> Option<u64> {
|
||||||
matches!(
|
matches!(
|
||||||
typ,
|
typ,
|
||||||
@@ -743,3 +704,114 @@ impl TaskResult {
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// inbuxa: a task queue row whose task type could not be read.
|
||||||
|
struct UnreadableDueRow {
|
||||||
|
due: u64,
|
||||||
|
id: u64,
|
||||||
|
value: Vec<u8>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// inbuxa: logs each unreadable queue row and repairs it from the task it
|
||||||
|
/// schedules. The task row says what the task is, so the queue row is
|
||||||
|
/// rewritten with that task's type; a row with no task behind it is removed.
|
||||||
|
///
|
||||||
|
/// Rescheduling an internal DMARC or TLS report wrote the report's object
|
||||||
|
/// type into the queue row instead of the task type. Such a row is the time
|
||||||
|
/// an administrator chose, so the task is moved to it as the reschedule
|
||||||
|
/// meant to do: the task row takes that due, and a queue row left at the
|
||||||
|
/// task's previous due is removed. Returns whether any row was repaired.
|
||||||
|
async fn repair_due_rows(server: &Server, rows: Vec<UnreadableDueRow>) -> bool {
|
||||||
|
let mut repaired = false;
|
||||||
|
for row in rows {
|
||||||
|
let UnreadableDueRow { due, id, value } = row;
|
||||||
|
trc::error!(
|
||||||
|
trc::StoreEvent::DataCorruption
|
||||||
|
.into_err()
|
||||||
|
.id(id)
|
||||||
|
.ctx(trc::Key::Due, trc::Value::Timestamp(due))
|
||||||
|
.ctx(
|
||||||
|
trc::Key::Key,
|
||||||
|
[due.to_be_bytes(), id.to_be_bytes()].concat()
|
||||||
|
)
|
||||||
|
.ctx(trc::Key::Value, value.clone())
|
||||||
|
.details("Unreadable task queue row skipped")
|
||||||
|
.caused_by(trc::location!())
|
||||||
|
);
|
||||||
|
|
||||||
|
let task_key = ValueClass::TaskQueue(TaskQueueClass::Task { id });
|
||||||
|
let due_key = ValueClass::TaskQueue(TaskQueueClass::Due { id, due });
|
||||||
|
let task = match server
|
||||||
|
.store()
|
||||||
|
.get_value::<Task>(ValueKey::from(task_key.clone()))
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
Ok(task) => task,
|
||||||
|
Err(err) => {
|
||||||
|
trc::error!(
|
||||||
|
err.id(id)
|
||||||
|
.details("Failed to read the task of an unreadable queue row.")
|
||||||
|
.caused_by(trc::location!())
|
||||||
|
);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
let mut batch = BatchBuilder::new();
|
||||||
|
let action = if let Some(mut task) = task {
|
||||||
|
let task_type = task.object_type();
|
||||||
|
batch.assert_value(task_key.clone(), AssertValue::Some);
|
||||||
|
if rescheduled_report_type(&value) == Some(task_type) {
|
||||||
|
let old_due = task.due_timestamp();
|
||||||
|
if old_due != due {
|
||||||
|
batch.clear(ValueClass::TaskQueue(TaskQueueClass::Due {
|
||||||
|
id,
|
||||||
|
due: old_due,
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
task.set_status(TaskStatus::at(due as i64));
|
||||||
|
}
|
||||||
|
batch
|
||||||
|
.set(due_key, task_type.to_id().serialize())
|
||||||
|
.set(task_key, task.to_pickled_vec());
|
||||||
|
"Rewrote the queue row from its task."
|
||||||
|
} else {
|
||||||
|
batch.clear(due_key);
|
||||||
|
"Removed a queue row with no task."
|
||||||
|
};
|
||||||
|
|
||||||
|
match server.store().write(batch.build_all()).await {
|
||||||
|
Ok(_) => {
|
||||||
|
repaired = true;
|
||||||
|
trc::event!(
|
||||||
|
TaskManager(TaskManagerEvent::TaskIgnored),
|
||||||
|
Id = id,
|
||||||
|
Due = trc::Value::Timestamp(due),
|
||||||
|
Reason = action,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
Err(err) if err.matches(trc::EventType::Store(trc::StoreEvent::AssertValueFailed)) => {
|
||||||
|
// The task went away meanwhile; the next scan looks again
|
||||||
|
}
|
||||||
|
Err(err) => {
|
||||||
|
trc::error!(
|
||||||
|
err.id(id)
|
||||||
|
.details("Failed to repair an unreadable queue row.")
|
||||||
|
.caused_by(trc::location!())
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
repaired
|
||||||
|
}
|
||||||
|
|
||||||
|
/// inbuxa: the task type a report reschedule meant, when a queue row holds
|
||||||
|
/// an internal report's object type (the value that reschedule wrote).
|
||||||
|
fn rescheduled_report_type(value: &[u8]) -> Option<TaskType> {
|
||||||
|
let id = u16::from_be_bytes(value.get(..2)?.try_into().ok()?);
|
||||||
|
match ObjectType::from_id(id)? {
|
||||||
|
ObjectType::DmarcInternalReport => Some(TaskType::DmarcReport),
|
||||||
|
ObjectType::TlsInternalReport => Some(TaskType::TlsReport),
|
||||||
|
_ => None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -35,9 +35,7 @@ pub mod scheduler;
|
|||||||
pub mod spam_classifier;
|
pub mod spam_classifier;
|
||||||
|
|
||||||
const QUEUE_REFRESH_INTERVAL: u64 = 60 * 5; // 5 minutes
|
const QUEUE_REFRESH_INTERVAL: u64 = 60 * 5; // 5 minutes
|
||||||
// inbuxa: the lock lifetime (one hour) lives in common::ipc::TaskLocks, per
|
const DEFAULT_LOCK_EXPIRY: u64 = 60 * 60; // 1 hour
|
||||||
// server, so a graceful stop can release the locks and the tests can shorten it
|
|
||||||
const CLAIM_RECHECK_INTERVAL: u64 = 60 * 5; // 5 minutes
|
|
||||||
|
|
||||||
pub(crate) struct TaskManagerIpc {
|
pub(crate) struct TaskManagerIpc {
|
||||||
txs: [mpsc::Sender<TaskJob>; TaskType::COUNT],
|
txs: [mpsc::Sender<TaskJob>; TaskType::COUNT],
|
||||||
|
|||||||
@@ -2,6 +2,8 @@
|
|||||||
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <[email protected]>
|
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <[email protected]>
|
||||||
*
|
*
|
||||||
* SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL
|
* SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL
|
||||||
|
*
|
||||||
|
* Modified by Coffey Labs in 2026 for INBUXA.
|
||||||
*/
|
*/
|
||||||
|
|
||||||
use registry::{
|
use registry::{
|
||||||
@@ -40,35 +42,49 @@ pub trait InternalReportIndex: ObjectImpl {
|
|||||||
|
|
||||||
fn primary_key(&self) -> ValueClass;
|
fn primary_key(&self) -> ValueClass;
|
||||||
|
|
||||||
|
/// Moves the report's delivery, and its queued task, to `at`.
|
||||||
|
///
|
||||||
|
/// inbuxa: the new queue row carries the task's type, as
|
||||||
|
/// `schedule_task_with_id` writes it, and the task row gets the new due
|
||||||
|
/// too. `queued` is the task as stored: its due, not the report's
|
||||||
|
/// `deliverAt`, is the queue row that exists (they differ once the task
|
||||||
|
/// has been retried).
|
||||||
fn reschedule_ops(
|
fn reschedule_ops(
|
||||||
&mut self,
|
&mut self,
|
||||||
batch: &mut BatchBuilder,
|
batch: &mut BatchBuilder,
|
||||||
item_id: u64,
|
item_id: u64,
|
||||||
revision: u64,
|
revision: u64,
|
||||||
at: UTCDateTime,
|
at: UTCDateTime,
|
||||||
|
queued: Option<&Task>,
|
||||||
) {
|
) {
|
||||||
let current_deliver_at = self.deliver_at();
|
let current_deliver_at = self.deliver_at();
|
||||||
|
let current_due = current_deliver_at.timestamp() as u64;
|
||||||
|
let queued_due = queued.map_or(current_due, |task| task.due_timestamp());
|
||||||
|
let new_due = at.timestamp() as u64;
|
||||||
|
|
||||||
if current_deliver_at != at {
|
if current_deliver_at != at || queued_due != new_due {
|
||||||
let object = Self::OBJECT;
|
let object = Self::OBJECT;
|
||||||
let object_id = object.to_id();
|
let object_id = object.to_id();
|
||||||
let key = ValueClass::Registry(RegistryClass::Item { object_id, item_id });
|
let key = ValueClass::Registry(RegistryClass::Item { object_id, item_id });
|
||||||
|
|
||||||
self.set_deliver_at(at);
|
self.set_deliver_at(at);
|
||||||
|
|
||||||
batch
|
batch.assert_value(key.clone(), AssertValue::Hash(revision));
|
||||||
.assert_value(key.clone(), AssertValue::Hash(revision))
|
if queued_due != new_due {
|
||||||
.clear(ValueClass::TaskQueue(TaskQueueClass::Due {
|
batch.clear(ValueClass::TaskQueue(TaskQueueClass::Due {
|
||||||
id: item_id,
|
id: item_id,
|
||||||
due: current_deliver_at.timestamp() as u64,
|
due: queued_due,
|
||||||
}))
|
}));
|
||||||
.set(
|
}
|
||||||
ValueClass::TaskQueue(TaskQueueClass::Due {
|
// A row an earlier reschedule left at the report's deliverAt
|
||||||
id: item_id,
|
if current_due != new_due && current_due != queued_due {
|
||||||
due: at.timestamp() as u64,
|
batch.clear(ValueClass::TaskQueue(TaskQueueClass::Due {
|
||||||
}),
|
id: item_id,
|
||||||
object_id.serialize(),
|
due: current_due,
|
||||||
)
|
}));
|
||||||
|
}
|
||||||
|
batch
|
||||||
|
.schedule_task_with_id(item_id, self.task(item_id))
|
||||||
.set(key, self.to_pickled_vec());
|
.set(key, self.to_pickled_vec());
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -10,9 +10,8 @@
|
|||||||
|
|
||||||
// inbuxa: 637 to 641 are the fork's SCIM events (SCIM-54); 642 is
|
// inbuxa: 637 to 641 are the fork's SCIM events (SCIM-54); 642 is
|
||||||
// auth.legacy-protocol-refused (legacy-protocols LP-6); 643 is
|
// auth.legacy-protocol-refused (legacy-protocols LP-6); 643 is
|
||||||
// security.legacy-protocols-changed (LP-8); 644 to 646 are the cluster
|
// security.legacy-protocols-changed (LP-8)
|
||||||
// coordinator's connection events
|
pub const TOTAL_EVENT_COUNT: usize = 644;
|
||||||
pub const TOTAL_EVENT_COUNT: usize = 647;
|
|
||||||
pub const TOTAL_METRIC_COUNT: usize = 369;
|
pub const TOTAL_METRIC_COUNT: usize = 369;
|
||||||
|
|
||||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
|
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
|
||||||
@@ -151,10 +150,6 @@ pub enum ClusterEvent {
|
|||||||
MessageSkipped = 47,
|
MessageSkipped = 47,
|
||||||
MessageInvalid = 49,
|
MessageInvalid = 49,
|
||||||
NodeIdRenewed = 275,
|
NodeIdRenewed = 275,
|
||||||
// inbuxa: the coordinator's connection
|
|
||||||
CoordinatorConnected = 644,
|
|
||||||
CoordinatorDisconnected = 645,
|
|
||||||
CoordinatorError = 646,
|
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
|
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
|
||||||
|
|||||||
@@ -81,10 +81,6 @@ impl EventType {
|
|||||||
b"cluster.message-skipped" => EventType::Cluster(ClusterEvent::MessageSkipped),
|
b"cluster.message-skipped" => EventType::Cluster(ClusterEvent::MessageSkipped),
|
||||||
b"cluster.message-invalid" => EventType::Cluster(ClusterEvent::MessageInvalid),
|
b"cluster.message-invalid" => EventType::Cluster(ClusterEvent::MessageInvalid),
|
||||||
b"cluster.node-id-renewed" => EventType::Cluster(ClusterEvent::NodeIdRenewed),
|
b"cluster.node-id-renewed" => EventType::Cluster(ClusterEvent::NodeIdRenewed),
|
||||||
// inbuxa: coordinator connection
|
|
||||||
b"cluster.coordinator-connected" => EventType::Cluster(ClusterEvent::CoordinatorConnected),
|
|
||||||
b"cluster.coordinator-disconnected" => EventType::Cluster(ClusterEvent::CoordinatorDisconnected),
|
|
||||||
b"cluster.coordinator-error" => EventType::Cluster(ClusterEvent::CoordinatorError),
|
|
||||||
b"dane.authentication-success" => EventType::Dane(DaneEvent::AuthenticationSuccess),
|
b"dane.authentication-success" => EventType::Dane(DaneEvent::AuthenticationSuccess),
|
||||||
b"dane.authentication-failure" => EventType::Dane(DaneEvent::AuthenticationFailure),
|
b"dane.authentication-failure" => EventType::Dane(DaneEvent::AuthenticationFailure),
|
||||||
b"dane.no-certificates-found" => EventType::Dane(DaneEvent::NoCertificatesFound),
|
b"dane.no-certificates-found" => EventType::Dane(DaneEvent::NoCertificatesFound),
|
||||||
@@ -746,14 +742,6 @@ impl EventType {
|
|||||||
EventType::Cluster(ClusterEvent::MessageSkipped) => "cluster.message-skipped",
|
EventType::Cluster(ClusterEvent::MessageSkipped) => "cluster.message-skipped",
|
||||||
EventType::Cluster(ClusterEvent::MessageInvalid) => "cluster.message-invalid",
|
EventType::Cluster(ClusterEvent::MessageInvalid) => "cluster.message-invalid",
|
||||||
EventType::Cluster(ClusterEvent::NodeIdRenewed) => "cluster.node-id-renewed",
|
EventType::Cluster(ClusterEvent::NodeIdRenewed) => "cluster.node-id-renewed",
|
||||||
// inbuxa: coordinator connection
|
|
||||||
EventType::Cluster(ClusterEvent::CoordinatorConnected) => {
|
|
||||||
"cluster.coordinator-connected"
|
|
||||||
}
|
|
||||||
EventType::Cluster(ClusterEvent::CoordinatorDisconnected) => {
|
|
||||||
"cluster.coordinator-disconnected"
|
|
||||||
}
|
|
||||||
EventType::Cluster(ClusterEvent::CoordinatorError) => "cluster.coordinator-error",
|
|
||||||
EventType::Dane(DaneEvent::AuthenticationSuccess) => "dane.authentication-success",
|
EventType::Dane(DaneEvent::AuthenticationSuccess) => "dane.authentication-success",
|
||||||
EventType::Dane(DaneEvent::AuthenticationFailure) => "dane.authentication-failure",
|
EventType::Dane(DaneEvent::AuthenticationFailure) => "dane.authentication-failure",
|
||||||
EventType::Dane(DaneEvent::NoCertificatesFound) => "dane.no-certificates-found",
|
EventType::Dane(DaneEvent::NoCertificatesFound) => "dane.no-certificates-found",
|
||||||
@@ -1536,10 +1524,6 @@ impl EventType {
|
|||||||
EventType::Cluster(ClusterEvent::MessageSkipped) => 47,
|
EventType::Cluster(ClusterEvent::MessageSkipped) => 47,
|
||||||
EventType::Cluster(ClusterEvent::MessageInvalid) => 49,
|
EventType::Cluster(ClusterEvent::MessageInvalid) => 49,
|
||||||
EventType::Cluster(ClusterEvent::NodeIdRenewed) => 275,
|
EventType::Cluster(ClusterEvent::NodeIdRenewed) => 275,
|
||||||
// inbuxa: coordinator connection
|
|
||||||
EventType::Cluster(ClusterEvent::CoordinatorConnected) => 644,
|
|
||||||
EventType::Cluster(ClusterEvent::CoordinatorDisconnected) => 645,
|
|
||||||
EventType::Cluster(ClusterEvent::CoordinatorError) => 646,
|
|
||||||
EventType::Dane(DaneEvent::AuthenticationSuccess) => 67,
|
EventType::Dane(DaneEvent::AuthenticationSuccess) => 67,
|
||||||
EventType::Dane(DaneEvent::AuthenticationFailure) => 66,
|
EventType::Dane(DaneEvent::AuthenticationFailure) => 66,
|
||||||
EventType::Dane(DaneEvent::NoCertificatesFound) => 69,
|
EventType::Dane(DaneEvent::NoCertificatesFound) => 69,
|
||||||
@@ -2192,10 +2176,6 @@ impl EventType {
|
|||||||
47 => Some(EventType::Cluster(ClusterEvent::MessageSkipped)),
|
47 => Some(EventType::Cluster(ClusterEvent::MessageSkipped)),
|
||||||
49 => Some(EventType::Cluster(ClusterEvent::MessageInvalid)),
|
49 => Some(EventType::Cluster(ClusterEvent::MessageInvalid)),
|
||||||
275 => Some(EventType::Cluster(ClusterEvent::NodeIdRenewed)),
|
275 => Some(EventType::Cluster(ClusterEvent::NodeIdRenewed)),
|
||||||
// inbuxa: coordinator connection
|
|
||||||
644 => Some(EventType::Cluster(ClusterEvent::CoordinatorConnected)),
|
|
||||||
645 => Some(EventType::Cluster(ClusterEvent::CoordinatorDisconnected)),
|
|
||||||
646 => Some(EventType::Cluster(ClusterEvent::CoordinatorError)),
|
|
||||||
67 => Some(EventType::Dane(DaneEvent::AuthenticationSuccess)),
|
67 => Some(EventType::Dane(DaneEvent::AuthenticationSuccess)),
|
||||||
66 => Some(EventType::Dane(DaneEvent::AuthenticationFailure)),
|
66 => Some(EventType::Dane(DaneEvent::AuthenticationFailure)),
|
||||||
69 => Some(EventType::Dane(DaneEvent::NoCertificatesFound)),
|
69 => Some(EventType::Dane(DaneEvent::NoCertificatesFound)),
|
||||||
@@ -3134,10 +3114,6 @@ impl EventType {
|
|||||||
EventType::Auth(AuthEvent::TooManyAttempts) => Level::Warn,
|
EventType::Auth(AuthEvent::TooManyAttempts) => Level::Warn,
|
||||||
EventType::Calendar(CalendarEvent::AlarmFailed) => Level::Warn,
|
EventType::Calendar(CalendarEvent::AlarmFailed) => Level::Warn,
|
||||||
EventType::Cluster(ClusterEvent::SubscriberDisconnected) => Level::Warn,
|
EventType::Cluster(ClusterEvent::SubscriberDisconnected) => Level::Warn,
|
||||||
// inbuxa: coordinator connection
|
|
||||||
EventType::Cluster(ClusterEvent::CoordinatorConnected) => Level::Info,
|
|
||||||
EventType::Cluster(ClusterEvent::CoordinatorDisconnected) => Level::Warn,
|
|
||||||
EventType::Cluster(ClusterEvent::CoordinatorError) => Level::Warn,
|
|
||||||
EventType::Delivery(DeliveryEvent::MissingOutboundHostname) => Level::Warn,
|
EventType::Delivery(DeliveryEvent::MissingOutboundHostname) => Level::Warn,
|
||||||
EventType::Delivery(DeliveryEvent::ConcurrencyLimitExceeded) => Level::Warn,
|
EventType::Delivery(DeliveryEvent::ConcurrencyLimitExceeded) => Level::Warn,
|
||||||
EventType::Delivery(DeliveryEvent::RateLimitExceeded) => Level::Warn,
|
EventType::Delivery(DeliveryEvent::RateLimitExceeded) => Level::Warn,
|
||||||
@@ -3268,10 +3244,6 @@ impl EventType {
|
|||||||
EventType::Cluster(ClusterEvent::MessageSkipped) => "PubSub message skipped",
|
EventType::Cluster(ClusterEvent::MessageSkipped) => "PubSub message skipped",
|
||||||
EventType::Cluster(ClusterEvent::MessageInvalid) => "Invalid PubSub message",
|
EventType::Cluster(ClusterEvent::MessageInvalid) => "Invalid PubSub message",
|
||||||
EventType::Cluster(ClusterEvent::NodeIdRenewed) => "Node ID renewed",
|
EventType::Cluster(ClusterEvent::NodeIdRenewed) => "Node ID renewed",
|
||||||
// inbuxa: coordinator connection
|
|
||||||
EventType::Cluster(ClusterEvent::CoordinatorConnected) => "Coordinator connected",
|
|
||||||
EventType::Cluster(ClusterEvent::CoordinatorDisconnected) => "Coordinator unavailable",
|
|
||||||
EventType::Cluster(ClusterEvent::CoordinatorError) => "Coordinator error",
|
|
||||||
EventType::Dane(DaneEvent::AuthenticationSuccess) => "DANE authentication successful",
|
EventType::Dane(DaneEvent::AuthenticationSuccess) => "DANE authentication successful",
|
||||||
EventType::Dane(DaneEvent::AuthenticationFailure) => "DANE authentication failed",
|
EventType::Dane(DaneEvent::AuthenticationFailure) => "DANE authentication failed",
|
||||||
EventType::Dane(DaneEvent::NoCertificatesFound) => "No certificates found for DANE",
|
EventType::Dane(DaneEvent::NoCertificatesFound) => "No certificates found for DANE",
|
||||||
@@ -4350,10 +4322,6 @@ impl EventType {
|
|||||||
EventType::Cluster(ClusterEvent::MessageSkipped),
|
EventType::Cluster(ClusterEvent::MessageSkipped),
|
||||||
EventType::Cluster(ClusterEvent::MessageInvalid),
|
EventType::Cluster(ClusterEvent::MessageInvalid),
|
||||||
EventType::Cluster(ClusterEvent::NodeIdRenewed),
|
EventType::Cluster(ClusterEvent::NodeIdRenewed),
|
||||||
// inbuxa: coordinator connection
|
|
||||||
EventType::Cluster(ClusterEvent::CoordinatorConnected),
|
|
||||||
EventType::Cluster(ClusterEvent::CoordinatorDisconnected),
|
|
||||||
EventType::Cluster(ClusterEvent::CoordinatorError),
|
|
||||||
EventType::Dane(DaneEvent::AuthenticationSuccess),
|
EventType::Dane(DaneEvent::AuthenticationSuccess),
|
||||||
EventType::Dane(DaneEvent::AuthenticationFailure),
|
EventType::Dane(DaneEvent::AuthenticationFailure),
|
||||||
EventType::Dane(DaneEvent::NoCertificatesFound),
|
EventType::Dane(DaneEvent::NoCertificatesFound),
|
||||||
|
|||||||
@@ -81,7 +81,7 @@ fn legacy_setting(name: &str, is_set: impl Fn(&str) -> bool) -> Option<String> {
|
|||||||
#[macro_export]
|
#[macro_export]
|
||||||
macro_rules! brand_version {
|
macro_rules! brand_version {
|
||||||
() => {
|
() => {
|
||||||
"2026.9.24.3"
|
"2026.9.24.4"
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -212,15 +212,10 @@ unchanged.
|
|||||||
- **MON-16.** With `indexTelemetry` on, storing a trace schedules an
|
- **MON-16.** With `indexTelemetry` on, storing a trace schedules an
|
||||||
`IndexTrace` task. The task builds one document for `SearchIndex::Tracing`
|
`IndexTrace` task. The task builds one document for `SearchIndex::Tracing`
|
||||||
with the fields named in `indexTracingFields`:
|
with the fields named in `indexTracingFields`:
|
||||||
- `eventType`: the trace's opening event, as its numeric id;
|
- `eventType`: every event type in the trace;
|
||||||
- `queueId`: the first `queueId` value, as an integer;
|
- `queueId`: every `queueId` value;
|
||||||
- `keywords`: every address in `from` and `to`, each address's domain, every
|
- `keywords`: every address in `from` and `to`, each address's domain, every
|
||||||
`domain`, `hostname`, `remoteIp`, `messageId` and `accountName` value,
|
`domain`, `hostname`, `remoteIp`, `messageId` and `accountName` value.
|
||||||
and every `queueId` value.
|
|
||||||
The event type and queue id are single integer columns on every search
|
|
||||||
backend (BIGINT on PostgreSQL and MySQL), so the `queueId` filter matches
|
|
||||||
the column or any queue id in the keywords, and a session that queued
|
|
||||||
several messages is found by each of them.
|
|
||||||
So searching `example.org` finds every trace to or from that domain, as the
|
So searching `example.org` finds every trace to or from that domain, as the
|
||||||
upstream suite expects. With `indexTelemetry` off nothing is indexed, and
|
upstream suite expects. With `indexTelemetry` off nothing is indexed, and
|
||||||
the `text` and `queueId` filters are refused (see "Interfaces").
|
the `text` and `queueId` filters are refused (see "Interfaces").
|
||||||
|
|||||||
Binary file not shown.
@@ -1 +1 @@
|
|||||||
XFI3xuKC_rH1KZyaVBF0uTIiRDXRqyYboijquiGz2eg
|
VbnFuwCOTBh0s2T-NuRhb2JaJr8Jl5s3LgXv4Pv2sTg
|
||||||
@@ -1,145 +0,0 @@
|
|||||||
/*
|
|
||||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
|
||||||
*
|
|
||||||
* SPDX-License-Identifier: AGPL-3.0-only
|
|
||||||
*/
|
|
||||||
|
|
||||||
//! A node that starts while its NATS coordinator is down joins the cluster
|
|
||||||
//! once NATS comes up, without a restart, and reports the coordinator's
|
|
||||||
//! connection on `/healthz/cluster` as it goes and comes back.
|
|
||||||
|
|
||||||
use crate::utils::server::TestServerBuilder;
|
|
||||||
use coordinator::Coordinator;
|
|
||||||
use registry::{
|
|
||||||
schema::{
|
|
||||||
enums::NetworkListenerProtocol,
|
|
||||||
structs::{Coordinator as CoordinatorSetting, NatsCoordinator},
|
|
||||||
},
|
|
||||||
types::map::Map,
|
|
||||||
};
|
|
||||||
use serde_json::{Value, json};
|
|
||||||
use std::time::{Duration, Instant};
|
|
||||||
use testcontainers::{
|
|
||||||
GenericImage, ImageExt, core::IntoContainerPort, core::WaitFor, runners::AsyncRunner,
|
|
||||||
};
|
|
||||||
|
|
||||||
const HTTP_PORT: u16 = 11_310;
|
|
||||||
const TOPIC: &str = "inbuxa-coordinator-test";
|
|
||||||
|
|
||||||
#[tokio::test(flavor = "multi_thread")]
|
|
||||||
pub async fn coordinator_reconnect_tests() {
|
|
||||||
println!("Running coordinator reconnect tests...");
|
|
||||||
|
|
||||||
// A port with no NATS server behind it, yet
|
|
||||||
let nats_port = std::net::TcpListener::bind("127.0.0.1:0")
|
|
||||||
.unwrap()
|
|
||||||
.local_addr()
|
|
||||||
.unwrap()
|
|
||||||
.port();
|
|
||||||
let config = NatsCoordinator {
|
|
||||||
addresses: Map::new(vec![format!("127.0.0.1:{nats_port}")]),
|
|
||||||
use_tls: false,
|
|
||||||
timeout_connection: 1_000u64.into(),
|
|
||||||
..Default::default()
|
|
||||||
};
|
|
||||||
|
|
||||||
// 1. The node starts, without a build error, while NATS is down, and
|
|
||||||
// says so
|
|
||||||
let test = TestServerBuilder::new("coordinator_reconnect_tests")
|
|
||||||
.await
|
|
||||||
.with_object(CoordinatorSetting::Nats(config.clone()))
|
|
||||||
.await
|
|
||||||
.with_listener(NetworkListenerProtocol::Http, "http", HTTP_PORT, true)
|
|
||||||
.await
|
|
||||||
.build()
|
|
||||||
.await;
|
|
||||||
let coordinator = test.server.core.storage.coordinator.clone();
|
|
||||||
assert!(
|
|
||||||
coordinator.is_enabled(),
|
|
||||||
"a coordinator, though not connected"
|
|
||||||
);
|
|
||||||
assert_eq!(coordinator.is_connected(), Some(false));
|
|
||||||
assert_eq!(
|
|
||||||
cluster_health().await,
|
|
||||||
(503, json!({"coordinator": "disconnected"}))
|
|
||||||
);
|
|
||||||
|
|
||||||
// A subscription made now, as the broadcast subscriber makes it at
|
|
||||||
// startup, has to work once NATS is up
|
|
||||||
let mut stream = coordinator.subscribe(TOPIC).await.unwrap();
|
|
||||||
|
|
||||||
// 2. NATS comes up: the node connects on its own
|
|
||||||
let nats = GenericImage::new("nats", "latest")
|
|
||||||
.with_wait_for(WaitFor::message_on_stderr("Server is ready"))
|
|
||||||
.with_mapped_port(nats_port, 4222.tcp())
|
|
||||||
.start()
|
|
||||||
.await
|
|
||||||
.expect("Failed to start NATS container");
|
|
||||||
wait_for_health(200, "connected").await;
|
|
||||||
let other_node = coordinator::backend::nats::NatsPubSub::open(config.clone())
|
|
||||||
.await
|
|
||||||
.unwrap();
|
|
||||||
wait_until_connected(&other_node).await;
|
|
||||||
round_trip(&other_node, &mut stream, b"after startup").await;
|
|
||||||
|
|
||||||
// 3. NATS goes away: the node reports it; and it comes back: the node
|
|
||||||
// reconnects and the same subscription carries on
|
|
||||||
nats.stop().await.unwrap();
|
|
||||||
wait_for_health(503, "disconnected").await;
|
|
||||||
nats.start().await.unwrap();
|
|
||||||
wait_for_health(200, "connected").await;
|
|
||||||
wait_until_connected(&other_node).await;
|
|
||||||
round_trip(&other_node, &mut stream, b"after reconnect").await;
|
|
||||||
|
|
||||||
drop(nats);
|
|
||||||
if test.is_reset() {
|
|
||||||
test.temp_dir.delete();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn cluster_health() -> (u16, Value) {
|
|
||||||
let response = reqwest::Client::builder()
|
|
||||||
.danger_accept_invalid_certs(true)
|
|
||||||
.timeout(Duration::from_secs(5))
|
|
||||||
.build()
|
|
||||||
.unwrap()
|
|
||||||
.get(format!("https://127.0.0.1:{HTTP_PORT}/healthz/cluster"))
|
|
||||||
.send()
|
|
||||||
.await
|
|
||||||
.unwrap();
|
|
||||||
let status = response.status().as_u16();
|
|
||||||
(status, response.json().await.unwrap())
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn wait_for_health(status: u16, state: &str) {
|
|
||||||
let started = Instant::now();
|
|
||||||
loop {
|
|
||||||
let health = cluster_health().await;
|
|
||||||
if health == (status, json!({"coordinator": state})) {
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
assert!(
|
|
||||||
started.elapsed() < Duration::from_secs(30),
|
|
||||||
"expected {status} {state}, still {health:?}"
|
|
||||||
);
|
|
||||||
tokio::time::sleep(Duration::from_millis(250)).await;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn wait_until_connected(coordinator: &Coordinator) {
|
|
||||||
let started = Instant::now();
|
|
||||||
while coordinator.is_connected() != Some(true) {
|
|
||||||
assert!(started.elapsed() < Duration::from_secs(30), "not connected");
|
|
||||||
tokio::time::sleep(Duration::from_millis(100)).await;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Another node publishes; this one's subscription receives it.
|
|
||||||
async fn round_trip(from: &Coordinator, stream: &mut coordinator::PubSubStream, payload: &[u8]) {
|
|
||||||
from.publish(TOPIC, payload.to_vec()).await.unwrap();
|
|
||||||
let message = tokio::time::timeout(Duration::from_secs(10), stream.next())
|
|
||||||
.await
|
|
||||||
.expect("no message within 10 seconds")
|
|
||||||
.expect("subscription ended");
|
|
||||||
assert_eq!(message.payload(), payload);
|
|
||||||
}
|
|
||||||
@@ -2,11 +2,7 @@
|
|||||||
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <[email protected]>
|
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <[email protected]>
|
||||||
*
|
*
|
||||||
* SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL
|
* SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL
|
||||||
*
|
|
||||||
* Modified by Coffey Labs in 2026 for INBUXA.
|
|
||||||
*/
|
*/
|
||||||
|
|
||||||
pub mod broadcast;
|
pub mod broadcast;
|
||||||
#[cfg(feature = "nats")]
|
|
||||||
pub mod coordinator; // inbuxa: coordinator reconnects
|
|
||||||
pub mod stress;
|
pub mod stress;
|
||||||
|
|||||||
@@ -2,9 +2,12 @@
|
|||||||
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <[email protected]>
|
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <[email protected]>
|
||||||
*
|
*
|
||||||
* SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL
|
* SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL
|
||||||
|
*
|
||||||
|
* Modified by Coffey Labs in 2026 for INBUXA.
|
||||||
*/
|
*/
|
||||||
|
|
||||||
pub mod analyze;
|
pub mod analyze;
|
||||||
pub mod dmarc;
|
pub mod dmarc;
|
||||||
|
pub mod reschedule; // inbuxa: report reschedules and unreadable queue rows
|
||||||
pub mod scheduler;
|
pub mod scheduler;
|
||||||
pub mod tls;
|
pub mod tls;
|
||||||
|
|||||||
@@ -0,0 +1,370 @@
|
|||||||
|
/*
|
||||||
|
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||||
|
*
|
||||||
|
* SPDX-License-Identifier: AGPL-3.0-only
|
||||||
|
*/
|
||||||
|
|
||||||
|
//! Rescheduling an internal DMARC or TLS report over JMAP moves its task: the
|
||||||
|
//! task runs at the new time, x:Task/get shows the new due, and tasks due
|
||||||
|
//! after it still run. A task queue row whose type can't be read is logged
|
||||||
|
//! and repaired rather than stopping every task due after it, including the
|
||||||
|
//! rows an earlier reschedule wrote with the report's object type.
|
||||||
|
|
||||||
|
use crate::utils::server::{TestServer, TestServerBuilder};
|
||||||
|
use common::{
|
||||||
|
Server,
|
||||||
|
config::smtp::report::AggregateFrequency,
|
||||||
|
ipc::{DmarcEvent, PolicyType, TlsEvent},
|
||||||
|
};
|
||||||
|
use mail_auth::{
|
||||||
|
common::parse::TxtRecordParser,
|
||||||
|
dmarc::Dmarc,
|
||||||
|
mta_sts::TlsRpt,
|
||||||
|
report::{ActionDisposition, DmarcResult, Record},
|
||||||
|
};
|
||||||
|
use registry::{
|
||||||
|
schema::{
|
||||||
|
enums::{TaskStoreMaintenanceType, TaskType},
|
||||||
|
prelude::{ObjectType, Property},
|
||||||
|
structs::{
|
||||||
|
DmarcInternalReport, DmarcReportSettings, Expression, Task, TaskStatus,
|
||||||
|
TaskStoreMaintenance, TlsInternalReport, TlsReportSettings,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
types::{EnumImpl, ObjectImpl, datetime::UTCDateTime},
|
||||||
|
};
|
||||||
|
use serde_json::json;
|
||||||
|
use smtp::reporting::{index::InternalReportIndex, send::MtaReportSend};
|
||||||
|
use std::{
|
||||||
|
sync::Arc,
|
||||||
|
time::{Duration, Instant},
|
||||||
|
};
|
||||||
|
use store::{
|
||||||
|
SerializeInfallible, ValueKey,
|
||||||
|
write::{BatchBuilder, RegistryClass, TaskQueueClass, ValueClass, now},
|
||||||
|
};
|
||||||
|
use types::id::Id;
|
||||||
|
use utils::snowflake::SnowflakeIdGenerator;
|
||||||
|
|
||||||
|
#[tokio::test(flavor = "multi_thread")]
|
||||||
|
#[serial_test::serial]
|
||||||
|
async fn report_reschedule() {
|
||||||
|
let mut test = TestServerBuilder::new("smtp_report_reschedule")
|
||||||
|
.await
|
||||||
|
.with_http_listener(19057)
|
||||||
|
.await
|
||||||
|
.capture_queue()
|
||||||
|
.build()
|
||||||
|
.await;
|
||||||
|
|
||||||
|
let admin = test.account("admin");
|
||||||
|
admin
|
||||||
|
.registry_create_object(TlsReportSettings {
|
||||||
|
max_report_size: Expression {
|
||||||
|
else_: "1024".into(),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
..Default::default()
|
||||||
|
})
|
||||||
|
.await;
|
||||||
|
admin
|
||||||
|
.registry_create_object(DmarcReportSettings {
|
||||||
|
aggregate_max_report_size: Expression {
|
||||||
|
else_: "1024".into(),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
..Default::default()
|
||||||
|
})
|
||||||
|
.await;
|
||||||
|
admin.reload_settings().await;
|
||||||
|
test.reload_core();
|
||||||
|
test.expect_reload_settings().await;
|
||||||
|
let admin = test.account("admin");
|
||||||
|
|
||||||
|
// A daily DMARC and TLS report, due a day from now
|
||||||
|
schedule_dmarc(&test, "foobar.org").await;
|
||||||
|
schedule_tls(&test, "foobar.org").await;
|
||||||
|
let dmarc_id = wait_for_report::<DmarcInternalReport>(&test, "foobar.org").await;
|
||||||
|
let tls_id = wait_for_report::<TlsInternalReport>(&test, "foobar.org").await;
|
||||||
|
|
||||||
|
// Reschedule both to a few seconds from now, with a task due after them
|
||||||
|
let at = now() + 3;
|
||||||
|
let later = marker_task(&test.server, at + 3).await;
|
||||||
|
for (object, id, task_type) in [
|
||||||
|
(
|
||||||
|
ObjectType::DmarcInternalReport,
|
||||||
|
dmarc_id,
|
||||||
|
TaskType::DmarcReport,
|
||||||
|
),
|
||||||
|
(ObjectType::TlsInternalReport, tls_id, TaskType::TlsReport),
|
||||||
|
] {
|
||||||
|
admin
|
||||||
|
.registry_update_object(
|
||||||
|
object,
|
||||||
|
id,
|
||||||
|
json!({
|
||||||
|
Property::DeliverAt: UTCDateTime::from_timestamp(at as i64),
|
||||||
|
}),
|
||||||
|
)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
// x:Task/get shows the new due, and the queue row carries the task's
|
||||||
|
// type. Upstream wrote the report's object type there and left the
|
||||||
|
// task at its old due
|
||||||
|
let task = admin.registry_get::<Task>(id).await;
|
||||||
|
assert_eq!(task.object_type(), task_type);
|
||||||
|
assert_eq!(
|
||||||
|
task.due_timestamp(),
|
||||||
|
at,
|
||||||
|
"{object:?} task due not moved: {task:?}"
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
queue_row(&test.server, id.id(), at).await,
|
||||||
|
Some(task_type.to_id().serialize()),
|
||||||
|
"{object:?} queue row"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Both reports go out at the new time, and the later task still runs
|
||||||
|
wait_until_run(&test.server, &[dmarc_id.id(), tls_id.id(), later]).await;
|
||||||
|
assert!(now() >= at, "the reports went out before their new time");
|
||||||
|
assert!(
|
||||||
|
admin
|
||||||
|
.registry_get_all::<DmarcInternalReport>()
|
||||||
|
.await
|
||||||
|
.is_empty()
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
admin
|
||||||
|
.registry_get_all::<TlsInternalReport>()
|
||||||
|
.await
|
||||||
|
.is_empty()
|
||||||
|
);
|
||||||
|
|
||||||
|
// Rows an earlier reschedule may have left in a store: one with the
|
||||||
|
// report's object type and the task left at its old due, and one that
|
||||||
|
// is unreadable and has no task behind it. Neither may hold back a task
|
||||||
|
// due after them.
|
||||||
|
schedule_dmarc(&test, "foobar.net").await;
|
||||||
|
let dmarc_id = wait_for_report::<DmarcInternalReport>(&test, "foobar.net").await;
|
||||||
|
let at = now() + 2;
|
||||||
|
let old_due = old_style_reschedule(&test.server, dmarc_id.id(), at).await;
|
||||||
|
let orphan = SnowflakeIdGenerator::global_id().unwrap();
|
||||||
|
let mut batch = BatchBuilder::new();
|
||||||
|
batch.set(
|
||||||
|
ValueClass::TaskQueue(TaskQueueClass::Due {
|
||||||
|
id: orphan,
|
||||||
|
due: at,
|
||||||
|
}),
|
||||||
|
vec![0xff, 0xff],
|
||||||
|
);
|
||||||
|
test.server.store().write(batch.build_all()).await.unwrap();
|
||||||
|
let later = marker_task(&test.server, at + 2).await;
|
||||||
|
|
||||||
|
wait_until_run(&test.server, &[dmarc_id.id(), later]).await;
|
||||||
|
assert!(
|
||||||
|
admin
|
||||||
|
.registry_get_all::<DmarcInternalReport>()
|
||||||
|
.await
|
||||||
|
.is_empty()
|
||||||
|
);
|
||||||
|
assert_eq!(queue_row(&test.server, orphan, at).await, None);
|
||||||
|
assert_eq!(queue_row(&test.server, dmarc_id.id(), at).await, None);
|
||||||
|
assert_eq!(queue_row(&test.server, dmarc_id.id(), old_due).await, None);
|
||||||
|
|
||||||
|
// x:Task/query by type skips an unreadable row rather than failing
|
||||||
|
let mut batch = BatchBuilder::new();
|
||||||
|
let due = now() + 3600;
|
||||||
|
batch.set(
|
||||||
|
ValueClass::TaskQueue(TaskQueueClass::Due { id: orphan, due }),
|
||||||
|
vec![0xff, 0xff],
|
||||||
|
);
|
||||||
|
test.server.store().write(batch.build_all()).await.unwrap();
|
||||||
|
admin
|
||||||
|
.registry_query_ids(
|
||||||
|
ObjectType::Task,
|
||||||
|
vec![(Property::Type, TaskType::DmarcReport.as_str())],
|
||||||
|
Vec::<&str>::new(),
|
||||||
|
)
|
||||||
|
.await;
|
||||||
|
let mut batch = BatchBuilder::new();
|
||||||
|
batch.clear(ValueClass::TaskQueue(TaskQueueClass::Due {
|
||||||
|
id: orphan,
|
||||||
|
due,
|
||||||
|
}));
|
||||||
|
test.server.store().write(batch.build_all()).await.unwrap();
|
||||||
|
|
||||||
|
if test.is_reset() {
|
||||||
|
test.temp_dir.delete();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn schedule_dmarc(test: &TestServer, domain: &str) {
|
||||||
|
test.server
|
||||||
|
.schedule_report(DmarcEvent {
|
||||||
|
domain: domain.to_string(),
|
||||||
|
report_record: Record::new()
|
||||||
|
.with_source_ip("192.168.1.2".parse().unwrap())
|
||||||
|
.with_action_disposition(ActionDisposition::Pass)
|
||||||
|
.with_dmarc_dkim_result(DmarcResult::Pass)
|
||||||
|
.with_dmarc_spf_result(DmarcResult::Fail)
|
||||||
|
.with_envelope_from("[email protected]")
|
||||||
|
.with_envelope_to("[email protected]")
|
||||||
|
.with_header_from("[email protected]"),
|
||||||
|
dmarc_record: Arc::new(
|
||||||
|
Dmarc::parse(format!("v=DMARC1; p=reject; rua=mailto:reports@{domain}").as_bytes())
|
||||||
|
.unwrap(),
|
||||||
|
),
|
||||||
|
interval: AggregateFrequency::Daily,
|
||||||
|
span_id: 0,
|
||||||
|
})
|
||||||
|
.await;
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn schedule_tls(test: &TestServer, domain: &str) {
|
||||||
|
test.server
|
||||||
|
.schedule_report(TlsEvent {
|
||||||
|
domain: domain.to_string(),
|
||||||
|
policy: PolicyType::None,
|
||||||
|
failure: None,
|
||||||
|
tls_record: Arc::new(
|
||||||
|
TlsRpt::parse(format!("v=TLSRPTv1;rua=mailto:reports@{domain}").as_bytes())
|
||||||
|
.unwrap(),
|
||||||
|
),
|
||||||
|
interval: AggregateFrequency::Daily,
|
||||||
|
span_id: 0,
|
||||||
|
})
|
||||||
|
.await;
|
||||||
|
}
|
||||||
|
|
||||||
|
trait ReportDomain: ObjectImpl {
|
||||||
|
fn report_domain(&self) -> &str;
|
||||||
|
}
|
||||||
|
|
||||||
|
impl ReportDomain for DmarcInternalReport {
|
||||||
|
fn report_domain(&self) -> &str {
|
||||||
|
&self.domain
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl ReportDomain for TlsInternalReport {
|
||||||
|
fn report_domain(&self) -> &str {
|
||||||
|
&self.domain
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn wait_for_report<T: ReportDomain>(test: &TestServer, domain: &str) -> Id {
|
||||||
|
let admin = test.account("admin");
|
||||||
|
for _ in 0..100 {
|
||||||
|
if let Some((id, _)) = admin
|
||||||
|
.registry_get_all::<T>()
|
||||||
|
.await
|
||||||
|
.into_iter()
|
||||||
|
.find(|(_, report)| report.report_domain() == domain)
|
||||||
|
{
|
||||||
|
return id;
|
||||||
|
}
|
||||||
|
tokio::time::sleep(Duration::from_millis(100)).await;
|
||||||
|
}
|
||||||
|
panic!("No {} for {domain}", T::OBJECT.as_str());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A task that succeeds when it runs, due at `due`.
|
||||||
|
async fn marker_task(server: &Server, due: u64) -> u64 {
|
||||||
|
let id = SnowflakeIdGenerator::global_id().unwrap();
|
||||||
|
let mut batch = BatchBuilder::new();
|
||||||
|
batch.schedule_task_with_id(
|
||||||
|
id,
|
||||||
|
Task::StoreMaintenance(TaskStoreMaintenance {
|
||||||
|
maintenance_type: TaskStoreMaintenanceType::RemoveLockDav,
|
||||||
|
shard_index: Some(0),
|
||||||
|
status: TaskStatus::at(due as i64),
|
||||||
|
}),
|
||||||
|
);
|
||||||
|
server.store().write(batch.build_all()).await.unwrap();
|
||||||
|
server.notify_task_queue();
|
||||||
|
id
|
||||||
|
}
|
||||||
|
|
||||||
|
/// What the reschedule before this fix wrote: the report's object type in
|
||||||
|
/// the new queue row, and the task row left at its old due. Returns that
|
||||||
|
/// old due.
|
||||||
|
async fn old_style_reschedule(server: &Server, item_id: u64, at: u64) -> u64 {
|
||||||
|
let object_id = ObjectType::DmarcInternalReport.to_id();
|
||||||
|
let key = ValueClass::Registry(RegistryClass::Item { object_id, item_id });
|
||||||
|
let mut report = server
|
||||||
|
.store()
|
||||||
|
.get_value::<DmarcInternalReport>(ValueKey::from(key.clone()))
|
||||||
|
.await
|
||||||
|
.unwrap()
|
||||||
|
.unwrap();
|
||||||
|
let old_due = report.deliver_at().timestamp() as u64;
|
||||||
|
report.set_deliver_at(UTCDateTime::from_timestamp(at as i64));
|
||||||
|
let mut batch = BatchBuilder::new();
|
||||||
|
batch
|
||||||
|
.clear(ValueClass::TaskQueue(TaskQueueClass::Due {
|
||||||
|
id: item_id,
|
||||||
|
due: old_due,
|
||||||
|
}))
|
||||||
|
.set(
|
||||||
|
ValueClass::TaskQueue(TaskQueueClass::Due {
|
||||||
|
id: item_id,
|
||||||
|
due: at,
|
||||||
|
}),
|
||||||
|
object_id.serialize(),
|
||||||
|
)
|
||||||
|
.set(key, report.to_pickled_vec());
|
||||||
|
server.store().write(batch.build_all()).await.unwrap();
|
||||||
|
server.notify_task_queue();
|
||||||
|
old_due
|
||||||
|
}
|
||||||
|
|
||||||
|
struct RawValue(Vec<u8>);
|
||||||
|
|
||||||
|
impl store::Deserialize for RawValue {
|
||||||
|
fn deserialize(bytes: &[u8]) -> trc::Result<Self> {
|
||||||
|
Ok(RawValue(bytes.to_vec()))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn queue_row(server: &Server, id: u64, due: u64) -> Option<Vec<u8>> {
|
||||||
|
server
|
||||||
|
.store()
|
||||||
|
.get_value::<RawValue>(ValueKey::from(ValueClass::TaskQueue(TaskQueueClass::Due {
|
||||||
|
id,
|
||||||
|
due,
|
||||||
|
})))
|
||||||
|
.await
|
||||||
|
.unwrap()
|
||||||
|
.map(|raw| raw.0)
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn task_exists(server: &Server, id: u64) -> bool {
|
||||||
|
server
|
||||||
|
.store()
|
||||||
|
.get_value::<Task>(ValueKey::from(ValueClass::TaskQueue(
|
||||||
|
TaskQueueClass::Task { id },
|
||||||
|
)))
|
||||||
|
.await
|
||||||
|
.unwrap()
|
||||||
|
.is_some()
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn wait_until_run(server: &Server, ids: &[u64]) {
|
||||||
|
let started = Instant::now();
|
||||||
|
loop {
|
||||||
|
let mut pending = Vec::new();
|
||||||
|
for id in ids {
|
||||||
|
if task_exists(server, *id).await {
|
||||||
|
pending.push(*id);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if pending.is_empty() {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if started.elapsed() > Duration::from_secs(30) {
|
||||||
|
panic!("tasks {pending:?} never ran");
|
||||||
|
}
|
||||||
|
tokio::time::sleep(Duration::from_millis(200)).await;
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -2,8 +2,6 @@
|
|||||||
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <[email protected]>
|
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <[email protected]>
|
||||||
*
|
*
|
||||||
* SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL
|
* SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL
|
||||||
*
|
|
||||||
* Modified by Coffey Labs in 2026 for INBUXA.
|
|
||||||
*/
|
*/
|
||||||
|
|
||||||
use crate::utils::{
|
use crate::utils::{
|
||||||
@@ -11,22 +9,14 @@ use crate::utils::{
|
|||||||
server::TestServer,
|
server::TestServer,
|
||||||
temp_dir::TempDir,
|
temp_dir::TempDir,
|
||||||
};
|
};
|
||||||
use ::registry::schema::{
|
use ::registry::schema::enums::CompressionAlgo;
|
||||||
enums::{CompressionAlgo, TaskStoreMaintenanceType},
|
|
||||||
prelude::ObjectType,
|
|
||||||
structs::Task,
|
|
||||||
};
|
|
||||||
use ahash::AHashSet;
|
use ahash::AHashSet;
|
||||||
use common::{
|
use common::{DATABASE_SCHEMA_VERSION, manager::backup::BackupParams};
|
||||||
DATABASE_SCHEMA_VERSION,
|
|
||||||
manager::{SPAM_CLASSIFIER_KEY, SPAM_TRAINER_KEY, backup::BackupParams},
|
|
||||||
};
|
|
||||||
use store::{
|
use store::{
|
||||||
rand,
|
rand,
|
||||||
write::{
|
write::{
|
||||||
AnyClass, AnyKey, BatchBuilder, BlobLink, BlobOp, Operation, QueueClass, QueueEvent,
|
AnyClass, AnyKey, BatchBuilder, BlobLink, BlobOp, Operation, QueueClass, QueueEvent,
|
||||||
RegistryClass, TaskQueueClass, ValueClass,
|
RegistryClass, ValueClass, key::KeySerializer,
|
||||||
key::{DeserializeBigEndian, KeySerializer},
|
|
||||||
},
|
},
|
||||||
*,
|
*,
|
||||||
};
|
};
|
||||||
@@ -177,50 +167,6 @@ pub async fn test(test: &TestServer) {
|
|||||||
}
|
}
|
||||||
db.write(batch.build_all()).await.unwrap();
|
db.write(batch.build_all()).await.unwrap();
|
||||||
|
|
||||||
// inbuxa: registry objects kept outside the registry subspace (archived
|
|
||||||
// items for undelete, spam training samples, directory entries) and the
|
|
||||||
// fork's own subspace. Exports used to leave the first two behind.
|
|
||||||
println!("Creating archived items, spam samples and fork data...");
|
|
||||||
let mut batch = BatchBuilder::new();
|
|
||||||
for item_id in [1u64, 2, 3] {
|
|
||||||
for object in [
|
|
||||||
ObjectType::ArchivedItem,
|
|
||||||
ObjectType::SpamTrainingSample,
|
|
||||||
ObjectType::Account,
|
|
||||||
] {
|
|
||||||
batch.set(
|
|
||||||
ValueClass::Registry(RegistryClass::Item {
|
|
||||||
object_id: object as u16,
|
|
||||||
item_id,
|
|
||||||
}),
|
|
||||||
random_bytes(item_id as usize * 64),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
batch.set(
|
|
||||||
ValueClass::Any(AnyClass {
|
|
||||||
subspace: SUBSPACE_INBUXA,
|
|
||||||
key: [b'U', b'x']
|
|
||||||
.into_iter()
|
|
||||||
.chain(item_id.to_be_bytes())
|
|
||||||
.collect(),
|
|
||||||
}),
|
|
||||||
random_bytes(32),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
db.write(batch.build_all()).await.unwrap();
|
|
||||||
|
|
||||||
// inbuxa: the trained spam classifier lives in blobs with fixed names
|
|
||||||
let mut named_blobs = Vec::new();
|
|
||||||
for key in [SPAM_CLASSIFIER_KEY, SPAM_TRAINER_KEY] {
|
|
||||||
let data = random_bytes(4096);
|
|
||||||
test.server
|
|
||||||
.blob_store()
|
|
||||||
.put_blob(key, &data, CompressionAlgo::Lz4)
|
|
||||||
.await
|
|
||||||
.unwrap();
|
|
||||||
named_blobs.push((key, data));
|
|
||||||
}
|
|
||||||
|
|
||||||
// Create directory data
|
// Create directory data
|
||||||
println!("Creating directory data...");
|
println!("Creating directory data...");
|
||||||
let mut batch = BatchBuilder::new();
|
let mut batch = BatchBuilder::new();
|
||||||
@@ -239,17 +185,6 @@ pub async fn test(test: &TestServer) {
|
|||||||
println!("Calculating store hash...");
|
println!("Calculating store hash...");
|
||||||
let snapshot = Snapshot::new(&db).await;
|
let snapshot = Snapshot::new(&db).await;
|
||||||
assert!(!snapshot.keys.is_empty(), "Store hash counts are empty",);
|
assert!(!snapshot.keys.is_empty(), "Store hash counts are empty",);
|
||||||
for subspace in [
|
|
||||||
SUBSPACE_DELETED_ITEMS,
|
|
||||||
SUBSPACE_SPAM_SAMPLES,
|
|
||||||
SUBSPACE_INBUXA,
|
|
||||||
] {
|
|
||||||
assert!(
|
|
||||||
snapshot.keys.iter().any(|k| k.subspace == subspace),
|
|
||||||
"No test data in subspace {}",
|
|
||||||
char::from(subspace)
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
// Export store
|
// Export store
|
||||||
println!("Exporting store...");
|
println!("Exporting store...");
|
||||||
@@ -275,188 +210,22 @@ pub async fn test(test: &TestServer) {
|
|||||||
.finalize(),
|
.finalize(),
|
||||||
);
|
);
|
||||||
db.write(batch.build_all()).await.unwrap();
|
db.write(batch.build_all()).await.unwrap();
|
||||||
for (key, _) in &named_blobs {
|
test.server.core.restore(temp_dir.path.clone()).await;
|
||||||
test.server.blob_store().delete_blob(key).await.unwrap();
|
|
||||||
}
|
|
||||||
let imported = test.server.core.restore(temp_dir.path.clone()).await;
|
|
||||||
let mut batch = BatchBuilder::new();
|
let mut batch = BatchBuilder::new();
|
||||||
batch.clear(ValueClass::NodeId(0));
|
batch.clear(ValueClass::NodeId(0));
|
||||||
db.write(batch.build_all()).await.unwrap();
|
db.write(batch.build_all()).await.unwrap();
|
||||||
for subspace in [
|
|
||||||
SUBSPACE_DELETED_ITEMS,
|
|
||||||
SUBSPACE_SPAM_SAMPLES,
|
|
||||||
SUBSPACE_INBUXA,
|
|
||||||
] {
|
|
||||||
assert!(
|
|
||||||
imported.contains(&subspace),
|
|
||||||
"Subspace {} was not exported",
|
|
||||||
char::from(subspace)
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
// Verify hash
|
// Verify hash
|
||||||
print!("Verifying store hash...");
|
print!("Verifying store hash...");
|
||||||
snapshot.assert_is_eq(&Snapshot::new(&db).await);
|
snapshot.assert_is_eq(&Snapshot::new(&db).await);
|
||||||
assert_named_blobs(test.server.blob_store(), &named_blobs).await;
|
|
||||||
println!(" GREAT SUCCESS!");
|
println!(" GREAT SUCCESS!");
|
||||||
|
|
||||||
// inbuxa: import the same export into a fresh store of another backend,
|
|
||||||
// the way a move from one database to another does it
|
|
||||||
#[cfg(all(feature = "rocks", feature = "sqlite"))]
|
|
||||||
cross_backend(test, &db, &temp_dir, &named_blobs).await;
|
|
||||||
|
|
||||||
// Destroy store
|
// Destroy store
|
||||||
for (key, _) in &named_blobs {
|
|
||||||
test.server.blob_store().delete_blob(key).await.unwrap();
|
|
||||||
}
|
|
||||||
store_destroy(&db).await;
|
store_destroy(&db).await;
|
||||||
store_assert_is_empty(&db, db.clone().into(), true).await;
|
store_assert_is_empty(&db, db.clone().into(), true).await;
|
||||||
temp_dir.delete();
|
temp_dir.delete();
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(all(feature = "rocks", feature = "sqlite"))]
|
|
||||||
async fn cross_backend(
|
|
||||||
test: &TestServer,
|
|
||||||
source: &Store,
|
|
||||||
export: &TempDir,
|
|
||||||
named_blobs: &[(&[u8], Vec<u8>)],
|
|
||||||
) {
|
|
||||||
let source_type = std::env::var("STORE").unwrap();
|
|
||||||
let target_type = if source_type.eq_ignore_ascii_case("sqlite") {
|
|
||||||
"RocksDb"
|
|
||||||
} else {
|
|
||||||
"Sqlite"
|
|
||||||
};
|
|
||||||
println!("Importing the export into a fresh {target_type} store...");
|
|
||||||
|
|
||||||
let target_dir = TempDir::new("art_vandelay_cross_backend", true);
|
|
||||||
let target = Store::build(
|
|
||||||
crate::utils::storage::build_data_store(target_type, &target_dir.path.to_string_lossy())
|
|
||||||
.await,
|
|
||||||
)
|
|
||||||
.await
|
|
||||||
.unwrap();
|
|
||||||
target.create_tables().await.unwrap();
|
|
||||||
store_destroy(&target).await;
|
|
||||||
|
|
||||||
let mut core = test.server.core.as_ref().clone();
|
|
||||||
core.storage.data = target.clone();
|
|
||||||
core.storage.blob = target.clone().into();
|
|
||||||
let imported = core.restore(export.path.clone()).await;
|
|
||||||
|
|
||||||
// Counters are stored differently by the SQL and key-value backends, so
|
|
||||||
// compare their keys here and their values through the counter API.
|
|
||||||
print!("Verifying {target_type} store hash...");
|
|
||||||
Snapshot::new_portable(source)
|
|
||||||
.await
|
|
||||||
.assert_is_eq(&Snapshot::new_portable(&target).await);
|
|
||||||
for subspace in [SUBSPACE_COUNTER, SUBSPACE_QUOTA] {
|
|
||||||
let mut keys = Vec::new();
|
|
||||||
source
|
|
||||||
.iterate(
|
|
||||||
IterateParams::new(
|
|
||||||
AnyKey {
|
|
||||||
subspace,
|
|
||||||
key: vec![0u8],
|
|
||||||
},
|
|
||||||
AnyKey {
|
|
||||||
subspace,
|
|
||||||
key: vec![u8::MAX; 10],
|
|
||||||
},
|
|
||||||
)
|
|
||||||
.no_values(),
|
|
||||||
|key, _| {
|
|
||||||
keys.push(key.to_vec());
|
|
||||||
Ok(true)
|
|
||||||
},
|
|
||||||
)
|
|
||||||
.await
|
|
||||||
.unwrap();
|
|
||||||
for key in keys {
|
|
||||||
let class = || {
|
|
||||||
ValueClass::Any(AnyClass {
|
|
||||||
subspace,
|
|
||||||
key: key.clone(),
|
|
||||||
})
|
|
||||||
};
|
|
||||||
assert_eq!(
|
|
||||||
source.get_counter(class()).await.unwrap(),
|
|
||||||
target.get_counter(class()).await.unwrap(),
|
|
||||||
"Counter mismatch in {} for {key:?}",
|
|
||||||
char::from(subspace)
|
|
||||||
);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
assert_named_blobs(&core.storage.blob, named_blobs).await;
|
|
||||||
println!(" GREAT SUCCESS!");
|
|
||||||
|
|
||||||
// The search index isn't exported; the import queues its rebuild
|
|
||||||
let queued = core.queue_reindex(&imported).await;
|
|
||||||
let expected = [
|
|
||||||
TaskStoreMaintenanceType::ReindexAccounts,
|
|
||||||
TaskStoreMaintenanceType::ReindexTelemetry,
|
|
||||||
];
|
|
||||||
assert_eq!(queued, expected);
|
|
||||||
let mut task_ids = Vec::new();
|
|
||||||
target
|
|
||||||
.iterate(
|
|
||||||
IterateParams::new(
|
|
||||||
AnyKey {
|
|
||||||
subspace: SUBSPACE_TASK_QUEUE,
|
|
||||||
key: vec![0u8],
|
|
||||||
},
|
|
||||||
AnyKey {
|
|
||||||
subspace: SUBSPACE_TASK_QUEUE,
|
|
||||||
key: vec![u8::MAX; 20],
|
|
||||||
},
|
|
||||||
)
|
|
||||||
.no_values(),
|
|
||||||
|key, _| {
|
|
||||||
if key.deserialize_be_u64(0)? == 0 {
|
|
||||||
task_ids.push(key.deserialize_be_u64(U64_LEN)?);
|
|
||||||
}
|
|
||||||
Ok(true)
|
|
||||||
},
|
|
||||||
)
|
|
||||||
.await
|
|
||||||
.unwrap();
|
|
||||||
let mut found = Vec::new();
|
|
||||||
for id in task_ids {
|
|
||||||
match target
|
|
||||||
.get_value::<Task>(ValueKey::from(ValueClass::TaskQueue(
|
|
||||||
TaskQueueClass::Task { id },
|
|
||||||
)))
|
|
||||||
.await
|
|
||||||
.unwrap()
|
|
||||||
{
|
|
||||||
Some(Task::StoreMaintenance(task)) => found.push(task.maintenance_type),
|
|
||||||
other => panic!("Unexpected task {other:?}"),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
found.sort_by_key(|t| *t as u16);
|
|
||||||
assert_eq!(found, expected, "Queued tasks don't match");
|
|
||||||
|
|
||||||
store_destroy(&target).await;
|
|
||||||
drop(core);
|
|
||||||
drop(target);
|
|
||||||
target_dir.delete();
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn assert_named_blobs(blob_store: &BlobStore, named_blobs: &[(&[u8], Vec<u8>)]) {
|
|
||||||
for (key, data) in named_blobs {
|
|
||||||
assert_eq!(
|
|
||||||
blob_store
|
|
||||||
.get_blob(key, 0..usize::MAX)
|
|
||||||
.await
|
|
||||||
.unwrap()
|
|
||||||
.as_ref(),
|
|
||||||
Some(data),
|
|
||||||
"Blob {} was not restored",
|
|
||||||
String::from_utf8_lossy(key)
|
|
||||||
);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, PartialEq, Eq)]
|
#[derive(Debug, PartialEq, Eq)]
|
||||||
struct Snapshot {
|
struct Snapshot {
|
||||||
keys: AHashSet<KeyValue>,
|
keys: AHashSet<KeyValue>,
|
||||||
@@ -471,19 +240,7 @@ struct KeyValue {
|
|||||||
|
|
||||||
impl Snapshot {
|
impl Snapshot {
|
||||||
async fn new(db: &Store) -> Self {
|
async fn new(db: &Store) -> Self {
|
||||||
Self::build(db, !db.is_sql(), true).await
|
let is_sql = db.is_sql();
|
||||||
}
|
|
||||||
|
|
||||||
/// Comparable across backends: no counter values, which the SQL and
|
|
||||||
/// key-value stores encode differently, and no blobs, which only live in
|
|
||||||
/// the data store when it doubles as the blob store.
|
|
||||||
#[cfg(all(feature = "rocks", feature = "sqlite"))]
|
|
||||||
async fn new_portable(db: &Store) -> Self {
|
|
||||||
Self::build(db, false, false).await
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn build(db: &Store, counter_values: bool, with_blobs: bool) -> Self {
|
|
||||||
let is_sql = !counter_values;
|
|
||||||
|
|
||||||
let mut keys = AHashSet::new();
|
let mut keys = AHashSet::new();
|
||||||
|
|
||||||
@@ -508,12 +265,7 @@ impl Snapshot {
|
|||||||
(SUBSPACE_QUOTA, !is_sql),
|
(SUBSPACE_QUOTA, !is_sql),
|
||||||
(SUBSPACE_REPORT_OUT, true),
|
(SUBSPACE_REPORT_OUT, true),
|
||||||
(SUBSPACE_REPORT_IN, true),
|
(SUBSPACE_REPORT_IN, true),
|
||||||
(SUBSPACE_DIRECTORY, true),
|
|
||||||
(SUBSPACE_INBUXA, true),
|
|
||||||
] {
|
] {
|
||||||
if subspace == SUBSPACE_BLOBS && !with_blobs {
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
let from_key = AnyKey {
|
let from_key = AnyKey {
|
||||||
subspace,
|
subspace,
|
||||||
key: vec![0u8],
|
key: vec![0u8],
|
||||||
|
|||||||
@@ -21,7 +21,6 @@ pub mod replica_cluster; // inbuxa: read replicas across nodes
|
|||||||
pub mod scaleout; // inbuxa: scale-out storage
|
pub mod scaleout; // inbuxa: scale-out storage
|
||||||
#[cfg(any(feature = "postgres", feature = "mysql"))]
|
#[cfg(any(feature = "postgres", feature = "mysql"))]
|
||||||
pub mod sql_timeout;
|
pub mod sql_timeout;
|
||||||
pub mod task_locks; // inbuxa: task locks across nodes
|
|
||||||
|
|
||||||
use crate::utils::server::TestServerBuilder;
|
use crate::utils::server::TestServerBuilder;
|
||||||
use std::io::Read;
|
use std::io::Read;
|
||||||
|
|||||||
@@ -2,8 +2,6 @@
|
|||||||
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <[email protected]>
|
* SPDX-FileCopyrightText: 2020 Stalwart Labs LLC <[email protected]>
|
||||||
*
|
*
|
||||||
* SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL
|
* SPDX-License-Identifier: AGPL-3.0-only OR LicenseRef-SEL
|
||||||
*
|
|
||||||
* Modified by Coffey Labs in 2026 for INBUXA.
|
|
||||||
*/
|
*/
|
||||||
|
|
||||||
use crate::{store::deflate_test_resource, utils::server::TestServer};
|
use crate::{store::deflate_test_resource, utils::server::TestServer};
|
||||||
@@ -124,10 +122,6 @@ pub async fn test(test: &TestServer) {
|
|||||||
println!("Running global id filtering tests...");
|
println!("Running global id filtering tests...");
|
||||||
test_global(store.clone()).await;
|
test_global(store.clone()).await;
|
||||||
|
|
||||||
// inbuxa: trace documents as the index task builds them
|
|
||||||
println!("Running trace document tests...");
|
|
||||||
test_trace_documents(store.clone()).await;
|
|
||||||
|
|
||||||
// Large document insert test
|
// Large document insert test
|
||||||
println!("Running large document insert tests...");
|
println!("Running large document insert tests...");
|
||||||
let mut large_text = String::with_capacity(20 * 1024 * 1024);
|
let mut large_text = String::with_capacity(20 * 1024 * 1024);
|
||||||
@@ -815,160 +809,3 @@ async fn test_global(store: SearchStore) {
|
|||||||
AHashSet::from_iter([3, 4, 5])
|
AHashSet::from_iter([3, 4, 5])
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
// inbuxa: MON-16: documents built by the index task from stored traces go
|
|
||||||
// into every search backend (the SQL backends type etyp and qid as BIGINT)
|
|
||||||
// and are found again by queue id and keyword.
|
|
||||||
async fn test_trace_documents(store: SearchStore) {
|
|
||||||
use registry::schema::{
|
|
||||||
enums::SearchTracingField,
|
|
||||||
structs::{
|
|
||||||
Trace, TraceEvent, TraceKeyValue, TraceValue, TraceValueString,
|
|
||||||
TraceValueUnsignedInt,
|
|
||||||
},
|
|
||||||
};
|
|
||||||
use services::task_manager::index::trace_search_document;
|
|
||||||
use trc::{DeliveryEvent, EventType, Key, SmtpEvent};
|
|
||||||
|
|
||||||
let kv_u = |key: Key, value: u64| TraceKeyValue {
|
|
||||||
key,
|
|
||||||
value: TraceValue::UnsignedInt(TraceValueUnsignedInt { value }),
|
|
||||||
};
|
|
||||||
let kv_s = |key: Key, value: &str| TraceKeyValue {
|
|
||||||
key,
|
|
||||||
value: TraceValue::String(TraceValueString {
|
|
||||||
value: value.to_string(),
|
|
||||||
}),
|
|
||||||
};
|
|
||||||
let event = |event: EventType, key_values: Vec<TraceKeyValue>| TraceEvent {
|
|
||||||
event,
|
|
||||||
key_values: key_values.into(),
|
|
||||||
..Default::default()
|
|
||||||
};
|
|
||||||
let fields = [
|
|
||||||
SearchTracingField::EventType,
|
|
||||||
SearchTracingField::QueueId,
|
|
||||||
SearchTracingField::Keywords,
|
|
||||||
];
|
|
||||||
|
|
||||||
// An SMTP session that queued two messages, and a delivery attempt
|
|
||||||
let session = Trace {
|
|
||||||
events: vec![
|
|
||||||
event(
|
|
||||||
EventType::Smtp(SmtpEvent::ConnectionStart),
|
|
||||||
vec![kv_s(Key::RemoteIp, "192.0.2.7")],
|
|
||||||
),
|
|
||||||
event(
|
|
||||||
EventType::Smtp(SmtpEvent::MailFrom),
|
|
||||||
vec![kv_s(Key::From, "[email protected]")],
|
|
||||||
),
|
|
||||||
event(
|
|
||||||
EventType::Smtp(SmtpEvent::RcptTo),
|
|
||||||
vec![kv_u(Key::QueueId, 9_000_000_001), kv_s(Key::To, "[email protected]")],
|
|
||||||
),
|
|
||||||
event(
|
|
||||||
EventType::Smtp(SmtpEvent::RcptTo),
|
|
||||||
vec![kv_u(Key::QueueId, 9_000_000_002)],
|
|
||||||
),
|
|
||||||
]
|
|
||||||
.into(),
|
|
||||||
};
|
|
||||||
let delivery = Trace {
|
|
||||||
events: vec![event(
|
|
||||||
EventType::Delivery(DeliveryEvent::AttemptStart),
|
|
||||||
vec![kv_u(Key::QueueId, 9_000_000_003), kv_s(Key::Hostname, "relay.example.net")],
|
|
||||||
)]
|
|
||||||
.into(),
|
|
||||||
};
|
|
||||||
let documents = vec![
|
|
||||||
trace_search_document(100, &session, &fields),
|
|
||||||
trace_search_document(101, &delivery, &fields),
|
|
||||||
];
|
|
||||||
assert!(
|
|
||||||
documents
|
|
||||||
.iter()
|
|
||||||
.all(|d| d.has_field(&SearchField::Tracing(TracingSearchField::QueueId))
|
|
||||||
&& d.has_field(&SearchField::Tracing(TracingSearchField::EventType))),
|
|
||||||
"trace documents carry a queue id and an event type"
|
|
||||||
);
|
|
||||||
store.index(documents).await.unwrap();
|
|
||||||
if let SearchStore::ElasticSearch(store) = &store {
|
|
||||||
store.refresh_index(SearchIndex::Tracing).await.unwrap();
|
|
||||||
}
|
|
||||||
|
|
||||||
let query = |filters: Vec<SearchFilter>| {
|
|
||||||
let store = store.clone();
|
|
||||||
async move {
|
|
||||||
store
|
|
||||||
.query_global(
|
|
||||||
SearchQuery::new(SearchIndex::Tracing)
|
|
||||||
.with_filter(SearchFilter::ge(SearchField::Id, 100u64))
|
|
||||||
.with_filters(filters),
|
|
||||||
)
|
|
||||||
.await
|
|
||||||
.unwrap()
|
|
||||||
.into_iter()
|
|
||||||
.collect::<AHashSet<_>>()
|
|
||||||
}
|
|
||||||
};
|
|
||||||
// By queue id, the way x:Trace/query asks: the queue id column, or any
|
|
||||||
// queue id in the keywords
|
|
||||||
let by_queue_id = |queue_id: u64| {
|
|
||||||
vec![
|
|
||||||
SearchFilter::Or,
|
|
||||||
SearchFilter::eq(TracingSearchField::QueueId, queue_id),
|
|
||||||
SearchFilter::has_text(
|
|
||||||
TracingSearchField::Keywords,
|
|
||||||
queue_id.to_string(),
|
|
||||||
Language::None,
|
|
||||||
),
|
|
||||||
SearchFilter::End,
|
|
||||||
]
|
|
||||||
};
|
|
||||||
assert_eq!(query(by_queue_id(9_000_000_001)).await, AHashSet::from_iter([100]));
|
|
||||||
assert_eq!(query(by_queue_id(9_000_000_002)).await, AHashSet::from_iter([100]));
|
|
||||||
assert_eq!(query(by_queue_id(9_000_000_003)).await, AHashSet::from_iter([101]));
|
|
||||||
assert_eq!(query(by_queue_id(9_000_000_004)).await, AHashSet::new());
|
|
||||||
assert_eq!(
|
|
||||||
query(vec![SearchFilter::eq(TracingSearchField::QueueId, 9_000_000_003u64)]).await,
|
|
||||||
AHashSet::from_iter([101])
|
|
||||||
);
|
|
||||||
// By opening event type
|
|
||||||
assert_eq!(
|
|
||||||
query(vec![SearchFilter::eq(
|
|
||||||
TracingSearchField::EventType,
|
|
||||||
EventType::Delivery(DeliveryEvent::AttemptStart).to_id() as u64,
|
|
||||||
)])
|
|
||||||
.await,
|
|
||||||
AHashSet::from_iter([101])
|
|
||||||
);
|
|
||||||
// By keyword: an address, lowercased, and its domain
|
|
||||||
assert_eq!(
|
|
||||||
query(vec![SearchFilter::has_text(
|
|
||||||
TracingSearchField::Keywords,
|
|
||||||
"example.org",
|
|
||||||
Language::None,
|
|
||||||
)])
|
|
||||||
.await,
|
|
||||||
AHashSet::from_iter([100])
|
|
||||||
);
|
|
||||||
assert_eq!(
|
|
||||||
query(vec![SearchFilter::has_text(
|
|
||||||
TracingSearchField::Keywords,
|
|
||||||
"relay.example.net",
|
|
||||||
Language::None,
|
|
||||||
)])
|
|
||||||
.await,
|
|
||||||
AHashSet::from_iter([101])
|
|
||||||
);
|
|
||||||
|
|
||||||
for id in [100u64, 101] {
|
|
||||||
store
|
|
||||||
.unindex(
|
|
||||||
SearchQuery::new(SearchIndex::Tracing)
|
|
||||||
.with_filter(SearchFilter::eq(SearchField::Id, id)),
|
|
||||||
)
|
|
||||||
.await
|
|
||||||
.unwrap();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -1,184 +0,0 @@
|
|||||||
/*
|
|
||||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
|
||||||
*
|
|
||||||
* SPDX-License-Identifier: AGPL-3.0-only
|
|
||||||
*/
|
|
||||||
|
|
||||||
//! Task locks across nodes: tasks claimed by a node that then disappears
|
|
||||||
//! run elsewhere once its locks expire, and a node that stops gracefully
|
|
||||||
//! hands its locks back at once. The other node is played by writing its
|
|
||||||
//! locks straight into the shared in-memory store, as a node that claimed
|
|
||||||
//! the tasks and died leaves them.
|
|
||||||
|
|
||||||
use crate::utils::server::TestServerBuilder;
|
|
||||||
use common::{KV_LOCK_TASK, Server};
|
|
||||||
use registry::schema::{
|
|
||||||
enums::IndexDocumentType,
|
|
||||||
structs::{Task, TaskIndexDocument, TaskStatus},
|
|
||||||
};
|
|
||||||
use services::task_manager::lock::{TaskLockManager, release_task_locks};
|
|
||||||
use std::time::{Duration, Instant};
|
|
||||||
use store::{
|
|
||||||
ValueKey,
|
|
||||||
write::{BatchBuilder, TaskQueueClass, ValueClass},
|
|
||||||
};
|
|
||||||
use utils::snowflake::SnowflakeIdGenerator;
|
|
||||||
|
|
||||||
// Short enough for a test, long enough that the recheck interval (a twelfth
|
|
||||||
// of it) is well below it
|
|
||||||
const LOCK_EXPIRY: u64 = 12;
|
|
||||||
|
|
||||||
#[tokio::test(flavor = "multi_thread")]
|
|
||||||
pub async fn task_lock_tests() {
|
|
||||||
let test = TestServerBuilder::new("task_lock_tests")
|
|
||||||
.await
|
|
||||||
.build()
|
|
||||||
.await;
|
|
||||||
let server = test.server.clone();
|
|
||||||
println!(
|
|
||||||
"Running task lock tests on {}...",
|
|
||||||
std::env::var("STORE").unwrap_or_default()
|
|
||||||
);
|
|
||||||
server.inner.ipc.task_locks.set_expiry(LOCK_EXPIRY);
|
|
||||||
|
|
||||||
// 1. Another node claimed the tasks and died. Its locks block them until
|
|
||||||
// they expire; then this node runs them, without waiting for anything
|
|
||||||
// else to wake it
|
|
||||||
let ids = new_task_ids(4);
|
|
||||||
for id in &ids {
|
|
||||||
assert!(foreign_lock(&server, *id, LOCK_EXPIRY).await);
|
|
||||||
}
|
|
||||||
schedule(&server, &ids).await;
|
|
||||||
let started = Instant::now();
|
|
||||||
server.notify_task_queue();
|
|
||||||
tokio::time::sleep(Duration::from_secs(3)).await;
|
|
||||||
assert_eq!(pending(&server, &ids).await, ids.len(), "held by the other node");
|
|
||||||
wait_until_done(&server, &ids, Duration::from_secs(LOCK_EXPIRY + 10)).await;
|
|
||||||
let elapsed = started.elapsed();
|
|
||||||
assert!(
|
|
||||||
elapsed >= Duration::from_secs(LOCK_EXPIRY - 2),
|
|
||||||
"ran before the other node's locks expired: {elapsed:?}"
|
|
||||||
);
|
|
||||||
|
|
||||||
// 2. The other node's locks outlive what this node expects: claimed just
|
|
||||||
// after this node looked, or by a node whose clock runs ahead. This node
|
|
||||||
// keeps checking at the recheck interval, so the tasks run soon after
|
|
||||||
// those locks expire, not a whole lock lifetime later
|
|
||||||
let held_for = LOCK_EXPIRY + LOCK_EXPIRY / 2;
|
|
||||||
let ids = new_task_ids(4);
|
|
||||||
for id in &ids {
|
|
||||||
assert!(foreign_lock(&server, *id, held_for).await);
|
|
||||||
}
|
|
||||||
schedule(&server, &ids).await;
|
|
||||||
let started = Instant::now();
|
|
||||||
server.notify_task_queue();
|
|
||||||
wait_until_done(&server, &ids, Duration::from_secs(held_for + 8)).await;
|
|
||||||
let elapsed = started.elapsed();
|
|
||||||
assert!(
|
|
||||||
elapsed >= Duration::from_secs(held_for - 2),
|
|
||||||
"ran before the other node's locks expired: {elapsed:?}"
|
|
||||||
);
|
|
||||||
|
|
||||||
// 3. A graceful stop releases the locks this node holds: another node
|
|
||||||
// can claim those tasks at once, and this one claims nothing more
|
|
||||||
let ids = new_task_ids(3);
|
|
||||||
for id in &ids {
|
|
||||||
assert!(server.try_lock_task(*id).await, "claim {id}");
|
|
||||||
}
|
|
||||||
assert_eq!(server.inner.ipc.task_locks.held(), ids.len());
|
|
||||||
for id in &ids {
|
|
||||||
assert!(
|
|
||||||
!foreign_lock(&server, *id, LOCK_EXPIRY).await,
|
|
||||||
"held while this node runs"
|
|
||||||
);
|
|
||||||
}
|
|
||||||
assert_eq!(release_task_locks(&server).await, ids.len());
|
|
||||||
assert_eq!(server.inner.ipc.task_locks.held(), 0);
|
|
||||||
for id in &ids {
|
|
||||||
assert!(
|
|
||||||
foreign_lock(&server, *id, LOCK_EXPIRY).await,
|
|
||||||
"released on stop: {id}"
|
|
||||||
);
|
|
||||||
}
|
|
||||||
let [id] = new_task_ids(1)[..] else {
|
|
||||||
unreachable!()
|
|
||||||
};
|
|
||||||
assert!(!server.try_lock_task(id).await, "a stopping node claims nothing");
|
|
||||||
|
|
||||||
for id in ids {
|
|
||||||
let _ = server
|
|
||||||
.in_memory_store()
|
|
||||||
.remove_lock(KV_LOCK_TASK, &id.to_be_bytes())
|
|
||||||
.await;
|
|
||||||
}
|
|
||||||
if test.is_reset() {
|
|
||||||
test.temp_dir.delete();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn new_task_ids(count: usize) -> Vec<u64> {
|
|
||||||
(0..count)
|
|
||||||
.map(|_| SnowflakeIdGenerator::global_id().unwrap())
|
|
||||||
.collect()
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The other node's claim on a task, as its task manager takes it.
|
|
||||||
async fn foreign_lock(server: &Server, id: u64, seconds: u64) -> bool {
|
|
||||||
server
|
|
||||||
.in_memory_store()
|
|
||||||
.try_lock(KV_LOCK_TASK, &id.to_be_bytes(), seconds)
|
|
||||||
.await
|
|
||||||
.unwrap()
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Unindex tasks for files that don't exist: files aren't search-indexed and
|
|
||||||
/// there is no undelete note, so running one only drops it from the queue.
|
|
||||||
async fn schedule(server: &Server, ids: &[u64]) {
|
|
||||||
let mut batch = BatchBuilder::new();
|
|
||||||
for (n, id) in ids.iter().enumerate() {
|
|
||||||
batch.schedule_task_with_id(
|
|
||||||
*id,
|
|
||||||
Task::UnindexDocument(TaskIndexDocument {
|
|
||||||
account_id: 0u32.into(),
|
|
||||||
document_id: (u32::MAX - n as u32).into(),
|
|
||||||
document_type: IndexDocumentType::File,
|
|
||||||
status: TaskStatus::now(),
|
|
||||||
}),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
server.store().write(batch.build_all()).await.unwrap();
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn pending(server: &Server, ids: &[u64]) -> usize {
|
|
||||||
let mut count = 0;
|
|
||||||
for id in ids {
|
|
||||||
if server
|
|
||||||
.store()
|
|
||||||
.get_value::<Task>(ValueKey::from(ValueClass::TaskQueue(
|
|
||||||
TaskQueueClass::Task { id: *id },
|
|
||||||
)))
|
|
||||||
.await
|
|
||||||
.unwrap()
|
|
||||||
.is_some()
|
|
||||||
{
|
|
||||||
count += 1;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
count
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn wait_until_done(server: &Server, ids: &[u64], within: Duration) {
|
|
||||||
let started = Instant::now();
|
|
||||||
loop {
|
|
||||||
let left = pending(server, ids).await;
|
|
||||||
if left == 0 {
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
assert!(
|
|
||||||
started.elapsed() < within,
|
|
||||||
"{left} task(s) still pending after {:?}",
|
|
||||||
started.elapsed()
|
|
||||||
);
|
|
||||||
tokio::time::sleep(Duration::from_millis(250)).await;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -148,73 +148,6 @@ pub async fn test(test: &mut TestServer) {
|
|||||||
"test 9: to"
|
"test 9: to"
|
||||||
);
|
);
|
||||||
|
|
||||||
// MON-16: the queueId filter finds the traces that name a queue id (the
|
|
||||||
// session that queued the message and its delivery attempt) through the
|
|
||||||
// search index, given as a string or a number (the index column is an
|
|
||||||
// integer)
|
|
||||||
fn queue_ids(value: &Value, out: &mut Vec<u64>) {
|
|
||||||
match value {
|
|
||||||
Value::Object(map) => {
|
|
||||||
if map.get("key").and_then(|k| k.as_str()) == Some("queueId")
|
|
||||||
&& let Some(id) = map
|
|
||||||
.get("value")
|
|
||||||
.and_then(|v| v.get("value").unwrap_or(v).as_u64())
|
|
||||||
{
|
|
||||||
out.push(id);
|
|
||||||
}
|
|
||||||
map.values().for_each(|v| queue_ids(v, out));
|
|
||||||
}
|
|
||||||
Value::Array(list) => list.iter().for_each(|v| queue_ids(v, out)),
|
|
||||||
_ => {}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
let with_ids = traces
|
|
||||||
.iter()
|
|
||||||
.map(|t| {
|
|
||||||
let mut ids = Vec::new();
|
|
||||||
queue_ids(t, &mut ids);
|
|
||||||
(t["id"].as_str().unwrap().to_string(), ids)
|
|
||||||
})
|
|
||||||
.collect::<Vec<_>>();
|
|
||||||
let queue_id = with_ids
|
|
||||||
.iter()
|
|
||||||
.find_map(|(_, ids)| ids.first().copied())
|
|
||||||
.expect("MON-16: a trace with a queue id");
|
|
||||||
let mut expected = with_ids
|
|
||||||
.iter()
|
|
||||||
.filter(|(_, ids)| ids.contains(&queue_id))
|
|
||||||
.map(|(id, _)| id.clone())
|
|
||||||
.collect::<Vec<_>>();
|
|
||||||
expected.sort();
|
|
||||||
for filter in [json!(queue_id.to_string()), json!(queue_id)] {
|
|
||||||
let response = admin
|
|
||||||
.jmap_method_call("x:Trace/query", json!({"filter": {"queueId": filter}}))
|
|
||||||
.await;
|
|
||||||
let mut found = response
|
|
||||||
.0
|
|
||||||
.pointer("/methodResponses/0/1/ids")
|
|
||||||
.and_then(|ids| ids.as_array())
|
|
||||||
.map(|ids| {
|
|
||||||
ids.iter()
|
|
||||||
.filter_map(|id| id.as_str().map(str::to_string))
|
|
||||||
.collect::<Vec<_>>()
|
|
||||||
})
|
|
||||||
.unwrap_or_default();
|
|
||||||
found.sort();
|
|
||||||
assert_eq!(found, expected, "MON-16: queueId {filter}: {response:?}");
|
|
||||||
}
|
|
||||||
let response = admin
|
|
||||||
.jmap_method_call(
|
|
||||||
"x:Trace/query",
|
|
||||||
json!({"filter": {"queueId": (queue_id ^ 0x5a5a_5a5a).to_string()}}),
|
|
||||||
)
|
|
||||||
.await;
|
|
||||||
assert_eq!(
|
|
||||||
response.0.pointer("/methodResponses/0/1/ids"),
|
|
||||||
Some(&json!([])),
|
|
||||||
"MON-16: an unknown queue id"
|
|
||||||
);
|
|
||||||
|
|
||||||
// Acceptance test 24: destroy removes a trace; create is refused
|
// Acceptance test 24: destroy removes a trace; create is refused
|
||||||
let trace_id = traces[0]["id"].as_str().unwrap().to_string();
|
let trace_id = traces[0]["id"].as_str().unwrap().to_string();
|
||||||
let response = admin
|
let response = admin
|
||||||
|
|||||||
Reference in New Issue
Block a user