Compare commits

..
Author SHA1 Message Date
jcoffey-dev 77404353d5 Merge pull request 'Release 2026.9.28.6' (#107) from hotfix/2026.9.28.6 into release/2026.9.28.6
publish / version (push) Successful in 11s
publish / publish-amd64 (push) Successful in 31m53s
publish / release (push) Successful in 1s
publish / publish-arm64 (push) Successful in 50m13s
publish / binaries (push) Successful in 42s
publish / announce (push) Successful in 22s
2026-09-29 01:29:48 +00:00
jcoffey-dev d2b0750e1d Release 2026.9.28.6
ci / fork-checks (pull_request) Successful in 45s
ci / build (pull_request) Successful in 6m10s
2026-09-28 18:23:26 -07:00
jcoffey-dev 0232e59283 Ports: each node checks the others' ports from outside 2026-09-28 18:16:33 -07:00
jcoffey-dev 9fd1d32799 Explain: don't prepare answers for date fields 2026-09-28 18:16:33 -07:00
jcoffey-dev eafe8bfbfd Webhooks: send one sample event to a saved webhook 2026-09-28 18:16:33 -07:00
48 changed files with 48 additions and 6342 deletions
Generated
-4
View File
@@ -3947,12 +3947,9 @@ name = "inbuxa-features"
version = "0.16.22"
dependencies = [
"ahash",
"aho-corasick",
"base64 0.23.1",
"flate2",
"jmap_proto",
"quick-xml 0.41.0",
"regex",
"registry",
"serde",
"serde_json",
@@ -3964,7 +3961,6 @@ dependencies = [
"types",
"utils",
"xxhash-rust",
"zip",
]
[[package]]
-11
View File
@@ -296,17 +296,6 @@ impl Default for DefaultPermissions {
default.superuser.push(permission);
default.tenant.push(permission);
}
// inbuxa: DLP and mail flow rules, and held mail, are the
// server's: never a tenant's (dlp-and-mail-flow-rules spec,
// settled answer 3)
Permission::SysMailRuleGet
| Permission::SysMailRuleUpdate
| Permission::SysDlpPolicyGet
| Permission::SysDlpPolicyUpdate
| Permission::SysDlpReviewGet
| Permission::SysDlpReviewUpdate => {
default.superuser.push(permission);
}
// inbuxa: AL-12: tenant administrators lock and delegate
// within their tenant
Permission::SysAccountLockGet
+7 -34
View File
@@ -65,10 +65,6 @@ const OFFICER: &[Permission] = &[
Permission::SysLegalHoldUpdate,
Permission::SysLegalHoldExport,
Permission::SysAccountLockGet,
// dlp-and-mail-flow-rules spec, §2.8: see DLP rules, review held mail
Permission::SysDlpPolicyGet,
Permission::SysDlpReviewGet,
Permission::SysDlpReviewUpdate,
];
/// What a tenant's officer holds besides [`READS`].
@@ -117,11 +113,6 @@ fn created_key(tenant: Option<Id>) -> ValueClass {
})
}
/// The server-level Compliance Officer role the server made, if it has.
pub async fn server_role(data: &Store) -> trc::Result<Option<Id>> {
recorded(data, None).await
}
async fn recorded(data: &Store, tenant: Option<Id>) -> trc::Result<Option<Id>> {
Ok(data
.get_value::<u64>(ValueKey::from(created_key(tenant)))
@@ -181,19 +172,13 @@ pub async fn ensure_compliance_roles(registry: &RegistryStore, data: &Store) ->
/// A new tenant gets its Compliance Officer role.
pub async fn tenant_created(registry: &RegistryStore, data: &Store, tenant: Id) -> trc::Result<()> {
create_once(registry, data, Some(tenant), tenant_role(tenant))
.await
.map(|_| ())
create_once(registry, data, Some(tenant), tenant_role(tenant)).await.map(|_| ())
}
/// Before a tenant is deleted: removes its Compliance Officer role if nobody
/// holds it, so the role doesn't block the delete. Returns whether it did,
/// so a delete refused for another reason can put it back.
pub async fn tenant_deleting(
registry: &RegistryStore,
data: &Store,
tenant: Id,
) -> trc::Result<bool> {
pub async fn tenant_deleting(registry: &RegistryStore, data: &Store, tenant: Id) -> trc::Result<bool> {
let Some(role) = recorded(data, Some(tenant)).await? else {
return Ok(false);
};
@@ -235,9 +220,7 @@ mod tests {
// Beyond what any user holds for their own account
for permission in all.into_iter().filter(|p| !user.contains(p)) {
let name = permission.as_str();
// Placing holds and reviewing held mail are the officer's
// job, not settings (settled answers 2 and 4)
let holds = name.starts_with("sysLegalHold") || name.starts_with("sysDlpReview");
let holds = name.starts_with("sysLegalHold");
assert!(
!(name.ends_with("Update") && !holds)
&& !(name.ends_with("Create") && !holds)
@@ -266,11 +249,7 @@ mod tests {
assert!(officer.contains(&hold));
assert!(!tenant.contains(&hold));
}
for both in [
Permission::SysComplianceGet,
Permission::SysAuditGet,
Permission::SysAccountGet,
] {
for both in [Permission::SysComplianceGet, Permission::SysAuditGet, Permission::SysAccountGet] {
assert!(officer.contains(&both) && tenant.contains(&both));
}
assert!(!officer.contains(&Permission::SysAuditSettingsUpdate));
@@ -278,15 +257,9 @@ mod tests {
#[test]
fn records_are_per_place() {
let ValueClass::Any(server) = created_key(None) else {
panic!()
};
let ValueClass::Any(a) = created_key(Some(Id::from(1u64))) else {
panic!()
};
let ValueClass::Any(b) = created_key(Some(Id::from(2u64))) else {
panic!()
};
let ValueClass::Any(server) = created_key(None) else { panic!() };
let ValueClass::Any(a) = created_key(Some(Id::from(1u64))) else { panic!() };
let ValueClass::Any(b) = created_key(Some(Id::from(2u64))) else { panic!() };
assert_eq!(server.key, b"Pc");
assert_ne!(a.key, b.key);
assert!(a.key.starts_with(b"Pc"));
@@ -46,21 +46,6 @@ const ADMIN_GRANTS: &[Permission] = &[
Permission::SysLegalHoldUpdate,
Permission::SysLegalHoldExport,
Permission::SysComplianceGet,
Permission::SysMailRuleGet,
Permission::SysMailRuleUpdate,
Permission::SysDlpPolicyGet,
Permission::SysDlpPolicyUpdate,
Permission::SysDlpReviewGet,
Permission::SysDlpReviewUpdate,
];
/// Granted to the server-level Compliance Officer role once it exists:
/// seeing DLP rules and reviewing held mail (dlp-and-mail-flow-rules spec,
/// §2.8, settled answer 4). A new install's role has them from the start.
const OFFICER_GRANTS: &[Permission] = &[
Permission::SysDlpPolicyGet,
Permission::SysDlpReviewGet,
Permission::SysDlpReviewUpdate,
];
/// Granted to the default tenant administrator roles: reading and exporting
@@ -80,16 +65,13 @@ const TENANT_GRANTS: &[Permission] = &[
enum Audience {
Admin,
Tenant,
Officer,
}
fn granted_key(permission: Permission, audience: Audience) -> ValueClass {
let mut key = b"Pg".to_vec();
// Admin grants keep the key they were first recorded under
match audience {
Audience::Admin => {}
Audience::Tenant => key.extend_from_slice(b"tenant:"),
Audience::Officer => key.extend_from_slice(b"officer:"),
if audience == Audience::Tenant {
key.extend_from_slice(b"tenant:");
}
key.extend_from_slice(permission.as_str().as_bytes());
ValueClass::Any(AnyClass {
@@ -100,8 +82,7 @@ fn granted_key(permission: Permission, audience: Audience) -> ValueClass {
pub(crate) async fn grant_new_admin_permissions(bp: &mut Bootstrap) -> trc::Result<()> {
grant(bp, Audience::Admin, ADMIN_GRANTS).await?;
grant(bp, Audience::Tenant, TENANT_GRANTS).await?;
grant(bp, Audience::Officer, OFFICER_GRANTS).await
grant(bp, Audience::Tenant, TENANT_GRANTS).await
}
async fn grant(bp: &mut Bootstrap, audience: Audience, grants: &[Permission]) -> trc::Result<()> {
@@ -120,47 +101,39 @@ async fn grant(bp: &mut Bootstrap, audience: Audience, grants: &[Permission]) ->
if pending.is_empty() {
return Ok(());
}
// The officer role is the one the server made, if it has made it yet: a
// new install makes it after this, with the permissions already in it
let admin_roles: Vec<Id> = if audience == Audience::Officer {
super::compliance_roles::server_role(&bp.data_store)
.await?
.into_iter()
.collect()
} else {
// An administrator's default roles include the plain User role, which
// every user also holds; only roles that are the audience's alone get it
bp.registry
.object::<Authentication>(Id::singleton())
.await?
.map(|auth| {
let (own, shared) = match audience {
Audience::Admin => (
auth.default_admin_role_ids.as_slice(),
[
auth.default_user_role_ids.as_slice(),
auth.default_group_role_ids.as_slice(),
auth.default_tenant_role_ids.as_slice(),
]
.concat(),
),
Audience::Tenant | Audience::Officer => (
// An administrator's default roles include the plain User role, which
// every user also holds; only roles that are the audience's alone get it
let admin_roles: Vec<Id> = bp
.registry
.object::<Authentication>(Id::singleton())
.await?
.map(|auth| {
let (own, shared) = match audience {
Audience::Admin => (
auth.default_admin_role_ids.as_slice(),
[
auth.default_user_role_ids.as_slice(),
auth.default_group_role_ids.as_slice(),
auth.default_tenant_role_ids.as_slice(),
[
auth.default_user_role_ids.as_slice(),
auth.default_group_role_ids.as_slice(),
auth.default_admin_role_ids.as_slice(),
]
.concat(),
),
};
own.iter()
.filter(|id| !shared.contains(id))
.copied()
.collect()
})
.unwrap_or_default()
};
]
.concat(),
),
Audience::Tenant => (
auth.default_tenant_role_ids.as_slice(),
[
auth.default_user_role_ids.as_slice(),
auth.default_group_role_ids.as_slice(),
auth.default_admin_role_ids.as_slice(),
]
.concat(),
),
};
own.iter()
.filter(|id| !shared.contains(id))
.copied()
.collect()
})
.unwrap_or_default();
// Fetched by id: the registry's listing doesn't reach stored roles
for role_id in admin_roles {
let Some(stored) = bp
-5
View File
@@ -21,11 +21,6 @@ base64 = "0.23"
sha2 = "0.11"
flate2 = "1.1"
tokio = { version = "1.53", features = ["sync", "rt"] }
# inbuxa: DLP detectors and attachment text (dlp-and-mail-flow-rules spec)
regex = "1.13.1"
aho-corasick = "1.1"
zip = "8.6"
quick-xml = "0.41"
[dev-dependencies]
tokio = { version = "1.53", features = ["macros", "rt"] }
-1
View File
@@ -23,7 +23,6 @@ pub mod audit;
pub mod branding;
pub mod hold;
pub mod lock;
pub mod mailflow;
pub mod masked_email;
pub mod privacy;
pub mod security;
-53
View File
@@ -1,53 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! The compiled rules, kept per node so a message doesn't read the store.
//! A change made on this node applies at once; one made on another node
//! within [`TTL`], when the copy here is next refreshed.
use super::{engine::Compiled, rules};
use std::{
sync::{Arc, RwLock},
time::{Duration, Instant},
};
use store::Store;
/// How long a node keeps its copy before reading the rules again.
pub const TTL: Duration = Duration::from_secs(30);
static CACHE: RwLock<Option<(Instant, Arc<Compiled>)>> = RwLock::new(None);
/// Forgets the copy, so the next message reads the rules again.
pub fn invalidate() {
if let Ok(mut cache) = CACHE.write() {
*cache = None;
}
}
/// The enabled rules, compiled. A rule that no longer compiles is left out
/// and reported, once per refresh.
pub async fn compiled(data: &Store) -> trc::Result<Arc<Compiled>> {
if let Ok(cache) = CACHE.read()
&& let Some((at, compiled)) = cache.as_ref()
&& at.elapsed() < TTL
{
return Ok(compiled.clone());
}
let (compiled, skipped) = Compiled::new(&rules::all(data).await?);
for (id, reason) in skipped {
trc::event!(
Store(trc::StoreEvent::DataCorruption),
Id = u64::from(id),
Reason = reason,
Details = "Mail rule skipped: it no longer compiles"
);
}
let compiled = Arc::new(compiled);
if let Ok(mut cache) = CACHE.write() {
*cache = Some((Instant::now(), compiled.clone()));
}
Ok(compiled)
}
@@ -1,49 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! African identifiers (§2.3): South Africa's ID number.
use super::{Detector, Findings, Region, Strength, checks, valid_short_date};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[Detector::new(
"za-id",
"South Africa: ID number",
Region::Africa,
Strength::Checked,
za_id,
)];
/// Birth date `YYMMDD`, four digits, citizenship (0, 1 or 2), 8 or 9, a Luhn
/// check digit. The date and the two fixed digits make it strong enough to
/// count alone.
static ZA_ID: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(r"\b(\d{2})(\d{2})(\d{2})\d{4}[012][89]\d\b").expect("detector pattern")
});
fn za_id(text: &str, findings: &mut Findings) {
for c in ZA_ID.captures_iter(text) {
let n = &c[0];
let num = |s: &str| s.parse::<u32>().unwrap_or(0);
if valid_short_date(num(&c[1]), num(&c[2]), num(&c[3])) && checks::luhn(n) {
findings.insert(n);
}
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
#[test]
fn south_africa() {
let detector = by_id("za-id").unwrap();
assert_eq!(detector.count("ID 8001015009087"), 1);
assert_eq!(detector.count("8001015009088"), 0);
assert_eq!(detector.count("8013015009087"), 0);
}
}
@@ -1,172 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Identifiers from the Americas outside the US and Canada (§2.3): Brazil's
//! CPF and CNPJ, and Mexico's CURP.
use super::{Detector, Findings, Region, Strength, digit_values, valid_short_date, word_near};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"br-cpf",
"Brazil: CPF",
Region::Americas,
Strength::Checked,
br_cpf,
),
Detector::new(
"br-cnpj",
"Brazil: CNPJ",
Region::Americas,
Strength::Checked,
br_cnpj,
),
Detector::new(
"mx-curp",
"Mexico: CURP",
Region::Americas,
Strength::Checked,
mx_curp,
),
];
fn re(pattern: &str) -> Regex {
Regex::new(pattern).expect("detector pattern")
}
/// Brazil's mod 11 check digit over `digits` with `weights`.
fn br_check(digits: &[u32], weights: &[u32]) -> u32 {
match digits.iter().zip(weights).map(|(a, w)| a * w).sum::<u32>() % 11 {
0 | 1 => 0,
r => 11 - r,
}
}
/// `111.444.777-35`, or eleven bare digits.
static CPF: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{3}(\.?)\d{3}(\.?)\d{3}(-?)\d{2}\b"));
pub fn cpf_valid(n: &str) -> bool {
let d = digit_values(n);
// A run of one digit passes the arithmetic but is never issued
d.len() == 11
&& d.iter().any(|x| *x != d[0])
&& br_check(&d[..9], &[10, 9, 8, 7, 6, 5, 4, 3, 2]) == d[9]
&& br_check(&d[..10], &[11, 10, 9, 8, 7, 6, 5, 4, 3, 2]) == d[10]
}
const CPF_WORDS: &[&str] = &[
"cpf",
"cadastro de pessoas físicas",
"cadastro de pessoa física",
];
fn br_cpf(text: &str, findings: &mut Findings) {
for c in CPF.captures_iter(text) {
let whole = c.get(0).unwrap();
let written = &c[1] == "." && &c[2] == "." && &c[3] == "-";
let n: String = whole
.as_str()
.chars()
.filter(char::is_ascii_digit)
.collect();
if cpf_valid(&n) && (written || word_near(text, whole.start(), whole.end(), CPF_WORDS)) {
findings.insert(n);
}
}
}
/// `11.222.333/0001-81`, or fourteen bare digits.
static CNPJ: LazyLock<Regex> =
LazyLock::new(|| re(r"\b\d{2}(\.?)\d{3}(\.?)\d{3}(/?)\d{4}(-?)\d{2}\b"));
pub fn cnpj_valid(n: &str) -> bool {
let d = digit_values(n);
d.len() == 14
&& d.iter().any(|x| *x != d[0])
&& br_check(&d[..12], &[5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2]) == d[12]
&& br_check(&d[..13], &[6, 5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2]) == d[13]
}
const CNPJ_WORDS: &[&str] = &["cnpj", "cadastro nacional da pessoa jurídica"];
fn br_cnpj(text: &str, findings: &mut Findings) {
for c in CNPJ.captures_iter(text) {
let whole = c.get(0).unwrap();
let written = &c[1] == "." && &c[2] == "." && &c[3] == "/" && &c[4] == "-";
let n: String = whole
.as_str()
.chars()
.filter(char::is_ascii_digit)
.collect();
if cnpj_valid(&n) && (written || word_near(text, whole.start(), whole.end(), CNPJ_WORDS)) {
findings.insert(n);
}
}
}
/// Four letters, the birth date, sex (H, M or X), the state, three
/// consonants, a character that tells the century apart, the check digit.
static CURP: LazyLock<Regex> = LazyLock::new(|| {
re(r"(?i)\b[A-Z]{4}(\d{2})(\d{2})(\d{2})[HMX][A-Z]{2}[B-DF-HJ-NP-TV-Z]{3}[A-Z0-9]\d\b")
});
/// RENAPO's check: each character's place in `0-9 A-N Ñ O-Z`, weighted 18
/// down to 2; the digit is 10 minus the sum mod 10 (10 becomes 0).
pub fn curp_valid(curp: &str) -> bool {
const ALPHABET: &str = "0123456789ABCDEFGHIJKLMNÑOPQRSTUVWXYZ";
let mut sum = 0u32;
for (i, c) in curp.chars().take(17).enumerate() {
let Some(value) = ALPHABET.chars().position(|a| a == c) else {
return false;
};
sum += value as u32 * (18 - i as u32);
}
curp.chars().nth(17).and_then(|c| c.to_digit(10)) == Some((10 - sum % 10) % 10)
}
fn mx_curp(text: &str, findings: &mut Findings) {
for c in CURP.captures_iter(text) {
let curp = c[0].to_ascii_uppercase();
if valid_short_date(num(&c[1]), num(&c[2]), num(&c[3])) && curp_valid(&curp) {
findings.insert(curp);
}
}
}
fn num(s: &str) -> u32 {
s.parse().unwrap_or(0)
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn brazil() {
assert_eq!(count("br-cpf", "CPF 111.444.777-35"), 1);
assert_eq!(count("br-cpf", "111.444.777-36"), 0);
assert_eq!(count("br-cpf", "pedido 11144477735"), 0);
assert_eq!(count("br-cpf", "cpf: 11144477735"), 1);
assert_eq!(count("br-cpf", "CPF 111.111.111-11"), 0);
assert_eq!(count("br-cnpj", "11.222.333/0001-81"), 1);
assert_eq!(count("br-cnpj", "11.222.333/0001-82"), 0);
assert_eq!(count("br-cnpj", "CNPJ 11222333000181"), 1);
}
#[test]
fn mexico() {
// python-stdnum's documented example
assert_eq!(count("mx-curp", "CURP BOXW310820HNERXN09"), 1);
assert_eq!(count("mx-curp", "BOXW310820HNERXN08"), 0);
assert_eq!(count("mx-curp", "BOXW311320HNERXN09"), 0);
}
}
@@ -1,511 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Detectors that aren't tied to one country (§2.3, region "Any").
use super::{
Detector, Findings, Region, Strength, checks, digits, stands_alone, valid_date, word_near,
};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"payment-card",
"Payment card number",
Region::Any,
Strength::Checked,
payment_card,
),
Detector::new("iban", "IBAN", Region::Any, Strength::Checked, iban),
Detector::new(
"swift-bic",
"SWIFT/BIC code",
Region::Any,
Strength::NeedsWord,
swift_bic,
),
Detector::new(
"email-addresses",
"Email addresses",
Region::Any,
Strength::Checked,
email_addresses,
),
Detector::new(
"phone-numbers",
"Phone numbers",
Region::Any,
Strength::NeedsWord,
phone_numbers,
),
Detector::new(
"date-of-birth",
"Date of birth",
Region::Any,
Strength::NeedsWord,
date_of_birth,
),
Detector::new(
"passport",
"Passport number",
Region::Any,
Strength::NeedsWord,
passport,
),
Detector::new(
"private-key",
"Private key",
Region::Any,
Strength::Checked,
private_key,
),
Detector::new(
"credentials",
"Cloud and service credentials",
Region::Any,
Strength::Checked,
credentials,
),
];
fn re(pattern: &str) -> Regex {
Regex::new(pattern).expect("detector pattern")
}
// --- Payment cards --------------------------------------------------------
/// Issuer prefixes (ISO/IEC 7812 IINs) and the lengths each network issues.
fn card_network(number: &str) -> bool {
let len = number.len();
let prefix = |n: usize| number[..n].parse::<u32>().unwrap_or(0);
match number.as_bytes()[0] {
// Visa
b'4' => matches!(len, 13 | 16 | 19),
b'5' => {
// Mastercard 51–55; Maestro 50, 56–58
(51..=55).contains(&prefix(2)) && len == 16
|| matches!(prefix(2), 50 | 56..=58) && (12..=19).contains(&len)
}
// Mastercard 2221–2720
b'2' => (2221..=2720).contains(&prefix(4)) && len == 16,
b'3' => {
// American Express 34, 37; JCB 3528–3589; Diners 300–305, 36, 38, 39
matches!(prefix(2), 34 | 37) && len == 15
|| (3528..=3589).contains(&prefix(4)) && (16..=19).contains(&len)
|| ((300..=305).contains(&prefix(3)) || matches!(prefix(2), 36 | 38 | 39))
&& (14..=19).contains(&len)
}
// Discover 6011, 644–649, 65; UnionPay 62; Maestro 6x
b'6' => (12..=19).contains(&len),
_ => false,
}
}
fn is_card(number: &str) -> bool {
(12..=19).contains(&number.len()) && card_network(number) && checks::luhn(number)
}
static CARD: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d(?:[ -]?\d){11,18}\b"));
fn payment_card(text: &str, findings: &mut Findings) {
for m in CARD.find_iter(text) {
if !stands_alone(text, m.start(), m.end()) {
continue;
}
let whole = digits(m.as_str());
if is_card(&whole) {
findings.insert(whole);
continue;
}
// Two numbers side by side ("4242 4242 4242 4242 2031"): try each
// run of whole groups
let groups: Vec<String> = m.as_str().split([' ', '-']).map(digits).collect();
'runs: for from in 0..groups.len() {
let mut number = String::new();
for group in &groups[from..] {
number.push_str(group);
if is_card(&number) {
findings.insert(number);
break 'runs;
}
}
}
}
}
// --- IBAN -----------------------------------------------------------------
static IBAN: LazyLock<Regex> =
LazyLock::new(|| re(r"\b[A-Za-z]{2}\d{2}(?:[ ]?[A-Za-z0-9]){11,30}"));
fn iban(text: &str, findings: &mut Findings) {
// The pattern can run on into the next words, even the next IBAN: after
// each hit, look again from where that IBAN ended
let mut from = 0;
while let Some(m) = IBAN.find_at(text, from) {
from = m.start() + 1;
let compact = m.as_str().replace(' ', "").to_ascii_uppercase();
let Some(len) = checks::iban_length(&compact[..2]) else {
continue;
};
if compact.len() < len {
continue;
}
// Where the country's length ends in the text, spaces counted
let mut seen = 0;
let Some(end) = m
.as_str()
.char_indices()
.find(|(_, c)| {
if *c != ' ' {
seen += 1;
}
seen == len
})
.map(|(i, c)| m.start() + i + c.len_utf8())
else {
continue;
};
let candidate = &compact[..len];
if stands_alone(text, m.start(), end) && checks::iban(candidate) {
findings.insert(candidate);
from = end;
}
}
}
// --- SWIFT/BIC ------------------------------------------------------------
static BIC: LazyLock<Regex> =
LazyLock::new(|| re(r"\b[A-Z]{4}[A-Z]{2}[A-Z0-9]{2}(?:[A-Z0-9]{3})?\b"));
const BIC_WORDS: &[&str] = &[
"swift",
"bic",
"swift/bic",
"bank",
"banque",
"bankverbindung",
];
fn swift_bic(text: &str, findings: &mut Findings) {
for m in BIC.find_iter(text) {
let code = m.as_str();
if checks::is_country(&code[4..6]) && word_near(text, m.start(), m.end(), BIC_WORDS) {
findings.insert(code);
}
}
}
// --- Contact lists --------------------------------------------------------
static EMAIL: LazyLock<Regex> =
LazyLock::new(|| re(r"(?i)\b[a-z0-9._%+-]+@[a-z0-9-]+(?:\.[a-z0-9-]+)*\.[a-z]{2,}\b"));
fn email_addresses(text: &str, findings: &mut Findings) {
for m in EMAIL.find_iter(text) {
findings.insert(m.as_str().to_lowercase());
}
}
/// International form: found alone. National form: only with a word.
static PHONE_INTL: LazyLock<Regex> = LazyLock::new(|| re(r"\+\d{1,3}(?:[ .-]?\(?\d{1,4}\)?){2,5}"));
static PHONE_NATIONAL: LazyLock<Regex> =
LazyLock::new(|| re(r"\(?\d{2,4}\)?[ .-]\d{3,4}[ .-]\d{3,4}"));
const PHONE_WORDS: &[&str] = &[
"phone",
"tel",
"telephone",
"mobile",
"cell",
"fax",
"telefon",
"téléphone",
"teléfono",
"telefono",
"handy",
"portable",
"móvil",
"cellulare",
"mobiel",
];
fn phone_numbers(text: &str, findings: &mut Findings) {
let mut international = Vec::new();
for m in PHONE_INTL.find_iter(text) {
let number = digits(m.as_str());
if (8..=15).contains(&number.len()) && stands_alone(text, m.start() + 1, m.end()) {
findings.insert(number);
international.push(m.range());
}
}
for m in PHONE_NATIONAL.find_iter(text) {
let number = digits(m.as_str());
// Not the tail of an international number already counted
if international.iter().any(|r| r.contains(&m.start())) {
continue;
}
if (9..=11).contains(&number.len())
&& stands_alone(text, m.start(), m.end())
&& !text[..m.start()].ends_with('+')
&& word_near(text, m.start(), m.end(), PHONE_WORDS)
{
findings.insert(number);
}
}
}
// --- Date of birth --------------------------------------------------------
static DATE_ISO: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{4})-(\d{2})-(\d{2})\b"));
static DATE_NUMERIC: LazyLock<Regex> =
LazyLock::new(|| re(r"\b(\d{1,2})[./-](\d{1,2})[./-](\d{4})\b"));
static DATE_WORDS: LazyLock<Regex> = LazyLock::new(|| {
re(
r"(?i)\b(?:(\d{1,2})\s+(jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)[a-z]*\.?,?\s+(\d{4})|(jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)[a-z]*\.?\s+(\d{1,2}),?\s+(\d{4}))\b",
)
});
const BIRTH_WORDS: &[&str] = &[
"born",
"birth",
"dob",
"d.o.b",
"birthday",
"birthdate",
"geburtsdatum",
"geboren",
"naissance",
"né le",
"née le",
"nacimiento",
"nacido",
"nacida",
"nascita",
"nato il",
"nata il",
"geboortedatum",
"födelsedatum",
"fødselsdato",
"syntymäaika",
"urodzenia",
"nascimento",
];
fn month_number(name: &str) -> u32 {
const MONTHS: [&str; 12] = [
"jan", "feb", "mar", "apr", "may", "jun", "jul", "aug", "sep", "oct", "nov", "dec",
];
let name = name.to_lowercase();
MONTHS
.iter()
.position(|m| *m == name)
.map_or(0, |i| i as u32 + 1)
}
fn date_of_birth(text: &str, findings: &mut Findings) {
let mut add = |start: usize, end: usize, key: String| {
if word_near(text, start, end, BIRTH_WORDS) {
findings.insert(key);
}
};
let num = |s: &str| s.parse::<u32>().unwrap_or(0);
for c in DATE_ISO.captures_iter(text) {
let (y, m, d) = (num(&c[1]), num(&c[2]), num(&c[3]));
let whole = c.get(0).unwrap();
if valid_date(y, m, d) {
add(whole.start(), whole.end(), format!("{y:04}{m:02}{d:02}"));
}
}
for c in DATE_NUMERIC.captures_iter(text) {
let (a, b, y) = (num(&c[1]), num(&c[2]), num(&c[3]));
let whole = c.get(0).unwrap();
// Day first or month first: either reading that is a real date
if valid_date(y, b, a) || valid_date(y, a, b) {
add(whole.start(), whole.end(), whole.as_str().to_string());
}
}
for c in DATE_WORDS.captures_iter(text) {
let whole = c.get(0).unwrap();
let (d, m, y) = match (c.get(1), c.get(4)) {
(Some(d), _) => (num(d.as_str()), month_number(&c[2]), num(&c[3])),
(_, Some(m)) => (num(&c[5]), month_number(m.as_str()), num(&c[6])),
_ => continue,
};
if valid_date(y, m, d) {
add(whole.start(), whole.end(), format!("{y:04}{m:02}{d:02}"));
}
}
}
// --- Passport -------------------------------------------------------------
static PASSPORT: LazyLock<Regex> = LazyLock::new(|| re(r"\b[A-Z0-9]{6,9}\b"));
const PASSPORT_WORDS: &[&str] = &[
"passport",
"passeport",
"reisepass",
"pasaporte",
"passaporto",
"paspoort",
"passnummer",
"pass-nr",
"passport no",
"pasaporte n.º",
"passaporte",
];
fn passport(text: &str, findings: &mut Findings) {
for m in PASSPORT.find_iter(text) {
let value = m.as_str();
if value.bytes().filter(u8::is_ascii_digit).count() >= 5
&& word_near(text, m.start(), m.end(), PASSPORT_WORDS)
{
findings.insert(value);
}
}
}
// --- Keys and credentials -------------------------------------------------
static PRIVATE_KEY: LazyLock<Regex> = LazyLock::new(|| {
re(
r"-----BEGIN (?:(?:RSA|EC|DSA|OPENSSH|ENCRYPTED|PGP) )?PRIVATE KEY(?: BLOCK)?-----\s*([A-Za-z0-9+/=:\s-]{0,64})",
)
});
fn private_key(text: &str, findings: &mut Findings) {
for c in PRIVATE_KEY.captures_iter(text) {
// Each key once, by the start of its body
let body: String = c[1].chars().filter(|c| !c.is_whitespace()).collect();
let whole = c.get(0).unwrap();
findings.insert(if body.is_empty() {
format!("@{}", whole.start())
} else {
body
});
}
}
/// Published token formats: AWS access key IDs, GitHub tokens, Slack
/// tokens, Stripe live secret and restricted keys, Google API keys.
static CREDENTIAL: LazyLock<Regex> = LazyLock::new(|| {
re(concat!(
r"\b(?:",
r"(?:AKIA|ASIA|ABIA|ACCA)[A-Z0-9]{16}",
r"|gh[pousr]_[A-Za-z0-9]{36}",
r"|github_pat_[A-Za-z0-9_]{82}",
r"|xox[abposr]-[A-Za-z0-9-]{10,72}",
r"|(?:sk|rk)_live_[A-Za-z0-9]{24,99}",
r"|AIza[0-9A-Za-z_-]{35}",
r")\b"
))
});
fn credentials(text: &str, findings: &mut Findings) {
for m in CREDENTIAL.find_iter(text) {
findings.insert(m.as_str());
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn payment_cards() {
// Networks' and processors' published test numbers
let text = "Visa 4242 4242 4242 4242, MC 5555-5555-5555-4444, Amex 378282246310005, \
Discover 6011111111111117, JCB 3566002020360505, Diners 30569309025904, \
UnionPay 6200000000000005, Mastercard 2-series 2223003122003222";
assert_eq!(count("payment-card", text), 8);
// Luhn fails, wrong network length, inside a longer number
assert_eq!(count("payment-card", "4242424242424241"), 0);
assert_eq!(count("payment-card", "378282246310005 0"), 1);
assert_eq!(count("payment-card", "order 94242424242424242 shipped"), 0);
// The same number twice counts once
assert_eq!(
count("payment-card", "4242424242424242 and 4242-4242-4242-4242"),
1
);
// A card followed by a year
assert_eq!(count("payment-card", "card 4242 4242 4242 4242 2031"), 1);
}
#[test]
fn ibans() {
let text =
"Pay GB29 NWBK 6016 1331 9268 19 or de89370400440532013000 (NL91ABNA0417164300).";
assert_eq!(count("iban", text), 3);
assert_eq!(count("iban", "GB29 NWBK 6016 1331 9268 18"), 0);
// Runs into the next word: still found at the country's length
assert_eq!(count("iban", "IBAN NL91ABNA0417164300 BIC ABNANL2A"), 1);
}
#[test]
fn swift_codes_need_a_word() {
assert_eq!(count("swift-bic", "SWIFT: DEUTDEFF500"), 1);
assert_eq!(count("swift-bic", "BIC NWBKGB2L"), 1);
assert_eq!(count("swift-bic", "HAPPYDAYS DEUTDEFF"), 0);
// Not a country in positions 5–6
assert_eq!(count("swift-bic", "BIC DEUTZZFF"), 0);
}
#[test]
fn email_and_phone_lists() {
let list = "[email protected], [email protected], [email protected], [email protected]";
assert_eq!(count("email-addresses", list), 3);
assert_eq!(
count("phone-numbers", "+44 20 7946 0958, +1 (415) 555-2671"),
2
);
assert_eq!(count("phone-numbers", "call 020 7946 0958"), 0);
assert_eq!(count("phone-numbers", "Tel: 020 7946 0958"), 1);
assert_eq!(count("phone-numbers", "invoice 020 7946 0958"), 0);
// One number, not also its national tail
assert_eq!(count("phone-numbers", "Tel: +44 20 7946 0958"), 1);
}
#[test]
fn dates_of_birth() {
assert_eq!(count("date-of-birth", "DOB: 1984-02-29"), 1);
assert_eq!(count("date-of-birth", "Geburtsdatum 31.12.1970"), 1);
assert_eq!(count("date-of-birth", "born on March 3, 1962"), 1);
assert_eq!(count("date-of-birth", "date of birth 3 Mar 1962"), 1);
// Not a real date, no word, a meeting
assert_eq!(count("date-of-birth", "DOB: 1985-02-29"), 0);
assert_eq!(count("date-of-birth", "invoice 1984-02-29"), 0);
assert_eq!(count("date-of-birth", "Meeting on 12/05/2026"), 0);
}
#[test]
fn passports_need_a_word() {
assert_eq!(count("passport", "Passport number: 533380006"), 1);
assert_eq!(count("passport", "Reisepass C01X00T47"), 1);
assert_eq!(count("passport", "Order 533380006 shipped"), 0);
// Mostly letters: a word, not a number
assert_eq!(count("passport", "passport PASSWORD"), 0);
}
#[test]
fn keys_and_credentials() {
let key = "-----BEGIN OPENSSH PRIVATE KEY-----\nb3BlbnNzaC1rZXktdjEAAAAABG5vbmUAAAAEbm9uZQ\n-----END OPENSSH PRIVATE KEY-----";
assert_eq!(count("private-key", key), 1);
assert_eq!(count("private-key", "-----BEGIN PUBLIC KEY-----\nMFkw"), 0);
// Documentation examples of each format
let tokens = "AKIAIOSFODNN7EXAMPLE ghp_0123456789abcdefghijklmnopqrstuvwxyz \
AIzaSyA-0123456789abcdefghijklmnopqrstu";
assert_eq!(count("credentials", tokens), 3);
assert_eq!(count("credentials", "AKIA123 ghp_short"), 0);
}
}
@@ -1,281 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Asian identifiers (§2.3): India's Aadhaar and PAN, China's resident ID,
//! Japan's My Number, Singapore's NRIC and FIN, and South Korea's resident
//! registration number.
use super::{
Detector, Findings, Region, Strength, digit_values, valid_date, valid_short_date, word_near,
};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"in-aadhaar",
"India: Aadhaar",
Region::Asia,
Strength::Checked,
in_aadhaar,
),
Detector::new(
"in-pan",
"India: PAN",
Region::Asia,
Strength::NeedsWord,
in_pan,
),
Detector::new(
"cn-resident-id",
"China: resident ID",
Region::Asia,
Strength::Checked,
cn_resident_id,
),
Detector::new(
"jp-my-number",
"Japan: My Number",
Region::Asia,
Strength::Checked,
jp_my_number,
),
Detector::new(
"sg-nric",
"Singapore: NRIC and FIN",
Region::Asia,
Strength::Checked,
sg_nric,
),
Detector::new(
"kr-rrn",
"South Korea: resident registration number",
Region::Asia,
Strength::NeedsWord,
kr_rrn,
),
];
fn re(pattern: &str) -> Regex {
Regex::new(pattern).expect("detector pattern")
}
/// Twelve digits written in fours, or bare.
static TWELVE: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{4})( ?)(\d{4})( ?)(\d{4})\b"));
const VERHOEFF_D: [[u8; 10]; 10] = [
[0, 1, 2, 3, 4, 5, 6, 7, 8, 9],
[1, 2, 3, 4, 0, 6, 7, 8, 9, 5],
[2, 3, 4, 0, 1, 7, 8, 9, 5, 6],
[3, 4, 0, 1, 2, 8, 9, 5, 6, 7],
[4, 0, 1, 2, 3, 9, 5, 6, 7, 8],
[5, 9, 8, 7, 6, 0, 4, 3, 2, 1],
[6, 5, 9, 8, 7, 1, 0, 4, 3, 2],
[7, 6, 5, 9, 8, 2, 1, 0, 4, 3],
[8, 7, 6, 5, 9, 3, 2, 1, 0, 4],
[9, 8, 7, 6, 5, 4, 3, 2, 1, 0],
];
const VERHOEFF_P: [[u8; 10]; 8] = [
[0, 1, 2, 3, 4, 5, 6, 7, 8, 9],
[1, 5, 7, 6, 2, 8, 3, 0, 9, 4],
[5, 8, 0, 3, 7, 9, 6, 1, 4, 2],
[8, 9, 1, 6, 0, 4, 3, 5, 2, 7],
[9, 4, 5, 3, 1, 2, 6, 8, 7, 0],
[4, 2, 8, 6, 5, 7, 3, 9, 0, 1],
[2, 7, 9, 3, 8, 0, 6, 4, 1, 5],
[7, 0, 4, 6, 9, 1, 3, 2, 5, 8],
];
/// The Verhoeff check (dihedral group D5).
pub fn verhoeff(n: &str) -> bool {
let mut c = 0u8;
for (i, b) in n.bytes().rev().enumerate() {
c = VERHOEFF_D[c as usize][VERHOEFF_P[i % 8][(b - b'0') as usize] as usize];
}
c == 0
}
const AADHAAR_WORDS: &[&str] = &["aadhaar", "aadhar", "uidai", "uid"];
fn in_aadhaar(text: &str, findings: &mut Findings) {
for c in TWELVE.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = &c[2] == " " && &c[4] == " ";
// Never starts with 0 or 1
if !n.starts_with(['0', '1'])
&& verhoeff(&n)
&& (written || word_near(text, whole.start(), whole.end(), AADHAAR_WORDS))
{
findings.insert(n);
}
}
}
/// Five letters (the fourth names the holder's type), four digits, a letter.
static PAN: LazyLock<Regex> = LazyLock::new(|| re(r"\b[A-Z]{3}[ABCFGHLJPTK][A-Z]\d{4}[A-Z]\b"));
const PAN_WORDS: &[&str] = &["pan", "pan card", "permanent account number", "income tax"];
fn in_pan(text: &str, findings: &mut Findings) {
for m in PAN.find_iter(text) {
if word_near(text, m.start(), m.end(), PAN_WORDS) {
findings.insert(m.as_str());
}
}
}
/// Region, birth date `YYYYMMDD`, sequence, then the ISO 7064 MOD 11-2
/// check (0–9 or X).
static CN_ID: LazyLock<Regex> =
LazyLock::new(|| re(r"(?i)\b[1-8]\d{5}(\d{4})(\d{2})(\d{2})\d{3}[\dX]\b"));
pub fn cn_id_valid(id: &str) -> bool {
const WEIGHTS: [u32; 17] = [7, 9, 10, 5, 8, 4, 2, 1, 6, 3, 7, 9, 10, 5, 8, 4, 2];
const CHECKS: &[u8] = b"10X98765432";
let sum: u32 = digit_values(&id[..17])
.iter()
.zip(WEIGHTS)
.map(|(a, w)| a * w)
.sum();
CHECKS[(sum % 11) as usize] == id.as_bytes()[17].to_ascii_uppercase()
}
fn cn_resident_id(text: &str, findings: &mut Findings) {
for c in CN_ID.captures_iter(text) {
let id = c[0].to_ascii_uppercase();
let (y, m, d) = (num(&c[1]), num(&c[2]), num(&c[3]));
if valid_date(y, m, d) && cn_id_valid(&id) {
findings.insert(id);
}
}
}
/// My Number: weights 2–7 then 2–6 from the right; a remainder of 0 or 1
/// gives 0, else 11 minus it.
pub fn my_number_valid(n: &str) -> bool {
let d = digit_values(n);
let sum: u32 = (1..=11)
.map(|i| d[11 - i] * if i <= 6 { i as u32 + 1 } else { i as u32 - 5 })
.sum();
let check = match sum % 11 {
0 | 1 => 0,
r => 11 - r,
};
check == d[11]
}
const MY_NUMBER_WORDS: &[&str] = &[
"my number",
"mynumber",
"マイナンバー",
"個人番号",
"kojin bango",
];
fn jp_my_number(text: &str, findings: &mut Findings) {
for c in TWELVE.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = &c[2] == " " && &c[4] == " ";
if my_number_valid(&n)
&& (written || word_near(text, whole.start(), whole.end(), MY_NUMBER_WORDS))
{
findings.insert(n);
}
}
}
static NRIC: LazyLock<Regex> = LazyLock::new(|| re(r"(?i)\b([STFGM])(\d{7})([A-Z])\b"));
/// Weights 2, 7, 6, 5, 4, 3, 2; T and G add 4, M adds 3; each series has its
/// own table of check letters.
fn nric_valid(prefix: u8, digits: &str, check: u8) -> bool {
let sum: u32 = digit_values(digits)
.iter()
.zip([2, 7, 6, 5, 4, 3, 2])
.map(|(a, w)| a * w)
.sum::<u32>()
+ match prefix {
b'T' | b'G' => 4,
b'M' => 3,
_ => 0,
};
let table: &[u8] = match prefix {
b'S' | b'T' => b"JZIHGFEDCBA",
b'F' | b'G' => b"XWUTRQPNMLK",
_ => b"KLJNPQRTUWX",
};
table[(sum % 11) as usize] == check
}
fn sg_nric(text: &str, findings: &mut Findings) {
for c in NRIC.captures_iter(text) {
let id = c[0].to_ascii_uppercase();
let bytes = id.as_bytes();
if nric_valid(bytes[0], &c[2], bytes[8]) {
findings.insert(id);
}
}
}
/// `YYMMDD-GNNNNNN`, the seventh digit giving sex and century.
static RRN: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{2})(\d{2})(\d{2})-?([1-8])\d{6}\b"));
const RRN_WORDS: &[&str] = &["주민등록번호", "주민번호", "resident registration", "rrn"];
fn kr_rrn(text: &str, findings: &mut Findings) {
for c in RRN.captures_iter(text) {
let whole = c.get(0).unwrap();
if valid_short_date(num(&c[1]), num(&c[2]), num(&c[3]))
&& word_near(text, whole.start(), whole.end(), RRN_WORDS)
{
findings.insert(whole.as_str().replace('-', ""));
}
}
}
fn num(s: &str) -> u32 {
s.parse().unwrap_or(0)
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn india() {
assert_eq!(count("in-aadhaar", "2345 6789 0124"), 1);
assert_eq!(count("in-aadhaar", "2345 6789 0125"), 0);
assert_eq!(count("in-aadhaar", "order 234567890124"), 0);
assert_eq!(count("in-aadhaar", "Aadhaar 234567890124"), 1);
assert_eq!(count("in-pan", "PAN: ABCPE1234F"), 1);
assert_eq!(count("in-pan", "ABCPE1234F"), 0);
}
#[test]
fn china_japan() {
assert_eq!(count("cn-resident-id", "11010519491231002X"), 1);
assert_eq!(count("cn-resident-id", "110105194912310021"), 0);
assert_eq!(count("cn-resident-id", "11010519491331002X"), 0);
assert_eq!(count("jp-my-number", "1234 5678 9018"), 1);
assert_eq!(count("jp-my-number", "1234 5678 9017"), 0);
assert_eq!(count("jp-my-number", "マイナンバー 123456789018"), 1);
}
#[test]
fn singapore_korea() {
assert_eq!(count("sg-nric", "S1234567D and T1234567J"), 2);
assert_eq!(count("sg-nric", "S1234567E"), 0);
assert_eq!(count("kr-rrn", "주민등록번호 800101-1234567"), 1);
assert_eq!(count("kr-rrn", "800101-1234567"), 0);
assert_eq!(count("kr-rrn", "RRN 801301-1234567"), 0);
}
}
@@ -1,128 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Australian identifiers (§2.3): the ATO's Tax File Number and the Medicare
//! card number.
use super::{Detector, Findings, Region, Strength, word_near};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"au-tfn",
"Australian Tax File Number",
Region::Australia,
Strength::Checked,
tfn,
),
Detector::new(
"au-medicare",
"Australian Medicare number",
Region::Australia,
Strength::Checked,
medicare,
),
];
fn re(pattern: &str) -> Regex {
Regex::new(pattern).expect("detector pattern")
}
/// `NNN NNN NNN` stands alone; bare digits (eight or nine) need a word.
static TFN: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{3})( ?)(\d{3})( ?)(\d{2,3})\b"));
/// Weighted sum mod 11, with the ATO's weights for 9- and 8-digit numbers.
pub fn tfn_valid(n: &str) -> bool {
let weights: &[u32] = match n.len() {
9 => &[1, 4, 3, 7, 5, 8, 6, 9, 10],
8 => &[10, 7, 8, 4, 6, 3, 5, 1],
_ => return false,
};
n.bytes()
.zip(weights)
.map(|(b, w)| u32::from(b - b'0') * w)
.sum::<u32>()
% 11
== 0
}
const TFN_WORDS: &[&str] = &["tfn", "tax file number", "tax file no"];
fn tfn(text: &str, findings: &mut Findings) {
for c in TFN.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = n.len() == 9 && c[2] == *" " && c[4] == *" ";
if tfn_valid(&n) && (written || word_near(text, whole.start(), whole.end(), TFN_WORDS)) {
findings.insert(n);
}
}
}
/// `NNNN NNNNN N` (and an optional issue number) stands alone; bare digits
/// need a word.
static MEDICARE: LazyLock<Regex> =
LazyLock::new(|| re(r"\b([2-6]\d{3})( ?)(\d{5})( ?)(\d)(?:[ -]?\d)?\b"));
/// The ninth digit is the weighted sum (1, 3, 7, 9, 1, 3, 7, 9) of the first
/// eight, mod 10.
pub fn medicare_valid(n: &str) -> bool {
let d: Vec<u32> = n.bytes().map(|b| u32::from(b - b'0')).collect();
d.len() >= 9
&& d[..8]
.iter()
.zip([1, 3, 7, 9, 1, 3, 7, 9])
.map(|(a, w)| a * w)
.sum::<u32>()
% 10
== d[8]
}
const MEDICARE_WORDS: &[&str] = &[
"medicare",
"medicare card",
"medicare no",
"medicare number",
];
fn medicare(text: &str, findings: &mut Findings) {
for c in MEDICARE.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = c[2] == *" " && c[4] == *" ";
if medicare_valid(&n)
&& (written || word_near(text, whole.start(), whole.end(), MEDICARE_WORDS))
{
findings.insert(n);
}
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn tax_file_numbers() {
assert_eq!(count("au-tfn", "TFN 123 456 782"), 1);
assert_eq!(count("au-tfn", "123 456 789"), 0);
assert_eq!(count("au-tfn", "order 123456782"), 0);
assert_eq!(count("au-tfn", "tax file number 123456782"), 1);
}
#[test]
fn medicare_numbers() {
assert_eq!(count("au-medicare", "2123 45670 1"), 1);
assert_eq!(count("au-medicare", "2123 45671 1"), 0);
assert_eq!(count("au-medicare", "ref 2123456701"), 0);
assert_eq!(count("au-medicare", "Medicare 2123456701"), 1);
}
}
@@ -1,66 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Canadian identifiers (§2.3): the Social Insurance Number.
use super::{Detector, Findings, Region, Strength, checks, word_near};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[Detector::new(
"ca-sin",
"Canadian Social Insurance Number",
Region::Canada,
Strength::Checked,
sin,
)];
/// `NNN NNN NNN` or `NNN-NNN-NNN` stands alone; nine bare digits need a word.
static SIN: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(r"\b(\d{3})([ -]?)(\d{3})([ -]?)(\d{3})\b").expect("detector pattern")
});
const SIN_WORDS: &[&str] = &[
"sin",
"social insurance",
"nas",
"numéro d'assurance sociale",
"assurance sociale",
];
fn sin(text: &str, findings: &mut Findings) {
for c in SIN.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = !c[2].is_empty() && c[2] == c[4];
// 0 and 8 are never issued as a first digit
if !n.starts_with(['0', '8'])
&& checks::luhn(&n)
&& (written || word_near(text, whole.start(), whole.end(), SIN_WORDS))
{
findings.insert(n);
}
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(text: &str) -> usize {
by_id("ca-sin").unwrap().count(text)
}
#[test]
fn social_insurance_numbers() {
assert_eq!(count("130 692 544 and 193-456-787"), 2);
assert_eq!(count("130 692 545"), 0);
// The government's printed example starts with 0, never issued
assert_eq!(count("046 454 286"), 0);
assert_eq!(count("order 130692544"), 0);
assert_eq!(count("SIN: 130692544"), 1);
}
}
@@ -1,227 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Check-digit algorithms, each from its public definition.
/// The Luhn check (ISO/IEC 7812-1, Annex B) over a string of ASCII digits.
pub fn luhn(digits: &str) -> bool {
if digits.len() < 2 || !digits.bytes().all(|b| b.is_ascii_digit()) {
return false;
}
let sum: u32 = digits
.bytes()
.rev()
.enumerate()
.map(|(i, b)| {
let d = u32::from(b - b'0');
if i % 2 == 1 {
let d = d * 2;
if d > 9 { d - 9 } else { d }
} else {
d
}
})
.sum();
sum.is_multiple_of(10)
}
/// ISO 13616 IBAN lengths, by country, from the IBAN registry.
const IBAN_LENGTHS: &[(&str, usize)] = &[
("AD", 24),
("AE", 23),
("AL", 28),
("AT", 20),
("AZ", 28),
("BA", 20),
("BE", 16),
("BG", 22),
("BH", 22),
("BI", 27),
("BR", 29),
("BY", 28),
("CH", 21),
("CR", 22),
("CY", 28),
("CZ", 24),
("DE", 22),
("DJ", 27),
("DK", 18),
("DO", 28),
("EE", 20),
("EG", 29),
("ES", 24),
("FI", 18),
("FK", 18),
("FO", 18),
("FR", 27),
("GB", 22),
("GE", 22),
("GI", 23),
("GL", 18),
("GR", 27),
("GT", 28),
("HN", 28),
("HR", 21),
("HU", 28),
("IE", 22),
("IL", 23),
("IQ", 23),
("IS", 26),
("IT", 27),
("JO", 30),
("KW", 30),
("KZ", 20),
("LB", 28),
("LC", 32),
("LI", 21),
("LT", 20),
("LU", 20),
("LV", 21),
("LY", 25),
("MC", 27),
("MD", 24),
("ME", 22),
("MK", 19),
("MN", 20),
("MR", 27),
("MT", 31),
("MU", 30),
("NI", 28),
("NL", 18),
("NO", 15),
("OM", 23),
("PK", 24),
("PL", 28),
("PS", 29),
("PT", 25),
("QA", 29),
("RO", 24),
("RS", 22),
("RU", 33),
("SA", 24),
("SC", 31),
("SD", 18),
("SE", 24),
("SI", 19),
("SK", 24),
("SM", 27),
("SO", 23),
("ST", 25),
("SV", 28),
("TL", 23),
("TN", 24),
("TR", 26),
("UA", 29),
("VA", 22),
("VG", 24),
("XK", 20),
("YE", 30),
];
/// The IBAN length for a country code, if the country uses IBANs.
pub fn iban_length(country: &str) -> Option<usize> {
IBAN_LENGTHS
.iter()
.find(|(code, _)| *code == country)
.map(|(_, len)| *len)
}
/// ISO 13616 / ISO 7064 MOD 97-10 over an IBAN with no spaces, upper case:
/// move the first four characters to the end, turn letters into 10–35, and
/// the number mod 97 must be 1. Also checks the country's length.
pub fn iban(iban: &str) -> bool {
if iban.len() < 5
|| !iban
.bytes()
.all(|b| b.is_ascii_uppercase() || b.is_ascii_digit())
{
return false;
}
if iban_length(&iban[..2]) != Some(iban.len())
|| !iban[2..4].bytes().all(|b| b.is_ascii_digit())
{
return false;
}
let mut remainder: u32 = 0;
for b in iban[4..].bytes().chain(iban[..4].bytes()) {
let value = if b.is_ascii_digit() {
u32::from(b - b'0')
} else {
u32::from(b - b'A') + 10
};
remainder = if value >= 10 {
(remainder * 100 + value) % 97
} else {
(remainder * 10 + value) % 97
};
}
remainder == 1
}
/// ISO 3166-1 alpha-2 country codes, for SWIFT/BIC positions 5–6.
const COUNTRIES: &str = "AD AE AF AG AI AL AM AO AQ AR AS AT AU AW AX AZ BA BB BD BE BF BG BH BI BJ \
BL BM BN BO BQ BR BS BT BV BW BY BZ CA CC CD CF CG CH CI CK CL CM CN CO CR CU CV CW CX CY CZ DE DJ \
DK DM DO DZ EC EE EG EH ER ES ET FI FJ FK FM FO FR GA GB GD GE GF GG GH GI GL GM GN GP GQ GR GS GT \
GU GW GY HK HM HN HR HT HU ID IE IL IM IN IO IQ IR IS IT JE JM JO JP KE KG KH KI KM KN KP KR KW KY \
KZ LA LB LC LI LK LR LS LT LU LV LY MA MC MD ME MF MG MH MK ML MM MN MO MP MQ MR MS MT MU MV MW MX \
MY MZ NA NC NE NF NG NI NL NO NP NR NU NZ OM PA PE PF PG PH PK PL PM PN PR PS PT PW PY QA RE RO RS \
RU RW SA SB SC SD SE SG SH SI SJ SK SL SM SN SO SR SS ST SV SX SY SZ TC TD TF TG TH TJ TK TL TM TN \
TO TR TT TV TW TZ UA UG UM US UY UZ VA VC VE VG VI VN VU WF WS XK YE YT ZA ZM ZW";
pub fn is_country(code: &str) -> bool {
code.len() == 2 && COUNTRIES.split(' ').any(|c| c == code)
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn luhn_known_numbers() {
// Published test card numbers
for good in [
"4242424242424242",
"5555555555554444",
"378282246310005",
"79927398713",
] {
assert!(luhn(good), "{good}");
}
for bad in ["4242424242424241", "79927398710", "1", "12a4"] {
assert!(!luhn(bad), "{bad}");
}
}
#[test]
fn iban_registry_examples() {
// The IBAN registry's own examples
for good in [
"GB29NWBK60161331926819",
"DE89370400440532013000",
"FR1420041010050500013M02606",
"NL91ABNA0417164300",
"BE68539007547034",
"NO9386011117947",
"CH9300762011623852957",
] {
assert!(iban(good), "{good}");
}
for bad in [
"GB29NWBK60161331926818", // check fails
"GB29NWBK6016133192681", // too short for GB
"ZZ29NWBK60161331926819", // no such country
"DE8937040044053201300A", // letters where DE has none still fail mod 97
] {
assert!(!iban(bad), "{bad}");
}
}
#[test]
fn countries() {
assert!(is_country("DE") && is_country("US") && is_country("XK"));
assert!(!is_country("ZZ") && !is_country("D"));
}
}
@@ -1,646 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! European Union national identifiers (§2.3), each from its issuer's
//! published rules. An identifier that is only digits and whose check a
//! random number passes often (mod 10, mod 11) counts alone only in its
//! written form, and as bare digits only beside a word.
use super::{
Detector, Findings, Region, Strength, checks, digit_values, stands_alone, valid_short_date,
word_near,
};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"de-tax-id",
"Germany: tax ID (Steuer-ID)",
Region::Eu,
Strength::Checked,
de_tax_id,
),
Detector::new(
"de-id-card",
"Germany: ID card number",
Region::Eu,
Strength::Checked,
de_id_card,
),
Detector::new(
"fr-nir",
"France: social security number (NIR)",
Region::Eu,
Strength::Checked,
fr_nir,
),
Detector::new(
"es-dni-nie",
"Spain: DNI and NIE",
Region::Eu,
Strength::Checked,
es_dni_nie,
),
Detector::new(
"it-codice-fiscale",
"Italy: codice fiscale",
Region::Eu,
Strength::Checked,
it_codice_fiscale,
),
Detector::new(
"nl-bsn",
"Netherlands: BSN",
Region::Eu,
Strength::Checked,
nl_bsn,
),
Detector::new(
"be-national-number",
"Belgium: national number",
Region::Eu,
Strength::Checked,
be_national_number,
),
Detector::new(
"pl-pesel",
"Poland: PESEL",
Region::Eu,
Strength::Checked,
pl_pesel,
),
Detector::new(
"se-personnummer",
"Sweden: personnummer",
Region::Eu,
Strength::Checked,
se_personnummer,
),
Detector::new(
"dk-cpr",
"Denmark: CPR number",
Region::Eu,
Strength::NeedsWord,
dk_cpr,
),
Detector::new(
"fi-hetu",
"Finland: personal identity code",
Region::Eu,
Strength::Checked,
fi_hetu,
),
Detector::new(
"ie-pps",
"Ireland: PPS number",
Region::Eu,
Strength::Checked,
ie_pps,
),
Detector::new(
"pt-nif",
"Portugal: NIF",
Region::Eu,
Strength::Checked,
pt_nif,
),
Detector::new(
"at-svnr",
"Austria: social insurance number",
Region::Eu,
Strength::Checked,
at_svnr,
),
];
fn re(pattern: &str) -> Regex {
Regex::new(pattern).expect("detector pattern")
}
fn num(s: &str) -> u32 {
s.parse().unwrap_or(0)
}
// --- Germany --------------------------------------------------------------
/// Eleven digits, written `86 095 742 719` on the BZSt's letters.
static DE_TAX: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{2}( ?)\d{3}( ?)\d{3}( ?)\d{3}\b"));
/// ISO 7064 MOD 11,10; no leading zero; in the first ten digits one digit
/// appears two or three times and every other at most once.
pub fn de_tax_id_valid(n: &str) -> bool {
let d = digit_values(n);
if d.len() != 11 || d[0] == 0 {
return false;
}
let mut counts = [0u8; 10];
for &x in &d[..10] {
counts[x as usize] += 1;
}
let repeated = counts.iter().filter(|&&c| c >= 2).count();
if repeated != 1 || counts.iter().any(|&c| c > 3) {
return false;
}
let mut product = 10;
for &x in &d[..10] {
let mut sum = (x + product) % 10;
if sum == 0 {
sum = 10;
}
product = (2 * sum) % 11;
}
let check = match 11 - product {
10 => 0,
c => c,
};
check == d[10]
}
const DE_TAX_WORDS: &[&str] = &[
"steuer-id",
"steueridentifikationsnummer",
"steuerliche identifikationsnummer",
"idnr",
"identifikationsnummer",
"tax id",
];
fn de_tax_id(text: &str, findings: &mut Findings) {
for c in DE_TAX.captures_iter(text) {
let whole = c.get(0).unwrap();
let written = [&c[1], &c[2], &c[3]].iter().all(|s| *s == " ");
let n: String = whole.as_str().replace(' ', "");
if de_tax_id_valid(&n)
&& (written || word_near(text, whole.start(), whole.end(), DE_TAX_WORDS))
{
findings.insert(n);
}
}
}
/// The ID card's document number: a letter from the card's alphabet, eight
/// more characters from it, then the check digit.
static DE_ID: LazyLock<Regex> =
LazyLock::new(|| re(r"\b[CFGHJKLMNPRTVWXYZ][CFGHJKLMNPRTVWXYZ0-9]{8}\d\b"));
/// ICAO 9303 check digit: weights 7, 3, 1; letters A=10 … Z=35.
pub fn icao_check(chars: &str, check: u32) -> bool {
let value = |c: char| c.to_digit(10).unwrap_or_else(|| c as u32 - 'A' as u32 + 10);
let sum: u32 = chars
.chars()
.zip([7, 3, 1].iter().cycle())
.map(|(c, w)| value(c) * w)
.sum();
sum % 10 == check
}
fn de_id_card(text: &str, findings: &mut Findings) {
for m in DE_ID.find_iter(text) {
let s = m.as_str();
if icao_check(&s[..9], num(&s[9..])) {
findings.insert(s);
}
}
}
// --- France ---------------------------------------------------------------
/// Sex, year, month, department (with Corsica's 2A and 2B), commune, order,
/// then the two-digit key, spaces allowed between groups.
static FR_NIR: LazyLock<Regex> = LazyLock::new(|| {
re(r"\b([1-478]) ?(\d{2}) ?(\d{2}) ?(\d{2}|2[AB]) ?(\d{3}) ?(\d{3}) ?(\d{2})\b")
});
fn fr_nir(text: &str, findings: &mut Findings) {
for c in FR_NIR.captures_iter(text) {
let month = num(&c[3]);
if !(matches!(month, 1..=12 | 20..=42 | 50..=99)) {
continue;
}
let department = match &c[4] {
"2A" => "19",
"2B" => "18",
d => d,
};
let body = format!(
"{}{}{}{}{}{}",
&c[1], &c[2], &c[3], department, &c[5], &c[6]
);
let Ok(value) = body.parse::<u64>() else {
continue;
};
if 97 - value % 97 == u64::from(num(&c[7])) {
findings.insert(format!(
"{}{}{}{}{}{}{}",
&c[1], &c[2], &c[3], &c[4], &c[5], &c[6], &c[7]
));
}
}
}
// --- Spain ----------------------------------------------------------------
static ES_ID: LazyLock<Regex> = LazyLock::new(|| re(r"(?i)\b([XYZ]?)[ -]?(\d{7,8})[ -]?([A-Z])\b"));
const DNI_LETTERS: &[u8] = b"TRWAGMYFPDXBNJZSQVHLCKE";
fn es_dni_nie(text: &str, findings: &mut Findings) {
for c in ES_ID.captures_iter(text) {
let prefix = c[1].to_ascii_uppercase();
let digits = &c[2];
// DNI: eight digits; NIE: X, Y or Z and seven digits
let number = match (prefix.as_str(), digits.len()) {
("", 8) => digits.to_string(),
("X", 7) => format!("0{digits}"),
("Y", 7) => format!("1{digits}"),
("Z", 7) => format!("2{digits}"),
_ => continue,
};
let letter = c[3].to_ascii_uppercase();
if DNI_LETTERS[(num(&number) % 23) as usize] == letter.as_bytes()[0] {
findings.insert(format!("{prefix}{digits}{letter}"));
}
}
}
// --- Italy ----------------------------------------------------------------
/// Surname and name letters, year, month letter, day, place code, check
/// letter; digits may be replaced by letters (omocodia).
static IT_CF: LazyLock<Regex> = LazyLock::new(|| {
let d = "[0-9LMNPQRSTUV]";
re(&format!(
r"(?i)\b[A-Z]{{6}}{d}{{2}}[ABCDEHLMPRST]{d}{{2}}[A-Z]{d}{{3}}[A-Z]\b"
))
});
/// The Ministry's odd-position values for 0–9 and A–Z.
const CF_ODD: [u32; 36] = [
1, 0, 5, 7, 9, 13, 15, 17, 19, 21, // 0-9
1, 0, 5, 7, 9, 13, 15, 17, 19, 21, 2, 4, 18, 20, 11, 3, 6, 8, 12, 14, 16, 10, 22, 25, 24,
23, // A-Z
];
pub fn codice_fiscale_valid(cf: &str) -> bool {
let index = |c: u8| {
if c.is_ascii_digit() {
(c - b'0') as usize
} else {
(c - b'A') as usize + 10
}
};
let even = |c: u8| {
if c.is_ascii_digit() {
u32::from(c - b'0')
} else {
u32::from(c - b'A')
}
};
let bytes = cf.as_bytes();
let sum: u32 = bytes[..15]
.iter()
.enumerate()
.map(|(i, &c)| {
if i % 2 == 0 {
CF_ODD[index(c)]
} else {
even(c)
}
})
.sum();
u32::from(bytes[15] - b'A') == sum % 26
}
fn it_codice_fiscale(text: &str, findings: &mut Findings) {
for m in IT_CF.find_iter(text) {
let cf = m.as_str().to_ascii_uppercase();
if codice_fiscale_valid(&cf) {
findings.insert(cf);
}
}
}
// --- Netherlands ----------------------------------------------------------
/// Nine digits, sometimes written `1112.22.333`.
static NL_BSN: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{4})(\.?)(\d{2})(\.?)(\d{3})\b"));
/// The eleven test: weights 9 down to 2, and −1 for the last digit.
pub fn bsn_valid(n: &str) -> bool {
let d = digit_values(n);
let sum: i64 = d[..8]
.iter()
.zip((2..=9).rev())
.map(|(a, w)| i64::from(a * w))
.sum::<i64>()
- i64::from(d[8]);
sum != 0 && sum % 11 == 0
}
const BSN_WORDS: &[&str] = &[
"bsn",
"burgerservicenummer",
"sofinummer",
"sofi-nummer",
"citizen service number",
];
fn nl_bsn(text: &str, findings: &mut Findings) {
for c in NL_BSN.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = &c[2] == "." && &c[4] == ".";
if bsn_valid(&n) && (written || word_near(text, whole.start(), whole.end(), BSN_WORDS)) {
findings.insert(n);
}
}
}
// --- Belgium --------------------------------------------------------------
/// `YY.MM.DD-XXX.CC` or eleven digits.
static BE_NN: LazyLock<Regex> =
LazyLock::new(|| re(r"\b(\d{2})\.?(\d{2})\.?(\d{2})-?(\d{3})\.?(\d{2})\b"));
fn be_national_number(text: &str, findings: &mut Findings) {
for c in BE_NN.captures_iter(text) {
let (month, day) = (num(&c[2]), num(&c[3]));
// Month 0 and day 0 mean unknown; bis numbers add 20 or 40 to the month
if !(month <= 12 || (20..=32).contains(&month) || (40..=52).contains(&month)) || day > 31 {
continue;
}
let body = format!("{}{}{}{}", &c[1], &c[2], &c[3], &c[4]);
let check = u64::from(num(&c[5]));
let before_2000 = 97 - body.parse::<u64>().unwrap_or(0) % 97;
let since_2000 = 97 - format!("2{body}").parse::<u64>().unwrap_or(0) % 97;
if check == before_2000 || check == since_2000 {
findings.insert(format!("{body}{}", &c[5]));
}
}
}
// --- Poland ---------------------------------------------------------------
static ELEVEN: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{11}\b"));
/// Weights 1, 3, 7, 9 repeating; the birth date encodes the century in the
/// month (+80 for the 1800s, +20 for the 2000s, and so on).
pub fn pesel_valid(n: &str) -> bool {
let d = digit_values(n);
let sum: u32 = d[..10]
.iter()
.zip([1, 3, 7, 9].iter().cycle())
.map(|(a, w)| a * w)
.sum();
let month = d[2] * 10 + d[3];
let (century, month) = match month {
81..=92 => (1800, month - 80),
1..=12 => (1900, month),
21..=32 => (2000, month - 20),
41..=52 => (2100, month - 40),
_ => return false,
};
let year = century + d[0] * 10 + d[1];
(10 - sum % 10) % 10 == d[10] && (1..=super::days_in(year, month)).contains(&(d[4] * 10 + d[5]))
}
const PESEL_WORDS: &[&str] = &["pesel", "numer pesel", "nr pesel"];
fn pl_pesel(text: &str, findings: &mut Findings) {
for m in ELEVEN.find_iter(text) {
if pesel_valid(m.as_str()) && word_near(text, m.start(), m.end(), PESEL_WORDS) {
findings.insert(m.as_str());
}
}
}
// --- Sweden ---------------------------------------------------------------
/// `YYMMDD-NNNN`, `YYYYMMDD-NNNN` (`+` after 100), or the bare digits.
static SE_PNR: LazyLock<Regex> =
LazyLock::new(|| re(r"\b(?:\d{2})?(\d{2})(\d{2})(\d{2})([-+]?)(\d{4})\b"));
const SE_WORDS: &[&str] = &[
"personnummer",
"personnr",
"person nr",
"samordningsnummer",
"pnr",
];
fn se_personnummer(text: &str, findings: &mut Findings) {
for c in SE_PNR.captures_iter(text) {
let whole = c.get(0).unwrap();
let (yy, month, day) = (num(&c[1]), num(&c[2]), num(&c[3]));
// Coordination numbers add 60 to the day
let day = if day > 60 { day - 60 } else { day };
let ten = format!("{}{}{}{}", &c[1], &c[2], &c[3], &c[5]);
let written = !c[4].is_empty();
if valid_short_date(yy, month, day)
&& checks::luhn(&ten)
&& (written || word_near(text, whole.start(), whole.end(), SE_WORDS))
{
findings.insert(ten);
}
}
}
// --- Denmark --------------------------------------------------------------
static DK_CPR: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{2})(\d{2})(\d{2})-?(\d{4})\b"));
const CPR_WORDS: &[&str] = &["cpr", "cpr-nr", "cpr nr", "cpr-nummer", "personnummer"];
fn dk_cpr(text: &str, findings: &mut Findings) {
for c in DK_CPR.captures_iter(text) {
let whole = c.get(0).unwrap();
if valid_short_date(num(&c[3]), num(&c[2]), num(&c[1]))
&& word_near(text, whole.start(), whole.end(), CPR_WORDS)
{
findings.insert(whole.as_str().replace('-', ""));
}
}
}
// --- Finland --------------------------------------------------------------
static FI_HETU: LazyLock<Regex> =
LazyLock::new(|| re(r"(?i)\b(\d{2})(\d{2})(\d{2})[-+ABCDEFYXWVU](\d{3})([0-9A-Y])\b"));
const HETU_CHECK: &[u8] = b"0123456789ABCDEFHJKLMNPRSTUVWXY";
fn fi_hetu(text: &str, findings: &mut Findings) {
for c in FI_HETU.captures_iter(text) {
let (day, month, yy) = (num(&c[1]), num(&c[2]), num(&c[3]));
let n: u64 = format!("{}{}{}{}", &c[1], &c[2], &c[3], &c[4])
.parse()
.unwrap_or(0);
let check = c[5].to_ascii_uppercase().as_bytes()[0];
if valid_short_date(yy, month, day) && HETU_CHECK[(n % 31) as usize] == check {
findings.insert(c[0].to_ascii_uppercase());
}
}
}
// --- Ireland --------------------------------------------------------------
static IE_PPS: LazyLock<Regex> = LazyLock::new(|| re(r"(?i)\b(\d{7})([A-W])([ABHW]?)\b"));
const PPS_CHECK: &[u8] = b"WABCDEFGHIJKLMNOPQRSTUV";
fn ie_pps(text: &str, findings: &mut Findings) {
for c in IE_PPS.captures_iter(text) {
let mut sum: u32 = digit_values(&c[1])
.iter()
.zip((2..=8).rev())
.map(|(a, w)| a * w)
.sum();
// The second letter counts, times 9; W (the old form) counts as 0
let second = c[3].to_ascii_uppercase();
if let Some(&letter) = second.as_bytes().first()
&& letter != b'W'
{
sum += u32::from(letter - b'A' + 1) * 9;
}
let check = c[2].to_ascii_uppercase().as_bytes()[0];
if PPS_CHECK[(sum % 23) as usize] == check {
findings.insert(c[0].to_ascii_uppercase());
}
}
}
// --- Portugal -------------------------------------------------------------
static NINE: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{9}\b"));
/// Mod 11 over weights 9 down to 2; a check of 10 or 11 becomes 0.
pub fn nif_valid(n: &str) -> bool {
let d = digit_values(n);
let sum: u32 = d[..8].iter().zip((2..=9).rev()).map(|(a, w)| a * w).sum();
let check = match 11 - sum % 11 {
10 | 11 => 0,
c => c,
};
matches!(d[0], 1 | 2 | 3 | 5 | 6 | 8 | 9) && check == d[8]
}
const NIF_WORDS: &[&str] = &[
"nif",
"contribuinte",
"número de identificação fiscal",
"numero de contribuinte",
];
fn pt_nif(text: &str, findings: &mut Findings) {
for m in NINE.find_iter(text) {
if nif_valid(m.as_str()) && word_near(text, m.start(), m.end(), NIF_WORDS) {
findings.insert(m.as_str());
}
}
}
// --- Austria --------------------------------------------------------------
/// A serial and check digit, then the birth date: `1237 010180`.
static AT_SVNR: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{3})(\d)( ?)(\d{2})(\d{2})(\d{2})\b"));
const SVNR_WORDS: &[&str] = &[
"sozialversicherungsnummer",
"svnr",
"sv-nr",
"sv-nummer",
"versicherungsnummer",
];
fn at_svnr(text: &str, findings: &mut Findings) {
for c in AT_SVNR.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}{}{}", &c[1], &c[2], &c[4], &c[5], &c[6]);
let d = digit_values(&n);
let sum: u32 = d
.iter()
.zip([3, 7, 9, 0, 5, 8, 4, 2, 1, 6])
.map(|(a, w)| a * w)
.sum();
let written = &c[3] == " ";
if d[0] != 0
&& sum % 11 == d[3]
&& valid_short_date(num(&c[6]), num(&c[5]), num(&c[4]))
&& (written || word_near(text, whole.start(), whole.end(), SVNR_WORDS))
&& stands_alone(text, whole.start(), whole.end())
{
findings.insert(n);
}
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn germany() {
assert_eq!(count("de-tax-id", "86 095 742 719"), 1);
assert_eq!(count("de-tax-id", "Steuer-ID: 86095742719"), 1);
assert_eq!(count("de-tax-id", "Rechnung 86095742719"), 0);
assert_eq!(count("de-tax-id", "86 095 742 718"), 0);
// ICAO 9303's German specimen card
assert_eq!(count("de-id-card", "Ausweis T220001293"), 1);
assert_eq!(count("de-id-card", "T220001294"), 0);
}
#[test]
fn france_spain_italy() {
assert_eq!(count("fr-nir", "2 55 08 14 168 025 38"), 1);
assert_eq!(count("fr-nir", "255081416802539"), 0);
assert_eq!(count("es-dni-nie", "DNI 12345678Z, NIE X-1234567-L"), 2);
assert_eq!(count("es-dni-nie", "12345678A"), 0);
assert_eq!(count("it-codice-fiscale", "CF: RSSMRA85T10A562S"), 1);
assert_eq!(count("it-codice-fiscale", "RSSMRA85T10A562T"), 0);
}
#[test]
fn benelux() {
assert_eq!(count("nl-bsn", "1112.22.333"), 1);
assert_eq!(count("nl-bsn", "BSN 111222333"), 1);
assert_eq!(count("nl-bsn", "order 111222333"), 0);
assert_eq!(count("nl-bsn", "BSN 111222334"), 0);
assert_eq!(count("be-national-number", "85.07.30-033.28"), 1);
assert_eq!(count("be-national-number", "85073003329"), 0);
}
#[test]
fn nordics() {
assert_eq!(count("se-personnummer", "811218-9876"), 1);
assert_eq!(count("se-personnummer", "811218-9875"), 0);
assert_eq!(count("se-personnummer", "order 8112189876"), 0);
assert_eq!(count("se-personnummer", "personnummer 198112189876"), 1);
assert_eq!(count("dk-cpr", "CPR-nr: 010170-1234"), 1);
assert_eq!(count("dk-cpr", "010170-1234"), 0);
assert_eq!(count("dk-cpr", "CPR 320170-1234"), 0);
assert_eq!(count("fi-hetu", "131052-308T"), 1);
assert_eq!(count("fi-hetu", "131052-308U"), 0);
}
#[test]
fn poland_ireland_portugal_austria() {
assert_eq!(count("pl-pesel", "PESEL 44051401359, pesel 02070803628"), 2);
assert_eq!(count("pl-pesel", "PESEL 44051401358"), 0);
assert_eq!(count("pl-pesel", "44051401359"), 0);
assert_eq!(count("ie-pps", "PPS 1234567T and 1234567FA"), 2);
assert_eq!(count("ie-pps", "1234567U"), 0);
assert_eq!(count("pt-nif", "NIF 123456789"), 1);
assert_eq!(count("pt-nif", "NIF 123456788"), 0);
assert_eq!(count("at-svnr", "1237 010180"), 1);
assert_eq!(count("at-svnr", "SVNR 1237010180"), 1);
assert_eq!(count("at-svnr", "1238 010180"), 0);
}
}
@@ -1,116 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! European identifiers outside the EU (§2.3): Norway's national identity
//! number and Switzerland's AHV number.
use super::{Detector, Findings, Region, Strength, digit_values, valid_short_date};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"no-fnr",
"Norway: national identity number",
Region::Europe,
Strength::Checked,
no_fnr,
),
Detector::new(
"ch-ahv",
"Switzerland: AHV number",
Region::Europe,
Strength::Checked,
ch_ahv,
),
];
static ELEVEN: LazyLock<Regex> =
LazyLock::new(|| Regex::new(r"\b\d{6} ?\d{5}\b").expect("detector pattern"));
/// Two mod 11 check digits over a birth date (D-numbers add 40 to the day,
/// H-numbers 40 to the month): strong enough to count alone.
pub fn fnr_valid(n: &str) -> bool {
let d = digit_values(n);
if d.len() != 11 {
return false;
}
let check =
|weights: &[u32]| match 11 - d.iter().zip(weights).map(|(a, w)| a * w).sum::<u32>() % 11 {
11 => Some(0),
10 => None,
c => Some(c),
};
let day = d[0] * 10 + d[1];
let month = d[2] * 10 + d[3];
let day = if day > 40 { day - 40 } else { day };
let month = if month > 40 { month - 40 } else { month };
valid_short_date(d[4] * 10 + d[5], month, day)
&& check(&[3, 7, 6, 1, 8, 9, 4, 5, 2]) == Some(d[9])
&& check(&[5, 4, 3, 2, 7, 6, 5, 4, 3, 2]) == Some(d[10])
}
fn no_fnr(text: &str, findings: &mut Findings) {
for m in ELEVEN.find_iter(text) {
let n = m.as_str().replace(' ', "");
if fnr_valid(&n) {
findings.insert(n);
}
}
}
/// `756.1234.5678.97`: the country prefix, then an EAN-13 check digit.
static AHV: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(r"\b756[. ]?\d{4}[. ]?\d{4}[. ]?\d{2}\b").expect("detector pattern")
});
pub fn ean13_valid(n: &str) -> bool {
let d = digit_values(n);
if d.len() != 13 {
return false;
}
let sum: u32 = d[..12]
.iter()
.enumerate()
.map(|(i, x)| if i % 2 == 0 { *x } else { x * 3 })
.sum();
(10 - sum % 10) % 10 == d[12]
}
fn ch_ahv(text: &str, findings: &mut Findings) {
for m in AHV.find_iter(text) {
let n: String = m.as_str().chars().filter(char::is_ascii_digit).collect();
if ean13_valid(&n) {
findings.insert(n);
}
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn norway() {
assert_eq!(count("no-fnr", "01019000083"), 1);
assert_eq!(count("no-fnr", "010190 00083"), 1);
assert_eq!(count("no-fnr", "01019000084"), 0);
// Not a date
assert_eq!(count("no-fnr", "32019000083"), 0);
}
#[test]
fn switzerland() {
// The federal example
assert_eq!(count("ch-ahv", "AHV 756.9217.0769.85"), 1);
assert_eq!(count("ch-ahv", "7569217076985"), 1);
assert_eq!(count("ch-ahv", "756.9217.0769.86"), 0);
}
}
@@ -1,278 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Detectors (dlp-and-mail-flow-rules spec, §2.3): each finds one kind of
//! identifier in text and reports the distinct ones it found.
//!
//! A detector is one of two strengths:
//!
//! - **Checked**: the identifier carries a published check digit or
//! checksum, so a random number rarely passes; found on its own.
//! - **Needs a word**: the format alone is too common, so a candidate counts
//! only with a corroborating word within [`WINDOW`] characters either
//! side.
//!
//! Findings are distinct normalized values (digits only, upper case), so the
//! same card number pasted twice counts once. They stay in memory: callers
//! read only [`Findings::len`].
pub mod africa;
pub mod americas;
pub mod any;
pub mod asia;
pub mod australia;
pub mod canada;
pub mod checks;
pub mod eu;
pub mod europe;
pub mod templates;
pub mod uk;
pub mod us;
use ahash::AHashSet;
/// How far, in characters, a corroborating word may be from a candidate.
pub const WINDOW: usize = 50;
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Strength {
Checked,
NeedsWord,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Region {
Any,
Us,
Uk,
Canada,
Australia,
Eu,
Europe,
Asia,
Americas,
Africa,
}
/// The distinct values one detector found.
#[derive(Debug, Default)]
pub struct Findings(AHashSet<String>);
impl Findings {
pub fn insert(&mut self, value: impl Into<String>) {
self.0.insert(value.into());
}
pub fn len(&self) -> usize {
self.0.len()
}
pub fn is_empty(&self) -> bool {
self.0.is_empty()
}
}
pub struct Detector {
/// Stable id, stored in rules: `payment-card`, `iban`, `us-ssn`.
pub id: &'static str,
pub name: &'static str,
pub region: Region,
pub strength: Strength,
find: fn(&str, &mut Findings),
}
impl Detector {
pub const fn new(
id: &'static str,
name: &'static str,
region: Region,
strength: Strength,
find: fn(&str, &mut Findings),
) -> Self {
Self {
id,
name,
region,
strength,
find,
}
}
/// Adds what this detector finds in `text` to `findings`. Call once per
/// piece of text (subject, each part, each attachment) with the same
/// `findings`, then read its length.
pub fn find(&self, text: &str, findings: &mut Findings) {
(self.find)(text, findings)
}
/// The distinct values found in one text.
pub fn count(&self, text: &str) -> usize {
let mut findings = Findings::default();
self.find(text, &mut findings);
findings.len()
}
}
/// Every detector, in the order the console lists them.
pub fn all() -> impl Iterator<Item = &'static Detector> {
[
any::DETECTORS,
us::DETECTORS,
uk::DETECTORS,
canada::DETECTORS,
australia::DETECTORS,
eu::DETECTORS,
europe::DETECTORS,
asia::DETECTORS,
americas::DETECTORS,
africa::DETECTORS,
]
.into_iter()
.flatten()
}
pub fn by_id(id: &str) -> Option<&'static Detector> {
all().find(|detector| detector.id == id)
}
/// Whether one of `words` appears, as a whole word and ignoring case, within
/// [`WINDOW`] characters before `start` or after `end` (byte offsets of the
/// candidate in `text`). The window is widened by the longest word, so a
/// word that reaches into it still counts whole.
pub fn word_near(text: &str, start: usize, end: usize, words: &[&str]) -> bool {
let reach = WINDOW + words.iter().map(|w| w.chars().count()).max().unwrap_or(0);
let before = text[..start]
.char_indices()
.rev()
.nth(reach - 1)
.map_or(0, |(i, _)| i);
let after = text[end..]
.char_indices()
.nth(reach)
.map_or(text.len(), |(i, _)| end + i);
let window = text[before..after].to_lowercase();
words.iter().any(|word| contains_word(&window, word))
}
/// Whether `word` (lower case) appears in `haystack` (lower case) with no
/// letter or digit on either side.
pub fn contains_word(haystack: &str, word: &str) -> bool {
haystack.match_indices(word).any(|(i, _)| {
let before_ok = haystack[..i]
.chars()
.next_back()
.is_none_or(|c| !c.is_alphanumeric());
let after_ok = haystack[i + word.len()..]
.chars()
.next()
.is_none_or(|c| !c.is_alphanumeric());
before_ok && after_ok
})
}
/// Whether the match at `start..end` stands alone: no digit or letter
/// directly before or after it, so `123-45-6789` isn't found inside a
/// longer run of digits.
pub fn stands_alone(text: &str, start: usize, end: usize) -> bool {
let before = text[..start].chars().next_back();
let after = text[end..].chars().next();
before.is_none_or(|c| !c.is_alphanumeric()) && after.is_none_or(|c| !c.is_alphanumeric())
}
/// Days in `month` of `year` (0 for a month that doesn't exist).
pub fn days_in(year: u32, month: u32) -> u32 {
match month {
1 | 3 | 5 | 7 | 8 | 10 | 12 => 31,
4 | 6 | 9 | 11 => 30,
2 if year.is_multiple_of(4) && (!year.is_multiple_of(100) || year.is_multiple_of(400)) => {
29
}
2 => 28,
_ => 0,
}
}
/// Whether `year`-`month`-`day` is a real date between 1900 and 2100.
pub fn valid_date(year: u32, month: u32, day: u32) -> bool {
(1900..=2100).contains(&year) && (1..=days_in(year, month)).contains(&day)
}
/// Whether a two-digit year, month and day make a real date in either the
/// 1900s or the 2000s.
pub fn valid_short_date(yy: u32, month: u32, day: u32) -> bool {
valid_date(1900 + yy, month, day) || valid_date(2000 + yy, month, day)
}
/// The value of each digit in `s`.
pub fn digit_values(s: &str) -> Vec<u32> {
s.bytes()
.filter(u8::is_ascii_digit)
.map(|b| u32::from(b - b'0'))
.collect()
}
/// The ASCII digits of `s`.
pub fn digits(s: &str) -> String {
s.chars().filter(char::is_ascii_digit).collect()
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn words_are_whole_and_near() {
let text = "Your passport number is X1234567, thanks";
let start = text.find("X123").unwrap();
assert!(word_near(text, start, start + 8, &["passport"]));
assert!(!word_near(text, start, start + 8, &["pass"]));
let far = format!("passport{}X1234567", " ".repeat(60));
let start = far.find("X123").unwrap();
assert!(!word_near(&far, start, start + 8, &["passport"]));
}
#[test]
fn near_counts_characters_not_bytes() {
// 45 two-byte characters between the word and the candidate: within
// 50 characters, though over 50 bytes
let text = format!("passport {} X1234567", "é".repeat(45));
let start = text.find("X123").unwrap();
assert!(word_near(&text, start, start + 8, &["passport"]));
}
/// An ordinary business email: order, invoice and tracking numbers,
/// dates, amounts, a street address. Nothing here is an identifier, so
/// no detector may fire, except the contact ones on the signature.
#[test]
fn ordinary_mail_finds_nothing() {
let text = "Hi Dana,\n\nThanks for order 4471-2290 placed 2026-09-14. Invoice INV-2026-00917 \
for $12,480.00 is due 10/31/2026; PO 7731902 covers lines 1-14. Tracking \
1Z999AA10123456784, parcel 3 of 5, 12.5 kg, box 40x30x20 cm. Meeting moved to \
Tuesday 9:30-10:15 in room 2B, building 1177. Ticket #5520318, case 20260914-0042. \
Version 2026.9.28.4, build 118822, commit 5a73a118. Serial SN-88213-X. \
Ship to 1600 Amphitheatre Pkwy, Mountain View, CA 94043. Revenue grew 18% to \
1,204,332 units; see figures 3.1-3.4 and table 12.\n\nBest,\nSam\n\
Sam Rivera | +1 (415) 555-2671 | [email protected]";
let quiet = ["email-addresses", "phone-numbers"];
for detector in all().filter(|d| !quiet.contains(&d.id)) {
assert_eq!(
detector.count(text),
0,
"{} fired on ordinary mail",
detector.id
);
}
}
#[test]
fn ids_are_unique() {
let mut seen = AHashSet::new();
for detector in all() {
assert!(seen.insert(detector.id), "duplicate id {}", detector.id);
assert!(by_id(detector.id).is_some());
}
}
}
@@ -1,97 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Templates (§2.3): named sets of detectors, so a policy doesn't pick forty
//! one at a time. Each is named for what it finds, never for a law, and is a
//! starting point: once added to a rule, its detectors can be changed.
pub struct Template {
pub id: &'static str,
pub name: &'static str,
pub detectors: &'static [&'static str],
}
pub static TEMPLATES: &[Template] = &[
Template {
id: "payment-and-bank",
name: "Payment cards and bank accounts",
detectors: &["payment-card", "iban", "swift-bic", "us-aba-routing"],
},
Template {
id: "us-personal",
name: "US personal identifiers",
detectors: &[
"us-ssn",
"us-itin",
"us-ein",
"us-drivers-license",
"passport",
"date-of-birth",
],
},
Template {
id: "uk-personal",
name: "UK personal identifiers",
detectors: &["uk-nino", "uk-utr", "uk-nhs", "passport", "date-of-birth"],
},
Template {
id: "eu-national",
name: "EU national identifiers",
detectors: &[
"de-tax-id",
"de-id-card",
"fr-nir",
"es-dni-nie",
"it-codice-fiscale",
"nl-bsn",
"be-national-number",
"pl-pesel",
"se-personnummer",
"dk-cpr",
"fi-hetu",
"ie-pps",
"pt-nif",
"at-svnr",
],
},
Template {
id: "health",
name: "Health identifiers",
detectors: &["uk-nhs", "us-mbi", "us-npi", "us-dea", "au-medicare"],
},
Template {
id: "credentials",
name: "Credentials and keys",
detectors: &["private-key", "credentials"],
},
Template {
id: "contact-lists",
name: "Contact lists",
detectors: &["email-addresses", "phone-numbers"],
},
];
pub fn by_id(id: &str) -> Option<&'static Template> {
TEMPLATES.iter().find(|template| template.id == id)
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn every_template_names_real_detectors() {
for template in TEMPLATES {
for id in template.detectors {
assert!(
super::super::by_id(id).is_some(),
"{}: no detector {id}",
template.id
);
}
}
}
}
@@ -1,155 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! United Kingdom identifiers (§2.3): HMRC's National Insurance number and
//! Unique Taxpayer Reference, and the NHS number.
use super::{Detector, Findings, Region, Strength, word_near};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"uk-nino",
"UK National Insurance number",
Region::Uk,
Strength::Checked,
nino,
),
Detector::new(
"uk-nhs",
"UK NHS number",
Region::Uk,
Strength::Checked,
nhs,
),
Detector::new(
"uk-utr",
"UK Unique Taxpayer Reference",
Region::Uk,
Strength::NeedsWord,
utr,
),
];
fn re(pattern: &str) -> Regex {
Regex::new(pattern).expect("detector pattern")
}
/// Two letters, six digits (often in pairs), a suffix A–D.
static NINO: LazyLock<Regex> =
LazyLock::new(|| re(r"(?i)\b([A-Z])([A-Z]) ?(\d{2}) ?(\d{2}) ?(\d{2}) ?([A-D])\b"));
/// HMRC's rules: D, F, I, Q, U and V are never used; O never second; and
/// BG, GB, KN, NK, NT, TN and ZZ are never allocated.
fn nino_prefix(first: char, second: char) -> bool {
const NEVER: &str = "DFIQUV";
let pair: String = [first, second].iter().collect();
!NEVER.contains(first)
&& !NEVER.contains(second)
&& second != 'O'
&& !["BG", "GB", "KN", "NK", "NT", "TN", "ZZ"].contains(&pair.as_str())
}
fn nino(text: &str, findings: &mut Findings) {
for c in NINO.captures_iter(text) {
let first = c[1].to_ascii_uppercase().chars().next().unwrap();
let second = c[2].to_ascii_uppercase().chars().next().unwrap();
if nino_prefix(first, second) {
findings.insert(format!(
"{first}{second}{}{}{}{}",
&c[3],
&c[4],
&c[5],
c[6].to_ascii_uppercase()
));
}
}
}
/// `NNN NNN NNNN` stands alone; ten bare digits need a word.
static NHS: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{3})([ -]?)(\d{3})([ -]?)(\d{4})\b"));
/// Mod 11: weights 10 down to 2 over the first nine digits; the check digit
/// is 11 minus the remainder (11 becomes 0; 10 is never issued).
pub fn nhs_valid(n: &str) -> bool {
let d: Vec<u32> = n.bytes().map(|b| u32::from(b - b'0')).collect();
let sum: u32 = d[..9].iter().zip((2..=10).rev()).map(|(a, w)| a * w).sum();
match 11 - sum % 11 {
11 => d[9] == 0,
10 => false,
check => d[9] == check,
}
}
const NHS_WORDS: &[&str] = &["nhs", "nhs number", "nhs no"];
fn nhs(text: &str, findings: &mut Findings) {
for c in NHS.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = !c[2].is_empty() && c[2] == c[4];
if nhs_valid(&n) && (written || word_near(text, whole.start(), whole.end(), NHS_WORDS)) {
findings.insert(n);
}
}
}
static UTR: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{5} ?\d{5}\b"));
const UTR_WORDS: &[&str] = &[
"utr",
"unique taxpayer reference",
"tax reference",
"self assessment",
];
fn utr(text: &str, findings: &mut Findings) {
for m in UTR.find_iter(text) {
if word_near(text, m.start(), m.end(), UTR_WORDS) {
findings.insert(m.as_str().replace(' ', ""));
}
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn national_insurance() {
assert_eq!(count("uk-nino", "NI: AB 12 34 56 C, ce123456d"), 2);
// Letters never used, pairs never allocated, a suffix past D
for bad in [
"QQ123456C",
"AO123456C",
"GB123456A",
"AB123456E",
"DA123456A",
] {
assert_eq!(count("uk-nino", bad), 0, "{bad}");
}
}
#[test]
fn nhs_numbers() {
// The NHS's own example
assert_eq!(count("uk-nhs", "943 476 5919"), 1);
assert_eq!(count("uk-nhs", "943 476 5918"), 0);
assert_eq!(count("uk-nhs", "order 9434765919"), 0);
assert_eq!(count("uk-nhs", "NHS number 9434765919"), 1);
}
#[test]
fn utr() {
assert_eq!(count("uk-utr", "UTR 12345 67890"), 1);
assert_eq!(count("uk-utr", "order 1234567890"), 0);
}
}
@@ -1,304 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! United States identifiers (§2.3), each from its issuer's published rules:
//! the SSA (SSN), the IRS (ITIN, EIN), the ABA (routing numbers), CMS (MBI,
//! NPI) and the DEA.
use super::{Detector, Findings, Region, Strength, checks, digits, stands_alone, word_near};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"us-ssn",
"US Social Security number",
Region::Us,
Strength::Checked,
ssn,
),
Detector::new("us-itin", "US ITIN", Region::Us, Strength::Checked, itin),
Detector::new("us-ein", "US EIN", Region::Us, Strength::NeedsWord, ein),
Detector::new(
"us-aba-routing",
"US bank routing number",
Region::Us,
Strength::NeedsWord,
aba_routing,
),
Detector::new(
"us-drivers-license",
"US driver's license",
Region::Us,
Strength::NeedsWord,
drivers_license,
),
Detector::new(
"us-mbi",
"US Medicare Beneficiary Identifier",
Region::Us,
Strength::Checked,
mbi,
),
Detector::new(
"us-npi",
"US National Provider Identifier",
Region::Us,
Strength::NeedsWord,
npi,
),
Detector::new(
"us-dea",
"US DEA registration number",
Region::Us,
Strength::Checked,
dea,
),
];
fn re(pattern: &str) -> Regex {
Regex::new(pattern).expect("detector pattern")
}
/// `AAA-GG-SSSS` (dashes or spaces), or nine bare digits.
static NINE: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{3})([ -]?)(\d{2})([ -]?)(\d{4})\b"));
/// Numbers the SSA has published as never valid: widely printed examples.
const SSN_EXAMPLES: &[&str] = &["078051120", "219099999"];
fn ssn_rules(area: u32, group: u32, serial: u32) -> bool {
area != 0 && area != 666 && area < 900 && group != 0 && serial != 0
}
const SSN_WORDS: &[&str] = &["ssn", "social security", "soc sec", "ss#", "ss no"];
fn ssn(text: &str, findings: &mut Findings) {
for c in NINE.captures_iter(text) {
let whole = c.get(0).unwrap();
let (area, group, serial) = (num(&c[1]), num(&c[3]), num(&c[5]));
let number = format!("{}{}{}", &c[1], &c[3], &c[5]);
// Written form (with both separators, the same one) stands alone;
// nine bare digits need a word
let written = !c[2].is_empty() && c[2] == c[4];
if ssn_rules(area, group, serial)
&& !SSN_EXAMPLES.contains(&number.as_str())
&& (written || word_near(text, whole.start(), whole.end(), SSN_WORDS))
{
findings.insert(number);
}
}
}
/// ITINs: 9XX, then a group in the IRS's ranges.
fn itin_group(group: u32) -> bool {
matches!(group, 50..=65 | 70..=88 | 90..=92 | 94..=99)
}
const ITIN_WORDS: &[&str] = &["itin", "taxpayer identification", "tax id"];
fn itin(text: &str, findings: &mut Findings) {
for c in NINE.captures_iter(text) {
let whole = c.get(0).unwrap();
let written = !c[2].is_empty() && c[2] == c[4];
if c[1].starts_with('9')
&& itin_group(num(&c[3]))
&& (written || word_near(text, whole.start(), whole.end(), ITIN_WORDS))
{
findings.insert(format!("{}{}{}", &c[1], &c[3], &c[5]));
}
}
}
static EIN: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{2})-?(\d{7})\b"));
/// The prefixes the IRS assigns to its campuses and internet EINs.
fn ein_prefix(prefix: u32) -> bool {
matches!(prefix, 1..=6 | 10..=16 | 20..=27 | 30..=48 | 50..=68 | 71..=77 | 80..=88 | 90..=95 | 98 | 99)
}
const EIN_WORDS: &[&str] = &[
"ein",
"fein",
"employer identification",
"tax id",
"tin",
"federal tax",
];
fn ein(text: &str, findings: &mut Findings) {
for c in EIN.captures_iter(text) {
let whole = c.get(0).unwrap();
if ein_prefix(num(&c[1])) && word_near(text, whole.start(), whole.end(), EIN_WORDS) {
findings.insert(format!("{}{}", &c[1], &c[2]));
}
}
}
static ROUTING: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{9}\b"));
/// The ABA check: 3, 7 and 1 weights, mod 10; and a Federal Reserve prefix.
pub fn aba_valid(n: &str) -> bool {
let d: Vec<u32> = n.bytes().map(|b| u32::from(b - b'0')).collect();
let prefix = d[0] * 10 + d[1];
matches!(prefix, 0..=12 | 21..=32 | 61..=72 | 80)
&& (3 * (d[0] + d[3] + d[6]) + 7 * (d[1] + d[4] + d[7]) + (d[2] + d[5] + d[8]))
.is_multiple_of(10)
}
const ROUTING_WORDS: &[&str] = &["routing", "aba", "rtn", "routing number", "transit"];
fn aba_routing(text: &str, findings: &mut Findings) {
for m in ROUTING.find_iter(text) {
// One random number in ten passes the check: always needs a word
if aba_valid(m.as_str()) && word_near(text, m.start(), m.end(), ROUTING_WORDS) {
findings.insert(m.as_str());
}
}
}
/// The shapes states issue: up to two letters, then 5–14 digits, dashes
/// allowed (Florida and Illinois print them).
static LICENSE: LazyLock<Regex> = LazyLock::new(|| re(r"\b[A-Z]{0,2}\d[\d-]{3,16}\d\b"));
const LICENSE_WORDS: &[&str] = &[
"driver's license",
"drivers license",
"driver license",
"driver's licence",
"dl",
"dl#",
"license number",
"lic no",
"dmv",
];
fn drivers_license(text: &str, findings: &mut Findings) {
for m in LICENSE.find_iter(text) {
let n = digits(m.as_str());
if (5..=14).contains(&n.len()) && word_near(text, m.start(), m.end(), LICENSE_WORDS) {
findings.insert(m.as_str().replace('-', ""));
}
}
}
/// CMS's MBI: 11 characters in a fixed pattern of digits, letters and
/// either, the letters S, L, O, I, B and Z never used; dashes may follow the
/// 4th and 7th.
static MBI: LazyLock<Regex> = LazyLock::new(|| {
let c = "[AC-HJKMNP-RT-Y]";
let an = "[AC-HJKMNP-RT-Y0-9]";
re(&format!(
r"\b[1-9]{c}{an}[0-9]-?{c}{an}[0-9]-?{c}{c}[0-9][0-9]\b"
))
});
fn mbi(text: &str, findings: &mut Findings) {
for m in MBI.find_iter(text) {
findings.insert(m.as_str().replace('-', ""));
}
}
static TEN: LazyLock<Regex> = LazyLock::new(|| re(r"\b[12]\d{9}\b"));
const NPI_WORDS: &[&str] = &["npi", "national provider", "provider id", "provider number"];
/// NPI: Luhn over the ISO card-issuer prefix 80840 and the number.
fn npi(text: &str, findings: &mut Findings) {
for m in TEN.find_iter(text) {
if checks::luhn(&format!("80840{}", m.as_str()))
&& word_near(text, m.start(), m.end(), NPI_WORDS)
{
findings.insert(m.as_str());
}
}
}
static DEA: LazyLock<Regex> = LazyLock::new(|| re(r"\b([ABCDEFGHJKLMPRSTUX][A-Z9])(\d{7})\b"));
/// DEA: (1st + 3rd + 5th) + 2 × (2nd + 4th + 6th) ends in the 7th digit.
fn dea(text: &str, findings: &mut Findings) {
for c in DEA.captures_iter(text) {
let d: Vec<u32> = c[2].bytes().map(|b| u32::from(b - b'0')).collect();
if ((d[0] + d[2] + d[4]) + 2 * (d[1] + d[3] + d[5])) % 10 == d[6] {
let whole = c.get(0).unwrap();
if stands_alone(text, whole.start(), whole.end()) {
findings.insert(whole.as_str());
}
}
}
}
fn num(s: &str) -> u32 {
s.parse().unwrap_or(0)
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn ssn() {
assert_eq!(count("us-ssn", "SSN 536-22-1234, also 536 22 1235"), 2);
// Bare digits: only with a word
assert_eq!(count("us-ssn", "ref 536221234"), 0);
assert_eq!(count("us-ssn", "social security: 536221234"), 1);
// Never issued, the SSA's printed examples, mixed separators
for bad in [
"000-12-3456",
"666-12-3456",
"912-12-3456",
"123-00-4567",
"123-45-0000",
"078-05-1120",
"536-22 1234",
] {
assert_eq!(count("us-ssn", bad), 0, "{bad}");
}
}
#[test]
fn itin_and_ein() {
assert_eq!(count("us-itin", "912-70-1234"), 1);
assert_eq!(count("us-itin", "912-69-1234"), 0);
assert_eq!(count("us-ssn", "912-70-1234"), 0);
assert_eq!(count("us-ein", "EIN: 12-3456789"), 1);
assert_eq!(count("us-ein", "part 12-3456789"), 0);
assert_eq!(count("us-ein", "EIN 07-3456789"), 0);
}
#[test]
fn routing_needs_a_word() {
assert_eq!(count("us-aba-routing", "Routing number 011000015"), 1);
assert_eq!(count("us-aba-routing", "ABA 021000021"), 1);
assert_eq!(count("us-aba-routing", "invoice 011000015"), 0);
assert_eq!(count("us-aba-routing", "routing 011000016"), 0);
}
#[test]
fn licenses() {
assert_eq!(count("us-drivers-license", "Driver's license: D1234567"), 1);
assert_eq!(count("us-drivers-license", "DL# S123-456-78-901-0"), 1);
assert_eq!(count("us-drivers-license", "Order D1234567"), 0);
}
#[test]
fn health_identifiers() {
// CMS's own MBI example
assert_eq!(count("us-mbi", "Medicare 1EG4-TE5-MK73"), 1);
assert_eq!(count("us-mbi", "1EG4TE5MK73"), 1);
assert_eq!(count("us-mbi", "1EG4-TE5-MK7S"), 0);
// CMS's NPI example
assert_eq!(count("us-npi", "NPI 1234567893"), 1);
assert_eq!(count("us-npi", "NPI 1234567894"), 0);
assert_eq!(count("us-npi", "call 1234567893"), 0);
assert_eq!(count("us-dea", "DEA AB1234563"), 1);
assert_eq!(count("us-dea", "AB1234564"), 0);
}
}
-695
View File
@@ -1,695 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Evaluating rules against a message (§2.1–§2.4). Rules are compiled once,
//! when they change: word lists become automata, patterns regexes. A message
//! is then checked against every enabled rule in order; each detector runs
//! at most once per message, and only when some rule asks for it.
//!
//! Pure: the caller parses the message, extracts attachment text
//! ([`super::extract`]) and knows the sender's groups and tenant. What comes
//! back is which rules matched, with each detector's count, and what DLP
//! decided; the matched text itself never leaves here (§2.7).
use super::{
detectors::{self, Findings},
extract::Extracted,
rules::{Action, Condition, Direction, Kind, Rule},
words::{Pattern, WordList},
};
use ahash::AHashMap;
use std::borrow::Cow;
/// Who sent a message, and to whom.
#[derive(Debug, Clone, Default)]
pub struct Envelope<'a> {
/// Outgoing (an authenticated sender) or incoming.
pub outgoing: bool,
pub sender: &'a str,
pub sender_groups: &'a [u32],
pub sender_tenant: Option<u32>,
pub recipients: Vec<Recipient<'a>>,
}
#[derive(Debug, Clone, Default)]
pub struct Recipient<'a> {
pub address: &'a str,
/// At a domain this server hosts.
pub local: bool,
pub groups: &'a [u32],
}
#[derive(Debug, Clone)]
pub struct Attachment<'a> {
pub name: Option<&'a str>,
/// Declared type, or detected where the caller knows better.
pub content_type: &'a str,
pub size: u64,
pub extracted: Extracted,
}
/// What rules look at.
#[derive(Debug, Clone, Default)]
pub struct Content<'a> {
pub subject: &'a str,
/// Each text and HTML part, as text.
pub bodies: Vec<Cow<'a, str>>,
pub headers: Vec<(&'a str, &'a str)>,
pub attachments: Vec<Attachment<'a>>,
pub size: u64,
/// Text past the inspection limit wasn't read.
pub truncated: bool,
}
impl Content<'_> {
fn texts(&self) -> impl Iterator<Item = &str> {
std::iter::once(self.subject)
.chain(self.bodies.iter().map(|b| b.as_ref()))
.chain(self.attachments.iter().filter_map(|a| match &a.extracted {
Extracted::Text(text) => Some(text.as_str()),
_ => None,
}))
}
fn cant_be_inspected(&self) -> bool {
self.truncated
|| self
.attachments
.iter()
.any(|a| matches!(a.extracted, Extracted::NotInspectable(_)))
}
}
enum Check {
Plain(Condition),
Words(WordList, u32),
Pattern(Pattern, u32),
Header {
name: String,
contains: Option<String>,
matches: Option<Pattern>,
},
AttachmentName(Pattern),
}
struct CompiledRule {
rule: Rule,
conditions: Vec<Check>,
exceptions: Vec<Check>,
}
/// The enabled rules, ready to run.
pub struct Compiled {
rules: Vec<CompiledRule>,
}
/// A rule reference, for notices and the audit record.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct RuleRef {
pub id: u32,
pub name: String,
pub notice: String,
}
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct Match {
pub rule_id: u32,
pub name: String,
pub kind: Kind,
pub actions: Vec<Action>,
/// Each detector (or `words`, `pattern`) that counted, and its count.
pub counts: Vec<(String, usize)>,
}
#[derive(Debug, Default)]
pub struct Outcome {
pub matched: Vec<Match>,
pub blocks: Vec<RuleRef>,
pub holds: Vec<(RuleRef, bool)>,
pub warns: Vec<RuleRef>,
}
/// What DLP decided, strictest first (§2.4).
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum Decision {
Pass,
Block(Vec<RuleRef>),
Hold {
rules: Vec<RuleRef>,
notify_sender: bool,
},
Warn(Vec<RuleRef>),
}
impl Outcome {
/// Block beats hold beats warn. An override (§2.5) answers the warnings
/// only: a block or hold still applies.
pub fn decision(&self, overridden: bool) -> Decision {
if !self.blocks.is_empty() {
Decision::Block(self.blocks.clone())
} else if !self.holds.is_empty() {
Decision::Hold {
rules: self.holds.iter().map(|(r, _)| r.clone()).collect(),
notify_sender: self.holds.iter().any(|(_, notify)| *notify),
}
} else if !self.warns.is_empty() && !overridden {
Decision::Warn(self.warns.clone())
} else {
Decision::Pass
}
}
}
fn compile_check(condition: &Condition) -> Result<Check, String> {
Ok(match condition {
Condition::Words { words, at_least } => Check::Words(WordList::new(words)?, *at_least),
Condition::Pattern { pattern, at_least } => {
Check::Pattern(Pattern::new(pattern)?, *at_least)
}
Condition::Header {
name,
contains,
matches,
} => Check::Header {
name: name.to_ascii_lowercase(),
contains: contains.as_ref().map(|c| c.to_lowercase()),
matches: matches.as_deref().map(Pattern::new).transpose()?,
},
Condition::AttachmentName { pattern } => Check::AttachmentName(Pattern::new(pattern)?),
other => Check::Plain(other.clone()),
})
}
impl Compiled {
/// Compiles the enabled rules; one that no longer compiles (a detector
/// renamed since it was saved) is skipped and named in the second list.
pub fn new(rules: &[Rule]) -> (Self, Vec<(u32, String)>) {
let mut compiled = Vec::new();
let mut skipped = Vec::new();
for rule in rules.iter().filter(|r| r.enabled) {
let result = rule.validate().map_err(|e| e.reason).and_then(|_| {
Ok(CompiledRule {
rule: rule.clone(),
conditions: rule
.conditions
.iter()
.map(compile_check)
.collect::<Result<_, _>>()?,
exceptions: rule
.exceptions
.iter()
.map(compile_check)
.collect::<Result<_, _>>()?,
})
});
match result {
Ok(c) => compiled.push(c),
Err(reason) => skipped.push((rule.id, reason)),
}
}
compiled.sort_by_key(|c| (c.rule.priority, c.rule.id));
(Self { rules: compiled }, skipped)
}
pub fn is_empty(&self) -> bool {
self.rules.is_empty()
}
/// Whether any rule could apply to mail going this way, so a caller can
/// skip parsing when none can.
pub fn applies_to(&self, outgoing: bool) -> bool {
self.rules
.iter()
.any(|c| direction_matches(c.rule.direction, outgoing))
}
pub fn evaluate(&self, envelope: &Envelope<'_>, content: &Content<'_>) -> Outcome {
let mut state = State {
content,
detected: AHashMap::new(),
};
let mut outcome = Outcome::default();
for compiled in &self.rules {
let rule = &compiled.rule;
if !direction_matches(rule.direction, envelope.outgoing) {
continue;
}
let mut counts = Vec::new();
let all_match = compiled
.conditions
.iter()
.all(|check| state.check(check, envelope, &mut counts));
if !all_match {
continue;
}
let mut ignored = Vec::new();
if compiled
.exceptions
.iter()
.any(|check| state.check(check, envelope, &mut ignored))
{
continue;
}
for action in &rule.actions {
let reference = |notice: &str| RuleRef {
id: rule.id,
name: rule.name.clone(),
notice: notice.to_string(),
};
match action {
Action::Block { notice } => outcome.blocks.push(reference(notice)),
Action::Hold {
notice,
notify_sender,
} => outcome.holds.push((reference(notice), *notify_sender)),
Action::Warn { notice } => outcome.warns.push(reference(notice)),
_ => {}
}
}
outcome.matched.push(Match {
rule_id: rule.id,
name: rule.name.clone(),
kind: rule.kind,
actions: rule.actions.clone(),
counts,
});
if rule.stop_processing {
break;
}
}
outcome
}
}
fn direction_matches(direction: Direction, outgoing: bool) -> bool {
match direction {
Direction::Any => true,
Direction::Outgoing => outgoing,
Direction::Incoming => !outgoing,
}
}
fn domain_of(address: &str) -> &str {
address.rsplit_once('@').map_or("", |(_, d)| d)
}
fn in_list(value: &str, list: &[String]) -> bool {
list.iter().any(|v| v.eq_ignore_ascii_case(value))
}
struct State<'c, 'a> {
content: &'c Content<'a>,
/// Each detector's count, run once per message.
detected: AHashMap<&'static str, usize>,
}
impl State<'_, '_> {
fn detector_count(&mut self, id: &str) -> usize {
let Some(detector) = detectors::by_id(id) else {
return 0;
};
if let Some(count) = self.detected.get(detector.id) {
return *count;
}
let mut findings = Findings::default();
for text in self.content.texts() {
detector.find(text, &mut findings);
}
self.detected.insert(detector.id, findings.len());
findings.len()
}
fn check(
&mut self,
check: &Check,
envelope: &Envelope<'_>,
counts: &mut Vec<(String, usize)>,
) -> bool {
let content = self.content;
match check {
Check::Words(list, at_least) => {
let n: usize = content.texts().map(|t| list.count(t)).sum();
counts.push(("words".into(), n));
n >= *at_least as usize
}
Check::Pattern(pattern, at_least) => {
let n: usize = content.texts().map(|t| pattern.count(t)).sum();
counts.push(("pattern".into(), n));
n >= *at_least as usize
}
Check::Header {
name,
contains,
matches,
} => content
.headers
.iter()
.filter(|(n, _)| n.eq_ignore_ascii_case(name))
.any(|(_, value)| match (contains, matches) {
(Some(needle), _) => value.to_lowercase().contains(needle.as_str()),
(_, Some(pattern)) => pattern.count(value) > 0,
_ => true,
}),
Check::AttachmentName(pattern) => content
.attachments
.iter()
.any(|a| a.name.is_some_and(|n| pattern.count(n) > 0)),
Check::Plain(condition) => match condition {
Condition::SenderAddress { addresses } => in_list(envelope.sender, addresses),
Condition::SenderDomain { domains } => in_list(domain_of(envelope.sender), domains),
Condition::SenderGroup { groups } => {
envelope.sender_groups.iter().any(|g| groups.contains(g))
}
Condition::SenderTenant { tenants } => {
envelope.sender_tenant.is_some_and(|t| tenants.contains(&t))
}
Condition::RecipientAddress { addresses } => envelope
.recipients
.iter()
.any(|r| in_list(r.address, addresses)),
Condition::RecipientDomain { domains } => envelope
.recipients
.iter()
.any(|r| in_list(domain_of(r.address), domains)),
Condition::RecipientGroup { groups } => envelope
.recipients
.iter()
.any(|r| r.groups.iter().any(|g| groups.contains(g))),
Condition::RecipientOutside => envelope.recipients.iter().any(|r| !r.local),
Condition::AttachmentType { types } => content.attachments.iter().any(|a| {
let ct = a.content_type.to_ascii_lowercase();
types
.iter()
.any(|t| ct.starts_with(&t.to_ascii_lowercase()))
}),
Condition::AttachmentExtension { extensions } => {
content.attachments.iter().any(|a| {
a.name
.and_then(|n| n.rsplit_once('.'))
.is_some_and(|(_, ext)| {
extensions
.iter()
.any(|e| e.trim_start_matches('.').eq_ignore_ascii_case(ext))
})
})
}
Condition::AttachmentSizeOver { bytes } => {
content.attachments.iter().any(|a| a.size > *bytes)
}
Condition::AttachmentCountOver { count } => {
content.attachments.len() > *count as usize
}
Condition::CantBeInspected => content.cant_be_inspected(),
Condition::MessageSizeOver { bytes } => content.size > *bytes,
Condition::Detected { detectors } => {
let mut any = false;
for d in detectors {
let n = self.detector_count(&d.id);
counts.push((d.id.clone(), n));
any |= n >= d.at_least as usize;
}
any
}
// Compiled into their own checks
Condition::Words { .. }
| Condition::Pattern { .. }
| Condition::Header { .. }
| Condition::AttachmentName { .. } => false,
},
}
}
}
#[cfg(test)]
mod tests {
use super::*;
use crate::mailflow::{
extract::Why,
rules::{DetectorMin, Position},
};
fn rule(id: u32, kind: Kind, conditions: Vec<Condition>, action: Action) -> Rule {
Rule {
id,
name: format!("rule {id}"),
description: String::new(),
kind,
enabled: true,
priority: id as i32,
direction: if kind == Kind::Dlp {
Direction::Outgoing
} else {
Direction::Any
},
conditions,
exceptions: vec![],
actions: vec![action],
stop_processing: false,
created_by: String::new(),
created_at: 0,
updated_at: 0,
}
}
fn envelope(outside: bool) -> Envelope<'static> {
Envelope {
outgoing: true,
sender: "[email protected]",
sender_groups: &[7],
sender_tenant: None,
recipients: vec![Recipient {
address: if outside {
"[email protected]"
} else {
"[email protected]"
},
local: !outside,
groups: &[],
}],
}
}
fn cards(n: usize) -> Content<'static> {
let body: String = [
"4242 4242 4242 4242",
"5555-5555-5555-4444",
"378282246310005",
"6011111111111117",
"3566002020360505",
]
.iter()
.take(n)
.map(|c| format!("card {c}\n"))
.collect();
Content {
subject: "Numbers",
bodies: vec![body.into()],
..Default::default()
}
}
fn five_cards_outside(action: Action) -> Rule {
rule(
1,
Kind::Dlp,
vec![
Condition::RecipientOutside,
Condition::Detected {
detectors: vec![DetectorMin {
id: "payment-card".into(),
at_least: 5,
}],
},
],
action,
)
}
#[test]
fn detector_threshold_and_recipients() {
let (rules, skipped) = Compiled::new(&[five_cards_outside(Action::Hold {
notice: "Held".into(),
notify_sender: true,
})]);
assert!(skipped.is_empty());
let outcome = rules.evaluate(&envelope(true), &cards(5));
assert_eq!(
outcome.matched[0].counts,
vec![("payment-card".to_string(), 5)]
);
assert!(matches!(
outcome.decision(false),
Decision::Hold {
notify_sender: true,
..
}
));
// Four cards, or everyone inside: nothing
assert_eq!(
rules.evaluate(&envelope(true), &cards(4)).decision(false),
Decision::Pass
);
assert_eq!(
rules.evaluate(&envelope(false), &cards(5)).decision(false),
Decision::Pass
);
}
#[test]
fn strictest_wins_and_override_answers_warnings_only() {
let warn = five_cards_outside(Action::Warn {
notice: "Sure?".into(),
});
let mut block = five_cards_outside(Action::Block {
notice: "No".into(),
});
block.id = 2;
let (rules, _) = Compiled::new(&[warn.clone(), block]);
let outcome = rules.evaluate(&envelope(true), &cards(5));
assert!(matches!(outcome.decision(true), Decision::Block(_)));
let (rules, _) = Compiled::new(&[warn]);
let outcome = rules.evaluate(&envelope(true), &cards(5));
assert!(matches!(outcome.decision(false), Decision::Warn(ref w) if w[0].notice == "Sure?"));
assert_eq!(outcome.decision(true), Decision::Pass);
}
#[test]
fn exceptions_order_and_stop_processing() {
let disclaimer = |id| {
rule(
id,
Kind::Transport,
vec![Condition::RecipientOutside],
Action::AddDisclaimer {
text: "t".into(),
html: None,
position: Position::Bottom,
},
)
};
let mut first = disclaimer(1);
first.stop_processing = true;
let (rules, _) = Compiled::new(&[disclaimer(2), first.clone()]);
let outcome = rules.evaluate(&envelope(true), &cards(0));
assert_eq!(
outcome
.matched
.iter()
.map(|m| m.rule_id)
.collect::<Vec<_>>(),
vec![1]
);
first.stop_processing = false;
first.exceptions = vec![Condition::SenderGroup { groups: vec![7] }];
let (rules, _) = Compiled::new(&[disclaimer(2), first]);
let outcome = rules.evaluate(&envelope(true), &cards(0));
assert_eq!(
outcome
.matched
.iter()
.map(|m| m.rule_id)
.collect::<Vec<_>>(),
vec![2]
);
}
#[test]
fn content_conditions() {
let content = Content {
subject: "Project Falcon",
bodies: vec!["see attached".into()],
headers: vec![("X-Class", "Internal only")],
attachments: vec![
Attachment {
name: Some("plan.docx"),
content_type: "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
size: 40_000,
extracted: Extracted::Text("IBAN GB29 NWBK 6016 1331 9268 19".into()),
},
Attachment {
name: Some("scan.pdf"),
content_type: "application/pdf",
size: 900_000,
extracted: Extracted::NotInspectable(Why::Pdf),
},
],
size: 1_000_000,
truncated: false,
};
let block = || Action::Block { notice: "n".into() };
let checks = [
(
Condition::Words {
words: vec!["project falcon".into()],
at_least: 1,
},
true,
),
(
Condition::Header {
name: "x-class".into(),
contains: Some("internal".into()),
matches: None,
},
true,
),
(
Condition::AttachmentExtension {
extensions: vec![".PDF".into()],
},
true,
),
(
Condition::AttachmentType {
types: vec!["image/".into()],
},
false,
),
(Condition::AttachmentSizeOver { bytes: 500_000 }, true),
(Condition::AttachmentCountOver { count: 2 }, false),
(Condition::CantBeInspected, true),
(Condition::MessageSizeOver { bytes: 2_000_000 }, false),
(
Condition::Detected {
detectors: vec![DetectorMin {
id: "iban".into(),
at_least: 1,
}],
},
true,
),
(
Condition::SenderDomain {
domains: vec!["EXAMPLE.com".into()],
},
true,
),
];
for (condition, expected) in checks {
let (rules, skipped) =
Compiled::new(&[rule(1, Kind::Dlp, vec![condition.clone()], block())]);
assert!(skipped.is_empty(), "{condition:?}");
let matched = !rules.evaluate(&envelope(true), &content).matched.is_empty();
assert_eq!(matched, expected, "{condition:?}");
}
}
#[test]
fn direction_and_disabled_rules() {
let mut r = five_cards_outside(Action::Block { notice: "n".into() });
let (rules, _) = Compiled::new(std::slice::from_ref(&r));
assert!(rules.applies_to(true) && !rules.applies_to(false));
let mut incoming = envelope(true);
incoming.outgoing = false;
assert_eq!(
rules.evaluate(&incoming, &cards(5)).decision(false),
Decision::Pass
);
r.enabled = false;
assert!(Compiled::new(&[r]).0.is_empty());
}
}
-694
View File
@@ -1,694 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! The text of an attachment, for the detectors (§2.3), or why there isn't
//! one.
//!
//! Read: text files (plain, CSV, JSON, XML, HTML), Office Open XML (DOCX,
//! XLSX, PPTX) and OpenDocument (ODT, ODS, ODP) documents, and ZIP archives
//! one level deep. **Can't be inspected**: encrypted or password-protected
//! files, PDF (settled answer 2), the older binary Office formats, archives
//! inside archives, and anything past the limits. Everything else (images,
//! audio, programs) has no text to read and is neither.
//!
//! Office files are ZIP archives of XML, read here with the `zip` and
//! `quick-xml` crates the server already uses: no outside converter runs.
use quick_xml::{Reader, XmlVersion, events::Event};
use std::io::{Cursor, Read};
/// How much may be unpacked from one attachment, and from how many entries.
#[derive(Debug, Clone, Copy)]
pub struct Limits {
pub max_unpacked: u64,
pub max_entries: usize,
}
impl Default for Limits {
fn default() -> Self {
Self {
max_unpacked: 50 * 1024 * 1024,
max_entries: 10_000,
}
}
}
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum Extracted {
/// The text to check.
Text(String),
/// A kind of file with no text in it: nothing to check, nothing missed.
NoText,
/// A file that may hold text the detectors couldn't read.
NotInspectable(Why),
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Why {
Encrypted,
Pdf,
LegacyOffice,
NestedArchive,
TooLarge,
Damaged,
}
impl Why {
pub fn as_str(&self) -> &'static str {
match self {
Why::Encrypted => "encrypted",
Why::Pdf => "pdf",
Why::LegacyOffice => "legacy-office",
Why::NestedArchive => "nested-archive",
Why::TooLarge => "too-large",
Why::Damaged => "damaged",
}
}
}
const OLE_MAGIC: &[u8] = &[0xD0, 0xCF, 0x11, 0xE0, 0xA1, 0xB1, 0x1A, 0xE1];
const ZIP_MAGIC: &[u8] = b"PK\x03\x04";
/// What an attachment says, from its declared type, its file name and, above
/// all, its first bytes.
pub fn extract(
content_type: &str,
file_name: Option<&str>,
data: &[u8],
limits: &Limits,
) -> Extracted {
extract_at(content_type, file_name, data, limits, 0)
}
fn extract_at(
content_type: &str,
file_name: Option<&str>,
data: &[u8],
limits: &Limits,
depth: u8,
) -> Extracted {
let content_type = content_type.to_ascii_lowercase();
let extension = file_name
.and_then(|name| name.rsplit_once('.'))
.map(|(_, ext)| ext.to_ascii_lowercase())
.unwrap_or_default();
if data.len() as u64 > limits.max_unpacked {
return Extracted::NotInspectable(Why::TooLarge);
}
if data.starts_with(b"%PDF-") || content_type == "application/pdf" || extension == "pdf" {
return Extracted::NotInspectable(Why::Pdf);
}
if data.starts_with(OLE_MAGIC) {
// An encrypted OOXML file is an OLE container holding the encrypted
// package; any other OLE file is a legacy .doc, .xls or .ppt
return Extracted::NotInspectable(if has_utf16(data, "EncryptedPackage") {
Why::Encrypted
} else {
Why::LegacyOffice
});
}
if data.starts_with(ZIP_MAGIC) {
if depth > 0 {
return Extracted::NotInspectable(Why::NestedArchive);
}
return zip(data, limits);
}
if is_text(&content_type, &extension) {
let text = decode_text(data);
return Extracted::Text(
if content_type == "text/html" || matches!(extension.as_str(), "html" | "htm") {
strip_html(&text)
} else {
text
},
);
}
Extracted::NoText
}
fn is_text(content_type: &str, extension: &str) -> bool {
content_type.starts_with("text/")
|| matches!(
content_type,
"application/json"
| "application/xml"
| "application/csv"
| "application/x-csv"
| "message/rfc822"
)
|| matches!(
extension,
"txt"
| "csv"
| "tsv"
| "json"
| "xml"
| "md"
| "log"
| "html"
| "htm"
| "eml"
| "ics"
| "vcf"
)
}
/// UTF-16 with a byte order mark, else UTF-8 (lossy).
fn decode_text(data: &[u8]) -> String {
let utf16 = |bytes: &[u8], big: bool| {
let units: Vec<u16> = bytes
.as_chunks::<2>()
.0
.iter()
.map(|&c| {
if big {
u16::from_be_bytes(c)
} else {
u16::from_le_bytes(c)
}
})
.collect();
String::from_utf16_lossy(&units)
};
match data {
[0xFF, 0xFE, rest @ ..] => utf16(rest, false),
[0xFE, 0xFF, rest @ ..] => utf16(rest, true),
[0xEF, 0xBB, 0xBF, rest @ ..] => String::from_utf8_lossy(rest).into_owned(),
_ => String::from_utf8_lossy(data).into_owned(),
}
}
fn has_utf16(data: &[u8], needle: &str) -> bool {
let needle: Vec<u8> = needle.encode_utf16().flat_map(u16::to_le_bytes).collect();
data.windows(needle.len()).any(|w| w == needle.as_slice())
}
/// Tags out, the common entities decoded, block ends as new lines.
fn strip_html(html: &str) -> String {
let mut out = String::with_capacity(html.len());
let mut in_tag = false;
let mut skip_until: Option<&str> = None;
let lower = html.to_ascii_lowercase();
let mut i = 0;
let bytes = html.as_bytes();
while i < bytes.len() {
if let Some(end) = skip_until {
match lower[i..].find(end) {
Some(at) => {
i += at + end.len();
skip_until = None;
}
None => break,
}
continue;
}
let c = bytes[i];
if in_tag {
if c == b'>' {
in_tag = false;
}
i += 1;
continue;
}
if c == b'<' {
if lower[i..].starts_with("<script") {
skip_until = Some("</script>");
} else if lower[i..].starts_with("<style") {
skip_until = Some("</style>");
} else {
if [
"<br", "<p", "</p", "<div", "</div", "<tr", "<li", "<td", "<th",
]
.iter()
.any(|t| lower[i..].starts_with(t))
{
out.push(
if lower[i..].starts_with("<td") || lower[i..].starts_with("<th") {
'\t'
} else {
'\n'
},
);
}
in_tag = true;
}
i += 1;
continue;
}
// Copy up to the next tag
let next = html[i..].find('<').map_or(html.len(), |at| i + at);
out.push_str(&html[i..next]);
i = next;
}
for (entity, text) in [
("&nbsp;", " "),
("&lt;", "<"),
("&gt;", ">"),
("&quot;", "\""),
("&#39;", "'"),
("&amp;", "&"),
] {
out = out.replace(entity, text);
}
out
}
/// A ZIP file: an Office document, an OpenDocument, or an archive.
fn zip(data: &[u8], limits: &Limits) -> Extracted {
let Ok(mut archive) = zip::ZipArchive::new(Cursor::new(data)) else {
return Extracted::NotInspectable(Why::Damaged);
};
if archive.len() > limits.max_entries {
return Extracted::NotInspectable(Why::TooLarge);
}
let mut names = Vec::with_capacity(archive.len());
let mut declared: u64 = 0;
for i in 0..archive.len() {
let Ok(entry) = archive.by_index_raw(i) else {
return Extracted::NotInspectable(Why::Damaged);
};
if entry.encrypted() {
return Extracted::NotInspectable(Why::Encrypted);
}
declared = declared.saturating_add(entry.size());
names.push(entry.name().to_string());
}
if declared > limits.max_unpacked {
return Extracted::NotInspectable(Why::TooLarge);
}
let mut budget = limits.max_unpacked;
let mut read =
|archive: &mut zip::ZipArchive<Cursor<&[u8]>>, name: &str| -> Result<Vec<u8>, Why> {
let entry = archive.by_name(name).map_err(|_| Why::Damaged)?;
let mut bytes = Vec::new();
// Declared sizes can lie: stop at the budget whatever they say
entry
.take(budget + 1)
.read_to_end(&mut bytes)
.map_err(|_| Why::Damaged)?;
if bytes.len() as u64 > budget {
return Err(Why::TooLarge);
}
budget -= bytes.len() as u64;
Ok(bytes)
};
let has = |name: &str| names.iter().any(|n| n == name);
let mut text = String::new();
let result: Result<(), Why> = (|| {
if has("[Content_Types].xml") {
// Office Open XML: the parts that hold what a person wrote
let mut shared = Vec::new();
if has("xl/sharedStrings.xml") {
shared = xml_strings(&read(&mut archive, "xl/sharedStrings.xml")?, "si");
}
for name in names.iter().filter(|n| ooxml_text_part(n)) {
let xml = read(&mut archive, name)?;
if name.starts_with("xl/worksheets/") {
xlsx_sheet(&xml, &mut text);
} else {
xml_text(&xml, &mut text);
}
text.push('\n');
}
text.extend(shared.iter().map(|s| format!("{s}\n")));
} else if names.first().is_some_and(|n| n == "mimetype")
&& read(&mut archive, "mimetype")?.starts_with(b"application/vnd.oasis.opendocument")
{
// OpenDocument: an encrypted one says so in its manifest
if has("META-INF/manifest.xml")
&& contains(
&read(&mut archive, "META-INF/manifest.xml")?,
b"encryption-data",
)
{
return Err(Why::Encrypted);
}
for name in ["content.xml", "styles.xml"] {
if has(name) {
xml_text(&read(&mut archive, name)?, &mut text);
text.push('\n');
}
}
} else {
// An archive: each file inside, one level deep
for name in names.iter().filter(|n| !n.ends_with('/')) {
let bytes = read(&mut archive, name)?;
match extract_at("", Some(name), &bytes, limits, 1) {
Extracted::Text(inner) => {
text.push_str(&inner);
text.push('\n');
}
Extracted::NoText => {}
Extracted::NotInspectable(why) => return Err(why),
}
}
}
Ok(())
})();
match result {
Ok(()) => Extracted::Text(text),
Err(why) => Extracted::NotInspectable(why),
}
}
fn ooxml_text_part(name: &str) -> bool {
let xml = name.ends_with(".xml");
xml && (name == "word/document.xml"
|| [
"word/header",
"word/footer",
"word/footnotes",
"word/endnotes",
"word/comments",
]
.iter()
.any(|p| name.starts_with(p))
|| name.starts_with("xl/worksheets/sheet")
|| name.starts_with("ppt/slides/slide")
|| name.starts_with("ppt/notesSlides/"))
}
fn contains(haystack: &[u8], needle: &[u8]) -> bool {
haystack.windows(needle.len()).any(|w| w == needle)
}
/// The local name of a tag, without its namespace prefix.
fn local(name: &[u8]) -> &[u8] {
name.rsplit(|b| *b == b':').next().unwrap_or(name)
}
fn push_entity(entity: &[u8], out: &mut String) {
match entity {
b"lt" => out.push('<'),
b"gt" => out.push('>'),
b"amp" => out.push('&'),
b"apos" => out.push('\''),
b"quot" => out.push('"'),
_ => {
let code = match entity {
[b'#', b'x' | b'X', hex @ ..] => std::str::from_utf8(hex)
.ok()
.and_then(|h| u32::from_str_radix(h, 16).ok()),
[b'#', dec @ ..] => std::str::from_utf8(dec).ok().and_then(|d| d.parse().ok()),
_ => None,
};
if let Some(c) = code.and_then(char::from_u32) {
out.push(c);
}
}
}
}
/// Every text node, runs joined as written, a new line after each paragraph
/// or row and a tab after each cell, so a number split across runs is whole
/// again.
fn xml_text(xml: &[u8], out: &mut String) {
let mut reader = Reader::from_reader(xml);
let mut buf = Vec::new();
loop {
match reader.read_event_into(&mut buf) {
Ok(Event::Text(t)) => {
if let Ok(text) = t.xml_content(XmlVersion::Implicit1_0) {
out.push_str(&text);
}
}
Ok(Event::CData(t)) => out.push_str(&String::from_utf8_lossy(&t)),
Ok(Event::GeneralRef(entity)) => push_entity(&entity, out),
Ok(Event::End(e)) => match local(e.name().as_ref()) {
b"p" | b"h" | b"tr" | b"row" | b"table-row" | b"br" => out.push('\n'),
b"tc" | b"c" | b"table-cell" | b"tab" => out.push('\t'),
_ => {}
},
Ok(Event::Empty(e)) => match local(e.name().as_ref()) {
b"br" | b"line-break" => out.push('\n'),
b"tab" | b"s" => out.push(' '),
_ => {}
},
Ok(Event::Eof) | Err(_) => break,
_ => {}
}
buf.clear();
}
}
/// The text of each `item` element (a shared string in XLSX).
fn xml_strings(xml: &[u8], item: &str) -> Vec<String> {
let mut reader = Reader::from_reader(xml);
let mut buf = Vec::new();
let mut items = Vec::new();
let mut current: Option<String> = None;
loop {
match reader.read_event_into(&mut buf) {
Ok(Event::Start(e)) if local(e.name().as_ref()) == item.as_bytes() => {
current = Some(String::new())
}
Ok(Event::End(e)) if local(e.name().as_ref()) == item.as_bytes() => {
items.extend(current.take());
}
Ok(Event::Text(t)) => {
if let (Some(s), Ok(text)) =
(current.as_mut(), t.xml_content(XmlVersion::Implicit1_0))
{
s.push_str(&text);
}
}
Ok(Event::GeneralRef(entity)) => {
if let Some(s) = current.as_mut() {
push_entity(&entity, s);
}
}
Ok(Event::Eof) | Err(_) => break,
_ => {}
}
buf.clear();
}
items
}
/// A worksheet's cell values: numbers and inline strings. Cells holding a
/// shared string are skipped here; the shared strings are read whole.
fn xlsx_sheet(xml: &[u8], out: &mut String) {
let mut reader = Reader::from_reader(xml);
let mut buf = Vec::new();
let mut shared_cell = false;
let mut in_value = false;
loop {
match reader.read_event_into(&mut buf) {
Ok(Event::Start(e)) => match local(e.name().as_ref()) {
b"c" => {
shared_cell = e
.attributes()
.flatten()
.any(|a| a.key.as_ref() == b"t" && a.value.as_ref() == b"s");
}
b"v" | b"t" => in_value = true,
_ => {}
},
Ok(Event::End(e)) => match local(e.name().as_ref()) {
b"v" | b"t" => in_value = false,
b"c" => out.push('\t'),
b"row" => out.push('\n'),
_ => {}
},
// A shared string's cell holds only its index: the string itself
// is added with the shared strings
Ok(Event::Text(t)) if in_value && !shared_cell => {
if let Ok(text) = t.xml_content(XmlVersion::Implicit1_0) {
out.push_str(&text);
}
}
Ok(Event::Eof) | Err(_) => break,
_ => {}
}
buf.clear();
}
}
#[cfg(test)]
mod tests {
use super::*;
use std::io::Write;
use zip::{ZipWriter, write::SimpleFileOptions};
fn zip_of(files: &[(&str, &str)]) -> Vec<u8> {
let mut zip = ZipWriter::new(Cursor::new(Vec::new()));
for (name, body) in files {
zip.start_file(*name, SimpleFileOptions::default()).unwrap();
zip.write_all(body.as_bytes()).unwrap();
}
zip.finish().unwrap().into_inner()
}
fn text_of(extracted: Extracted) -> String {
match extracted {
Extracted::Text(text) => text,
other => panic!("expected text, got {other:?}"),
}
}
#[test]
fn plain_text_and_html() {
let limits = Limits::default();
assert_eq!(
text_of(extract("text/plain", None, b"card 4242", &limits)),
"card 4242"
);
let utf16: Vec<u8> = [0xFF, 0xFE]
.into_iter()
.chain("héllo".encode_utf16().flat_map(u16::to_le_bytes))
.collect();
assert_eq!(
text_of(extract(
"application/octet-stream",
Some("a.csv"),
&utf16,
&limits
)),
"héllo"
);
let html = "<html><style>p{}</style><p>Card&nbsp;4242</p><script>x()</script><td>a</td><td>b</td></html>";
let text = text_of(extract("text/html", None, html.as_bytes(), &limits));
assert!(
text.contains("Card 4242") && !text.contains("x()") && !text.contains("p{}"),
"{text:?}"
);
assert_eq!(
extract("image/png", Some("a.png"), b"\x89PNG....", &limits),
Extracted::NoText
);
}
#[test]
fn docx_joins_split_runs() {
let doc = r#"<w:document xmlns:w="w"><w:body><w:p><w:r><w:t>Card 4242 42</w:t></w:r><w:r><w:t>42 4242 4242</w:t></w:r></w:p><w:p><w:r><w:t>A &amp; B</w:t></w:r></w:p></w:body></w:document>"#;
let docx = zip_of(&[
("[Content_Types].xml", "<Types/>"),
("word/document.xml", doc),
]);
let text = text_of(extract(
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
Some("a.docx"),
&docx,
&Limits::default(),
));
assert!(text.contains("Card 4242 4242 4242 4242\nA & B"), "{text:?}");
}
#[test]
fn xlsx_numbers_and_shared_strings() {
let sheet = r#"<worksheet><sheetData><row><c r="A1" t="s"><v>0</v></c><c r="B1"><v>4242424242424242</v></c></row></sheetData></worksheet>"#;
let shared = r#"<sst><si><t>IBAN GB29 NWBK 6016 1331 9268 19</t></si></sst>"#;
let xlsx = zip_of(&[
("[Content_Types].xml", "<Types/>"),
("xl/sharedStrings.xml", shared),
("xl/worksheets/sheet1.xml", sheet),
]);
let text = text_of(extract("", Some("book.xlsx"), &xlsx, &Limits::default()));
assert!(
text.contains("4242424242424242") && text.contains("GB29 NWBK 6016 1331 9268 19"),
"{text:?}"
);
// The shared string's index isn't read as a value
assert!(
!text.contains("\t0\t") && !text.starts_with('0'),
"{text:?}"
);
}
#[test]
fn opendocument_and_encrypted_opendocument() {
let content = r#"<office:document-content xmlns:text="t"><text:p>SSN 078-05-1120</text:p></office:document-content>"#;
let odt = zip_of(&[
("mimetype", "application/vnd.oasis.opendocument.text"),
("content.xml", content),
]);
assert!(
text_of(extract("", Some("a.odt"), &odt, &Limits::default()))
.contains("SSN 078-05-1120")
);
let manifest = r#"<manifest:manifest><manifest:file-entry><manifest:encryption-data/></manifest:file-entry></manifest:manifest>"#;
let locked = zip_of(&[
("mimetype", "application/vnd.oasis.opendocument.text"),
("META-INF/manifest.xml", manifest),
("content.xml", "x"),
]);
assert_eq!(
extract("", Some("a.odt"), &locked, &Limits::default()),
Extracted::NotInspectable(Why::Encrypted)
);
}
#[test]
fn archives() {
let limits = Limits::default();
let archive = zip_of(&[
("notes/a.txt", "card 4242424242424242"),
("b.png", "\u{89}PNG"),
]);
assert!(
text_of(extract("application/zip", Some("x.zip"), &archive, &limits))
.contains("4242424242424242")
);
let nested = zip_of(&[(
"inner.zip",
std::str::from_utf8(&[b'P', b'K', 3, 4]).unwrap(),
)]);
assert_eq!(
extract("application/zip", Some("x.zip"), &nested, &limits),
Extracted::NotInspectable(Why::NestedArchive)
);
// Password-protected
let mut zip = ZipWriter::new(Cursor::new(Vec::new()));
zip.start_file(
"secret.txt",
SimpleFileOptions::default().with_aes_encryption(zip::AesMode::Aes256, "pw"),
)
.unwrap();
zip.write_all(b"4242424242424242").unwrap();
let locked = zip.finish().unwrap().into_inner();
assert_eq!(
extract("application/zip", Some("x.zip"), &locked, &limits),
Extracted::NotInspectable(Why::Encrypted)
);
// Past the limits
let small = Limits {
max_unpacked: 10,
max_entries: 1,
};
assert_eq!(
extract("application/zip", Some("x.zip"), &archive, &small),
Extracted::NotInspectable(Why::TooLarge)
);
assert_eq!(
extract("application/zip", None, b"PK\x03\x04garbage", &limits),
Extracted::NotInspectable(Why::Damaged)
);
}
#[test]
fn not_inspectable_kinds() {
let limits = Limits::default();
assert_eq!(
extract("application/octet-stream", None, b"%PDF-1.7 ...", &limits),
Extracted::NotInspectable(Why::Pdf)
);
let mut ole = OLE_MAGIC.to_vec();
ole.extend(std::iter::repeat_n(0, 64));
assert_eq!(
extract("", Some("old.doc"), &ole, &limits),
Extracted::NotInspectable(Why::LegacyOffice)
);
ole.extend("EncryptedPackage".encode_utf16().flat_map(u16::to_le_bytes));
assert_eq!(
extract("", Some("new.docx"), &ole, &limits),
Extracted::NotInspectable(Why::Encrypted)
);
}
}
-29
View File
@@ -1,29 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Data loss prevention and mail flow rules (dlp-and-mail-flow-rules spec).
//!
//! Mostly pure functions over text and attachment bytes, unit-tested
//! without a server:
//!
//! - [`detectors`]: find identifiers in text (payment cards, IBANs,
//! national ID numbers, keys), each by its published format and check
//! (§2.3);
//! - [`words`]: an organization's own word lists and patterns;
//! - [`extract`]: the text of an attachment, or why it can't be read;
//! - [`rules`]: what a rule is, its checks, and where rules are kept;
//! - [`engine`]: rules compiled and run against a message;
//! - [`cache`]: each node's compiled copy.
//!
//! Nothing here writes what it finds anywhere: callers get counts, and the
//! matched text never leaves the evaluation (§2.7).
pub mod cache;
pub mod detectors;
pub mod engine;
pub mod extract;
pub mod rules;
pub mod words;
-717
View File
@@ -1,717 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Mail flow rules and DLP rules (dlp-and-mail-flow-rules spec, §2.2–§2.4):
//! what a rule is, what makes one valid, and where it's kept.
//!
//! Kept in the fork's subspace (`store::SUBSPACE_INBUXA`), never in the
//! registry, so an upstream schema import never touches them. Every key
//! starts with `R`, then one byte for the kind:
//!
//! - `r` + rule id (u32): the rule, as JSON.
//!
//! Numbers are big-endian. There are few rules, so they're read whole.
use super::{detectors, words};
use serde::{Deserialize as SerdeDeserialize, Serialize as SerdeSerialize, de::DeserializeOwned};
use store::{
Deserialize, IterateParams, SUBSPACE_INBUXA, Serialize, Store, ValueKey,
write::{AnyClass, BatchBuilder, ValueClass, assert::AssertValue},
};
use trc::AddContext;
const FEATURE: u8 = b'R';
const KIND_RULE: u8 = b'r';
const CREATE_ATTEMPTS: usize = 5;
/// Longest text a rule may carry (a notice, a disclaimer), in bytes.
const MAX_TEXT: usize = 16 * 1024;
/// Most entries in one list (words, addresses, domains).
const MAX_LIST: usize = 5_000;
#[derive(Debug, Clone, Copy, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
pub enum Kind {
Dlp,
Transport,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
pub enum Direction {
/// Mail an authenticated sender submits, over SMTP or JMAP.
Outgoing,
/// Everything else the server accepts.
Incoming,
Any,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
pub enum Position {
Top,
Bottom,
}
fn one() -> u32 {
1
}
/// A detector and the least it must find.
#[derive(Debug, Clone, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
pub struct DetectorMin {
pub id: String,
#[serde(default = "one")]
pub at_least: u32,
}
#[derive(Debug, Clone, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
#[serde(
tag = "type",
rename_all = "camelCase",
rename_all_fields = "camelCase"
)]
pub enum Condition {
SenderAddress {
addresses: Vec<String>,
},
SenderDomain {
domains: Vec<String>,
},
SenderGroup {
groups: Vec<u32>,
},
SenderTenant {
tenants: Vec<u32>,
},
/// Any recipient is one of these.
RecipientAddress {
addresses: Vec<String>,
},
RecipientDomain {
domains: Vec<String>,
},
RecipientGroup {
groups: Vec<u32>,
},
/// Any recipient isn't at a domain this server hosts.
RecipientOutside,
/// Words or phrases in the subject, body or readable attachments.
Words {
words: Vec<String>,
#[serde(default = "one")]
at_least: u32,
},
/// The organization's regular expression, in the same places.
Pattern {
pattern: String,
#[serde(default = "one")]
at_least: u32,
},
/// A header exists, or its value contains or matches.
Header {
name: String,
#[serde(default)]
contains: Option<String>,
#[serde(default)]
matches: Option<String>,
},
/// An attachment's declared or detected type starts with one of these.
AttachmentType {
types: Vec<String>,
},
AttachmentExtension {
extensions: Vec<String>,
},
AttachmentName {
pattern: String,
},
AttachmentSizeOver {
bytes: u64,
},
AttachmentCountOver {
count: u32,
},
/// An attachment is encrypted, a PDF, a legacy Office file, an archive
/// inside an archive, or past the inspection limit.
CantBeInspected,
MessageSizeOver {
bytes: u64,
},
/// Any of these detectors finds at least its minimum (DLP rules only).
Detected {
detectors: Vec<DetectorMin>,
},
}
#[derive(Debug, Clone, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
#[serde(
tag = "type",
rename_all = "camelCase",
rename_all_fields = "camelCase"
)]
pub enum Action {
// Transport actions
AddDisclaimer {
text: String,
#[serde(default)]
html: Option<String>,
position: Position,
},
AddHeader {
name: String,
value: String,
},
RemoveHeader {
name: String,
},
PrefixSubject {
text: String,
},
AddRecipient {
address: String,
},
Redirect {
addresses: Vec<String>,
},
Refuse {
text: String,
},
Route {
queue: String,
},
// DLP actions
Block {
notice: String,
},
Warn {
notice: String,
},
Hold {
notice: String,
#[serde(default)]
notify_sender: bool,
},
}
impl Action {
pub fn is_dlp(&self) -> bool {
matches!(
self,
Action::Block { .. } | Action::Warn { .. } | Action::Hold { .. }
)
}
}
#[derive(Debug, Clone, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
#[serde(rename_all = "camelCase")]
pub struct Rule {
#[serde(default)]
pub id: u32,
pub name: String,
#[serde(default)]
pub description: String,
pub kind: Kind,
#[serde(default = "enabled")]
pub enabled: bool,
#[serde(default)]
pub priority: i32,
pub direction: Direction,
#[serde(default)]
pub conditions: Vec<Condition>,
#[serde(default)]
pub exceptions: Vec<Condition>,
pub actions: Vec<Action>,
#[serde(default)]
pub stop_processing: bool,
#[serde(default)]
pub created_by: String,
#[serde(default)]
pub created_at: u64,
#[serde(default)]
pub updated_at: u64,
}
fn enabled() -> bool {
true
}
/// Why a rule can't be saved: the property at fault, and a sentence.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct Invalid {
pub property: &'static str,
pub reason: String,
}
fn invalid(property: &'static str, reason: impl Into<String>) -> Invalid {
Invalid {
property,
reason: reason.into(),
}
}
impl Rule {
/// Everything that can be checked without the rest of the server: the
/// shape (§2.2, §2.4), the detectors, word lists and patterns.
pub fn validate(&self) -> Result<(), Invalid> {
if self.name.trim().is_empty() {
return Err(invalid("name", "A rule needs a name."));
}
if self.name.len() > 200 || self.description.len() > MAX_TEXT {
return Err(invalid("name", "The name or description is too long."));
}
if self.actions.is_empty() {
return Err(invalid("actions", "A rule needs something to do."));
}
let dlp_actions = self.actions.iter().filter(|a| a.is_dlp()).count();
match self.kind {
Kind::Dlp => {
if self.direction != Direction::Outgoing {
return Err(invalid("direction", "DLP rules check outgoing mail only."));
}
if dlp_actions != 1 || self.actions.len() != 1 {
return Err(invalid(
"actions",
"A DLP rule has exactly one action: block, warn or hold.",
));
}
}
Kind::Transport => {
if dlp_actions > 0 {
return Err(invalid(
"actions",
"Block, warn and hold belong to DLP rules.",
));
}
if self
.conditions
.iter()
.chain(&self.exceptions)
.any(|c| matches!(c, Condition::Detected { .. }))
{
return Err(invalid("conditions", "Detectors belong to DLP rules."));
}
}
}
for (property, list) in [
("conditions", &self.conditions),
("exceptions", &self.exceptions),
] {
for condition in list {
validate_condition(condition).map_err(|reason| invalid(property, reason))?;
}
}
for action in &self.actions {
validate_action(action).map_err(|reason| invalid("actions", reason))?;
}
Ok(())
}
}
fn nonempty_list<T>(list: &[T], what: &str) -> Result<(), String> {
if list.is_empty() {
Err(format!("The {what} list is empty."))
} else if list.len() > MAX_LIST {
Err(format!(
"The {what} list is longer than {MAX_LIST} entries."
))
} else {
Ok(())
}
}
fn header_name(name: &str) -> Result<(), String> {
if !name.is_empty()
&& name.len() <= 100
&& name.bytes().all(|b| b.is_ascii_graphic() && b != b':')
{
Ok(())
} else {
Err(format!("\"{name}\" isn't a header name."))
}
}
fn text(value: &str, what: &str) -> Result<(), String> {
if value.trim().is_empty() {
Err(format!("The {what} is empty."))
} else if value.len() > MAX_TEXT {
Err(format!("The {what} is longer than {MAX_TEXT} bytes."))
} else {
Ok(())
}
}
fn validate_condition(condition: &Condition) -> Result<(), String> {
match condition {
Condition::SenderAddress { addresses } | Condition::RecipientAddress { addresses } => {
nonempty_list(addresses, "address")
}
Condition::SenderDomain { domains } | Condition::RecipientDomain { domains } => {
nonempty_list(domains, "domain")
}
Condition::SenderGroup { groups } | Condition::RecipientGroup { groups } => {
nonempty_list(groups, "group")
}
Condition::SenderTenant { tenants } => nonempty_list(tenants, "tenant"),
Condition::Words { words, at_least } => {
nonempty_list(words, "word")?;
if *at_least == 0 {
return Err("The least number of words must be 1 or more.".into());
}
words::WordList::new(words).map(|_| ())
}
Condition::Pattern { pattern, at_least } => {
if *at_least == 0 {
return Err("The least number of matches must be 1 or more.".into());
}
words::Pattern::new(pattern).map(|_| ())
}
Condition::Header {
name,
contains,
matches,
} => {
header_name(name)?;
if let Some(pattern) = matches {
words::Pattern::new(pattern)?;
}
if contains.is_some() && matches.is_some() {
return Err("A header condition is either contains or matches.".into());
}
Ok(())
}
Condition::AttachmentType { types } => nonempty_list(types, "type"),
Condition::AttachmentExtension { extensions } => nonempty_list(extensions, "extension"),
Condition::AttachmentName { pattern } => words::Pattern::new(pattern).map(|_| ()),
Condition::Detected { detectors } => {
nonempty_list(detectors, "detector")?;
for d in detectors {
if detectors::by_id(&d.id).is_none() {
return Err(format!("There is no detector \"{}\".", d.id));
}
if d.at_least == 0 {
return Err("A detector's least count must be 1 or more.".into());
}
}
Ok(())
}
Condition::RecipientOutside
| Condition::AttachmentSizeOver { .. }
| Condition::AttachmentCountOver { .. }
| Condition::CantBeInspected
| Condition::MessageSizeOver { .. } => Ok(()),
}
}
fn validate_action(action: &Action) -> Result<(), String> {
match action {
Action::AddDisclaimer { text: t, html, .. } => {
text(t, "disclaimer")?;
html.as_deref()
.map_or(Ok(()), |h| text(h, "disclaimer's HTML"))
}
Action::AddHeader { name, value } => {
header_name(name)?;
if value.len() > 998 || value.contains(['\r', '\n']) {
Err("A header value is one line of at most 998 characters.".into())
} else {
Ok(())
}
}
Action::RemoveHeader { name } => header_name(name),
Action::PrefixSubject { text: t } => text(t, "subject prefix"),
Action::AddRecipient { address } => {
if address.contains('@') {
Ok(())
} else {
Err(format!("\"{address}\" isn't an address."))
}
}
Action::Redirect { addresses } => {
nonempty_list(addresses, "address")?;
match addresses.iter().find(|a| !a.contains('@')) {
Some(a) => Err(format!("\"{a}\" isn't an address.")),
None => Ok(()),
}
}
Action::Refuse { text: t } => text(t, "refusal text"),
Action::Route { queue } => text(queue, "queue"),
Action::Block { notice } | Action::Warn { notice } | Action::Hold { notice, .. } => {
text(notice, "notice")
}
}
}
// --- Storage --------------------------------------------------------------
struct Json<T>(T);
impl<T: SerdeSerialize> Serialize for Json<T> {
fn serialize(&self) -> trc::Result<Vec<u8>> {
serde_json::to_vec(&self.0).map_err(|err| {
trc::StoreEvent::UnexpectedError
.into_err()
.details("Failed to serialize mail rule")
.reason(err)
})
}
}
impl<T: DeserializeOwned + Sync + Send> Deserialize for Json<T> {
fn deserialize(bytes: &[u8]) -> trc::Result<Self> {
serde_json::from_slice(bytes).map(Json).map_err(|err| {
trc::StoreEvent::DataCorruption
.into_err()
.details("Invalid mail rule")
.reason(err)
})
}
}
fn class(id: u32) -> ValueClass {
let mut key = Vec::with_capacity(6);
key.push(FEATURE);
key.push(KIND_RULE);
key.extend_from_slice(&id.to_be_bytes());
ValueClass::Any(AnyClass {
subspace: SUBSPACE_INBUXA,
key,
})
}
fn key(id: u32) -> ValueKey<ValueClass> {
ValueKey::from(class(id))
}
pub async fn get(data: &Store, id: u32) -> trc::Result<Option<Rule>> {
Ok(data
.get_value::<Json<Rule>>(key(id))
.await
.caused_by(trc::location!())?
.map(|Json(rule)| rule))
}
/// Every rule, in the order they run: by priority, then oldest first.
pub async fn all(data: &Store) -> trc::Result<Vec<Rule>> {
let mut rules = Vec::new();
data.iterate(IterateParams::new(key(0), key(u32::MAX)), |_, value| {
if let Ok(Json(rule)) = Json::<Rule>::deserialize(value) {
rules.push(rule);
}
Ok(true)
})
.await
.caused_by(trc::location!())?;
rules.sort_by_key(|rule| (rule.priority, rule.id));
Ok(rules)
}
/// Writes a new rule under the next free id, which it returns. Two nodes
/// creating rules at once can't take the same id: the key must be absent.
pub async fn create(data: &Store, rule: &Rule) -> trc::Result<u32> {
let mut attempt = 0;
loop {
attempt += 1;
let id = all(data).await?.iter().map(|r| r.id).max().unwrap_or(0) + 1;
let stored = Rule { id, ..rule.clone() };
let mut batch = BatchBuilder::new();
batch.assert_value(class(id), AssertValue::None);
batch.set(class(id), Json(&stored).serialize()?);
match data.write(batch.build_all()).await {
Ok(_) => {
super::cache::invalidate();
return Ok(id);
}
Err(err)
if attempt < CREATE_ATTEMPTS
&& matches!(
err.as_ref(),
trc::EventType::Store(trc::StoreEvent::AssertValueFailed)
) => {}
Err(err) => return Err(err.caused_by(trc::location!())),
}
}
}
/// Replaces a stored rule (same id).
pub async fn update(data: &Store, rule: &Rule) -> trc::Result<()> {
let mut batch = BatchBuilder::new();
batch.set(class(rule.id), Json(rule).serialize()?);
data.write(batch.build_all())
.await
.caused_by(trc::location!())?;
super::cache::invalidate();
Ok(())
}
pub async fn delete(data: &Store, id: u32) -> trc::Result<()> {
let mut batch = BatchBuilder::new();
batch.clear(class(id));
data.write(batch.build_all())
.await
.caused_by(trc::location!())?;
super::cache::invalidate();
Ok(())
}
#[cfg(test)]
mod tests {
use super::*;
fn rule(kind: Kind, actions: Vec<Action>) -> Rule {
Rule {
id: 0,
name: "Cards outside".into(),
description: String::new(),
kind,
enabled: true,
priority: 0,
direction: Direction::Outgoing,
conditions: vec![Condition::RecipientOutside],
exceptions: vec![],
actions,
stop_processing: false,
created_by: String::new(),
created_at: 0,
updated_at: 0,
}
}
#[test]
fn wire_format() {
let json = r#"{"name":"Cards","kind":"dlp","direction":"outgoing",
"conditions":[{"type":"recipientOutside"},{"type":"detected","detectors":[{"id":"payment-card","atLeast":5}]}],
"actions":[{"type":"hold","notice":"Held for review","notifySender":true}]}"#;
let parsed: Rule = serde_json::from_str(json).unwrap();
assert!(parsed.enabled);
assert_eq!(
parsed.conditions[1],
Condition::Detected {
detectors: vec![DetectorMin {
id: "payment-card".into(),
at_least: 5
}]
}
);
assert_eq!(
parsed.actions[0],
Action::Hold {
notice: "Held for review".into(),
notify_sender: true
}
);
assert!(parsed.validate().is_ok());
let back = serde_json::to_value(&parsed).unwrap();
assert_eq!(back["actions"][0]["notifySender"], true);
}
#[test]
fn dlp_rules_have_one_dlp_action_on_outgoing_mail() {
let block = Action::Block {
notice: "No.".into(),
};
assert!(rule(Kind::Dlp, vec![block.clone()]).validate().is_ok());
let two = rule(
Kind::Dlp,
vec![
block.clone(),
Action::Warn {
notice: "Hm.".into(),
},
],
);
assert_eq!(two.validate().unwrap_err().property, "actions");
let mixed = rule(
Kind::Dlp,
vec![block.clone(), Action::PrefixSubject { text: "[x]".into() }],
);
assert_eq!(mixed.validate().unwrap_err().property, "actions");
let mut inbound = rule(Kind::Dlp, vec![block.clone()]);
inbound.direction = Direction::Incoming;
assert_eq!(inbound.validate().unwrap_err().property, "direction");
assert_eq!(
rule(Kind::Transport, vec![block])
.validate()
.unwrap_err()
.property,
"actions"
);
}
#[test]
fn conditions_and_actions_are_checked() {
let disclaimer = Action::AddDisclaimer {
text: "Sent from Example Co.".into(),
html: None,
position: Position::Bottom,
};
let mut r = rule(Kind::Transport, vec![disclaimer]);
assert!(r.validate().is_ok());
r.conditions.push(Condition::Detected {
detectors: vec![DetectorMin {
id: "iban".into(),
at_least: 1,
}],
});
assert_eq!(r.validate().unwrap_err().property, "conditions");
let mut r = rule(
Kind::Dlp,
vec![Action::Block {
notice: "No.".into(),
}],
);
r.conditions = vec![Condition::Detected {
detectors: vec![DetectorMin {
id: "nope".into(),
at_least: 1,
}],
}];
assert!(r.validate().unwrap_err().reason.contains("nope"));
r.conditions = vec![Condition::Pattern {
pattern: "(".into(),
at_least: 1,
}];
assert!(r.validate().is_err());
r.conditions = vec![Condition::Words {
words: vec![],
at_least: 1,
}];
assert!(r.validate().is_err());
r.exceptions = vec![Condition::Header {
name: "X-Bad: yes".into(),
contains: None,
matches: None,
}];
r.conditions = vec![];
assert_eq!(r.validate().unwrap_err().property, "exceptions");
let header = rule(
Kind::Transport,
vec![Action::AddHeader {
name: "X-Tag".into(),
value: "a\r\nBcc: x@y".into(),
}],
);
assert!(header.validate().is_err());
let redirect = rule(
Kind::Transport,
vec![Action::Redirect {
addresses: vec!["nobody".into()],
}],
);
assert!(redirect.validate().is_err());
let mut unnamed = rule(
Kind::Transport,
vec![Action::RemoveHeader {
name: "X-Tag".into(),
}],
);
unnamed.name = " ".into();
assert_eq!(unnamed.validate().unwrap_err().property, "name");
}
}
-109
View File
@@ -1,109 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! An organization's own word lists and patterns (§2.3). Both count
//! occurrences, not distinct values: "confidential" three times is three.
use aho_corasick::{AhoCorasick, AhoCorasickBuilder, MatchKind};
use regex::{Regex, RegexBuilder};
/// How large a compiled pattern may grow. Keeps a rule someone writes from
/// making every message slow to send.
const PATTERN_SIZE_LIMIT: usize = 1 << 20;
/// Words and phrases, matched whole and ignoring case.
#[derive(Debug, Clone)]
pub struct WordList {
matcher: AhoCorasick,
}
impl WordList {
/// Builds a list from words or phrases; empty entries are skipped.
pub fn new<I, S>(words: I) -> Result<Self, String>
where
I: IntoIterator<Item = S>,
S: AsRef<str>,
{
let words: Vec<String> = words
.into_iter()
.map(|w| w.as_ref().trim().to_lowercase())
.filter(|w| !w.is_empty())
.collect();
if words.is_empty() {
return Err("The list has no words".into());
}
AhoCorasickBuilder::new()
.match_kind(MatchKind::LeftmostLongest)
.build(&words)
.map(|matcher| Self { matcher })
.map_err(|err| err.to_string())
}
/// How many times any word of the list appears in `text`.
pub fn count(&self, text: &str) -> usize {
let text = text.to_lowercase();
self.matcher
.find_iter(&text)
.filter(|m| super::detectors::stands_alone(&text, m.start(), m.end()))
.count()
}
}
/// An organization's regular expression.
#[derive(Debug, Clone)]
pub struct Pattern {
regex: Regex,
}
impl Pattern {
/// Compiles `pattern`, or says why it can't be used. Matching ignores
/// case unless the pattern turns that off with `(?-i)`.
pub fn new(pattern: &str) -> Result<Self, String> {
RegexBuilder::new(pattern)
.case_insensitive(true)
.size_limit(PATTERN_SIZE_LIMIT)
.build()
.map(|regex| Self { regex })
.map_err(|err| err.to_string())
}
/// How many times the pattern matches in `text`.
pub fn count(&self, text: &str) -> usize {
self.regex.find_iter(text).filter(|m| !m.is_empty()).count()
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn words_whole_and_any_case() {
let list = WordList::new(["Project Falcon", "confidential", " "]).unwrap();
assert_eq!(
list.count(
"CONFIDENTIAL: project falcon notes. Not confidentiality, not projectfalcon."
),
2
);
assert_eq!(list.count("Confidential, confidential and confidential"), 3);
// Non-ASCII case folding
let list = WordList::new(["GEHEIM", "Straße"]).unwrap();
assert_eq!(list.count("streng geheim, STRASSE ist nicht Straße"), 2);
assert!(WordList::new(["", " "]).is_err());
}
#[test]
fn patterns() {
let pattern = Pattern::new(r"\bPRJ-\d{4}\b").unwrap();
assert_eq!(pattern.count("prj-1234 and PRJ-5678, not PRJ-12"), 2);
assert!(Pattern::new("(unclosed").is_err());
// Too large to compile within the limit
assert!(Pattern::new(r"\w{1000}\w{1000}\w{1000}").is_err());
// Empty matches don't count
assert_eq!(Pattern::new("x*").unwrap().count("abc"), 0);
}
}
@@ -1,218 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! `inbuxa:MailRule/get` and `/set` under `urn:inbuxa:jmap`: mail flow rules
//! and DLP rules (dlp-and-mail-flow-rules spec, §2.2). `kind` says which,
//! and which permissions reach it. The set call's `reason` argument, if
//! given, goes into the audit log with the change.
use crate::{
object::{AnyId, JmapObject, JmapObjectId},
request::deserialize::DeserializeArguments,
};
use jmap_tools::{Element, Key, Property};
use std::{borrow::Cow, str::FromStr};
use types::id::Id;
#[derive(Debug, Clone, Default)]
pub struct MailRule;
#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash)]
pub enum MailRuleProperty {
Id,
Name,
Description,
/// `dlp` or `transport`.
Kind,
Enabled,
/// Lower runs first.
Priority,
/// `outgoing`, `incoming` or `any`.
Direction,
Conditions,
Exceptions,
Actions,
StopProcessing,
CreatedBy,
CreatedAt,
UpdatedAt,
}
#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash)]
pub enum MailRuleValue {
Id(Id),
}
impl Property for MailRuleProperty {
fn try_parse(parent: Option<&Key<'_, Self>>, value: &str) -> Option<Self> {
// Keys inside conditions and actions stay plain keys
match parent {
None => MailRuleProperty::parse(value),
Some(_) => None,
}
}
fn to_cow(&self) -> Cow<'static, str> {
match self {
MailRuleProperty::Id => "id",
MailRuleProperty::Name => "name",
MailRuleProperty::Description => "description",
MailRuleProperty::Kind => "kind",
MailRuleProperty::Enabled => "enabled",
MailRuleProperty::Priority => "priority",
MailRuleProperty::Direction => "direction",
MailRuleProperty::Conditions => "conditions",
MailRuleProperty::Exceptions => "exceptions",
MailRuleProperty::Actions => "actions",
MailRuleProperty::StopProcessing => "stopProcessing",
MailRuleProperty::CreatedBy => "createdBy",
MailRuleProperty::CreatedAt => "createdAt",
MailRuleProperty::UpdatedAt => "updatedAt",
}
.into()
}
}
impl MailRuleProperty {
fn parse(value: &str) -> Option<Self> {
hashify::tiny_map!(value.as_bytes(),
b"id" => MailRuleProperty::Id,
b"name" => MailRuleProperty::Name,
b"description" => MailRuleProperty::Description,
b"kind" => MailRuleProperty::Kind,
b"enabled" => MailRuleProperty::Enabled,
b"priority" => MailRuleProperty::Priority,
b"direction" => MailRuleProperty::Direction,
b"conditions" => MailRuleProperty::Conditions,
b"exceptions" => MailRuleProperty::Exceptions,
b"actions" => MailRuleProperty::Actions,
b"stopProcessing" => MailRuleProperty::StopProcessing,
b"createdBy" => MailRuleProperty::CreatedBy,
b"createdAt" => MailRuleProperty::CreatedAt,
b"updatedAt" => MailRuleProperty::UpdatedAt,
)
}
}
impl FromStr for MailRuleProperty {
type Err = ();
fn from_str(s: &str) -> Result<Self, Self::Err> {
MailRuleProperty::parse(s).ok_or(())
}
}
impl Element for MailRuleValue {
type Property = MailRuleProperty;
fn try_parse<P>(key: &Key<'_, Self::Property>, value: &str) -> Option<Self> {
match key {
Key::Property(MailRuleProperty::Id) => Id::from_str(value).ok().map(MailRuleValue::Id),
_ => None,
}
}
fn to_cow(&self) -> Cow<'static, str> {
match self {
MailRuleValue::Id(id) => id.to_string().into(),
}
}
}
/// The set call's own argument: why, for the audit log.
#[derive(Debug, Clone, Default)]
pub struct MailRuleSetArguments {
pub reason: Option<String>,
}
impl<'de> DeserializeArguments<'de> for MailRuleSetArguments {
fn deserialize_argument<A>(&mut self, key: &str, map: &mut A) -> Result<(), A::Error>
where
A: serde::de::MapAccess<'de>,
{
if key == "reason" {
self.reason = map.next_value()?;
} else {
let _ = map.next_value::<serde::de::IgnoredAny>()?;
}
Ok(())
}
}
impl JmapObject for MailRule {
type Property = MailRuleProperty;
type Element = MailRuleValue;
type Id = Id;
type Filter = ();
type Comparator = ();
type GetArguments = ();
type SetArguments<'de> = MailRuleSetArguments;
type QueryArguments = ();
type CopyArguments = ();
type ParseArguments = ();
const ID_PROPERTY: Self::Property = MailRuleProperty::Id;
}
impl From<Id> for MailRuleValue {
fn from(id: Id) -> Self {
MailRuleValue::Id(id)
}
}
impl JmapObjectId for MailRuleValue {
fn as_id(&self) -> Option<Id> {
match self {
MailRuleValue::Id(id) => Some(*id),
}
}
fn as_any_id(&self) -> Option<AnyId> {
match self {
MailRuleValue::Id(id) => Some(AnyId::Id(*id)),
}
}
fn as_id_ref(&self) -> Option<&str> {
None
}
fn try_set_id(&mut self, new_id: AnyId) -> bool {
if let AnyId::Id(id) = new_id {
*self = MailRuleValue::Id(id);
true
} else {
false
}
}
}
impl JmapObjectId for MailRuleProperty {
fn as_id(&self) -> Option<Id> {
None
}
fn as_any_id(&self) -> Option<AnyId> {
None
}
fn as_id_ref(&self) -> Option<&str> {
None
}
fn try_set_id(&mut self, _: AnyId) -> bool {
false
}
}
-1
View File
@@ -28,7 +28,6 @@ pub mod inbuxa_data_inventory; // inbuxa: personal-data catalog
pub mod inbuxa_inventory_snapshot; // inbuxa: personal-data catalog
pub mod inbuxa_audit; // inbuxa: the audit log
pub mod inbuxa_legal_hold; // inbuxa: legal hold
pub mod inbuxa_mail_rule; // inbuxa: DLP and mail flow rules
pub mod inbuxa_hold_export; // inbuxa: legal hold exports
pub mod inbuxa_explanation; // inbuxa: "Explain this" with the local model
pub mod inbuxa_protocol_policy; // inbuxa: legacy protocols off
-3
View File
@@ -82,9 +82,6 @@ impl Response<'_> {
GetResponseMethod::LegalHold(response) => {
response.eval_jptr(path, &mut results)
}
GetResponseMethod::MailRule(response) => {
response.eval_jptr(path, &mut results)
}
GetResponseMethod::HoldExport(response) => {
response.eval_jptr(path, &mut results)
}
@@ -53,7 +53,6 @@ impl Response<'_> {
GetRequestMethod::AuditSettings(request) => request.resolve_references(self)?,
GetRequestMethod::AccountLock(request) => request.resolve_references(self)?,
GetRequestMethod::LegalHold(request) => request.resolve_references(self)?,
GetRequestMethod::MailRule(request) => request.resolve_references(self)?,
GetRequestMethod::HoldExport(request) => request.resolve_references(self)?,
GetRequestMethod::ProtocolPolicy(request) => request.resolve_references(self)?,
GetRequestMethod::TenantProtocolPolicy(request) => {
@@ -123,9 +122,6 @@ impl Response<'_> {
SetRequestMethod::LegalHold(request) => {
request.resolve_references(self, 1, false)?
}
SetRequestMethod::MailRule(request) => {
request.resolve_references(self, 1, false)?
}
SetRequestMethod::HoldExport(request) => {
request.resolve_references(self, 1, false)?
}
+1 -9
View File
@@ -65,8 +65,6 @@ pub enum MethodObject {
LegalHold,
HoldExport,
ProtocolPolicy,
// inbuxa: DLP and mail flow rules
MailRule,
TenantProtocolPolicy,
}
@@ -104,8 +102,7 @@ impl MethodObject {
| MethodObject::AuditVerification
| MethodObject::AccountLock
| MethodObject::LegalHold
| MethodObject::HoldExport
| MethodObject::MailRule => Capability::Inbuxa,
| MethodObject::HoldExport => Capability::Inbuxa,
MethodObject::ProtocolPolicy => Capability::Inbuxa,
MethodObject::TenantProtocolPolicy => Capability::Inbuxa,
}
@@ -299,8 +296,6 @@ impl MethodName {
(MethodFunction::Set, MethodObject::AccountLock) => "inbuxa:AccountLock/set",
(MethodFunction::Get, MethodObject::LegalHold) => "inbuxa:LegalHold/get",
(MethodFunction::Set, MethodObject::LegalHold) => "inbuxa:LegalHold/set",
(MethodFunction::Get, MethodObject::MailRule) => "inbuxa:MailRule/get",
(MethodFunction::Set, MethodObject::MailRule) => "inbuxa:MailRule/set",
(MethodFunction::Get, MethodObject::HoldExport) => "inbuxa:HoldExport/get",
(MethodFunction::Set, MethodObject::HoldExport) => "inbuxa:HoldExport/set",
(MethodFunction::Set, MethodObject::AuditVerification) => {
@@ -453,8 +448,6 @@ impl MethodName {
"inbuxa:AccountLock/set" => (MethodObject::AccountLock, MethodFunction::Set),
"inbuxa:LegalHold/get" => (MethodObject::LegalHold, MethodFunction::Get),
"inbuxa:LegalHold/set" => (MethodObject::LegalHold, MethodFunction::Set),
"inbuxa:MailRule/get" => (MethodObject::MailRule, MethodFunction::Get),
"inbuxa:MailRule/set" => (MethodObject::MailRule, MethodFunction::Set),
"inbuxa:HoldExport/get" => (MethodObject::HoldExport, MethodFunction::Get),
"inbuxa:HoldExport/set" => (MethodObject::HoldExport, MethodFunction::Set),
"inbuxa:AuditVerification/set" => (MethodObject::AuditVerification, MethodFunction::Set),
@@ -525,7 +518,6 @@ impl Display for MethodObject {
MethodObject::AuditVerification => "inbuxa:AuditVerification",
MethodObject::AccountLock => "inbuxa:AccountLock",
MethodObject::LegalHold => "inbuxa:LegalHold",
MethodObject::MailRule => "inbuxa:MailRule",
MethodObject::HoldExport => "inbuxa:HoldExport",
MethodObject::ProtocolPolicy => "inbuxa:ProtocolPolicy",
MethodObject::TenantProtocolPolicy => "inbuxa:TenantProtocolPolicy",
-2
View File
@@ -123,7 +123,6 @@ pub enum GetRequestMethod {
AuditSettings(Box<GetRequest<crate::object::inbuxa_audit::AuditSettings>>),
AccountLock(Box<GetRequest<crate::object::inbuxa_account_lock::AccountLock>>),
LegalHold(Box<GetRequest<crate::object::inbuxa_legal_hold::LegalHold>>),
MailRule(Box<GetRequest<crate::object::inbuxa_mail_rule::MailRule>>),
HoldExport(Box<GetRequest<crate::object::inbuxa_hold_export::HoldExport>>),
ProtocolPolicy(Box<GetRequest<crate::object::inbuxa_protocol_policy::ProtocolPolicy>>),
TenantProtocolPolicy(
@@ -159,7 +158,6 @@ pub enum SetRequestMethod<'x> {
AuditVerification(Box<SetRequest<'x, crate::object::inbuxa_audit::AuditVerification>>),
AccountLock(Box<SetRequest<'x, crate::object::inbuxa_account_lock::AccountLock>>),
LegalHold(Box<SetRequest<'x, crate::object::inbuxa_legal_hold::LegalHold>>),
MailRule(Box<SetRequest<'x, crate::object::inbuxa_mail_rule::MailRule>>),
HoldExport(Box<SetRequest<'x, crate::object::inbuxa_hold_export::HoldExport>>),
ProtocolPolicy(Box<SetRequest<'x, crate::object::inbuxa_protocol_policy::ProtocolPolicy>>),
TenantProtocolPolicy(
-15
View File
@@ -609,21 +609,6 @@ impl<'de> Visitor<'de> for CallVisitor {
return Err(de::Error::invalid_length(1, &self));
}
},
// inbuxa: DLP and mail flow rules
(MethodFunction::Get, MethodObject::MailRule) => match seq.next_element() {
Ok(Some(value)) => RequestMethod::Get(GetRequestMethod::MailRule(value)),
Err(err) => RequestMethod::invalid(err),
Ok(None) => {
return Err(de::Error::invalid_length(1, &self));
}
},
(MethodFunction::Set, MethodObject::MailRule) => match seq.next_element() {
Ok(Some(value)) => RequestMethod::Set(SetRequestMethod::MailRule(value)),
Err(err) => RequestMethod::invalid(err),
Ok(None) => {
return Err(de::Error::invalid_length(1, &self));
}
},
// inbuxa: legal hold
(MethodFunction::Get, MethodObject::LegalHold) => match seq.next_element() {
Ok(Some(value)) => RequestMethod::Get(GetRequestMethod::LegalHold(value)),
-14
View File
@@ -110,7 +110,6 @@ pub enum GetResponseMethod {
AuditSettings(GetResponse<crate::object::inbuxa_audit::AuditSettings>),
AccountLock(GetResponse<crate::object::inbuxa_account_lock::AccountLock>),
LegalHold(GetResponse<crate::object::inbuxa_legal_hold::LegalHold>),
MailRule(GetResponse<crate::object::inbuxa_mail_rule::MailRule>),
HoldExport(GetResponse<crate::object::inbuxa_hold_export::HoldExport>),
ProtocolPolicy(GetResponse<crate::object::inbuxa_protocol_policy::ProtocolPolicy>),
TenantProtocolPolicy(
@@ -146,7 +145,6 @@ pub enum SetResponseMethod {
AuditVerification(Box<SetResponse<crate::object::inbuxa_audit::AuditVerification>>),
AccountLock(Box<SetResponse<crate::object::inbuxa_account_lock::AccountLock>>),
LegalHold(Box<SetResponse<crate::object::inbuxa_legal_hold::LegalHold>>),
MailRule(Box<SetResponse<crate::object::inbuxa_mail_rule::MailRule>>),
HoldExport(Box<SetResponse<crate::object::inbuxa_hold_export::HoldExport>>),
Explanation(Box<SetResponse<crate::object::inbuxa_explanation::Explanation>>),
ProtocolPolicy(Box<SetResponse<crate::object::inbuxa_protocol_policy::ProtocolPolicy>>),
@@ -801,18 +799,6 @@ impl<'x> From<SetResponse<crate::object::inbuxa_account_lock::AccountLock>> for
}
// inbuxa: legal hold
impl<'x> From<GetResponse<crate::object::inbuxa_mail_rule::MailRule>> for ResponseMethod<'x> {
fn from(value: GetResponse<crate::object::inbuxa_mail_rule::MailRule>) -> Self {
ResponseMethod::Get(GetResponseMethod::MailRule(value))
}
}
impl<'x> From<SetResponse<crate::object::inbuxa_mail_rule::MailRule>> for ResponseMethod<'x> {
fn from(value: SetResponse<crate::object::inbuxa_mail_rule::MailRule>) -> Self {
ResponseMethod::Set(SetResponseMethod::MailRule(Box::new(value)))
}
}
impl<'x> From<GetResponse<crate::object::inbuxa_legal_hold::LegalHold>> for ResponseMethod<'x> {
fn from(value: GetResponse<crate::object::inbuxa_legal_hold::LegalHold>) -> Self {
ResponseMethod::Get(GetResponseMethod::LegalHold(value))
-24
View File
@@ -103,16 +103,6 @@ impl JmapAuthorization for AccessToken {
// inbuxa: account lock (AL-12)
GetRequestMethod::AccountLock(_) => Permission::SysAccountLockGet,
GetRequestMethod::LegalHold(_) => Permission::SysLegalHoldGet,
// inbuxa: DLP and mail flow rules share an object; either
// permission reaches it, and the handler shows each kind
// only to those who may see it
GetRequestMethod::MailRule(_) => {
if self.has_permission(Permission::SysMailRuleGet) {
Permission::SysMailRuleGet
} else {
Permission::SysDlpPolicyGet
}
}
GetRequestMethod::HoldExport(_) => Permission::SysLegalHoldExport,
// inbuxa: legacy protocols off. It takes listeners away and
// puts them back, so it takes the listener's permissions
@@ -257,19 +247,6 @@ impl JmapAuthorization for AccessToken {
Permission::SysLegalHoldUpdate,
Permission::SysLegalHoldUpdate,
),
// inbuxa: DLP and mail flow rules: either change
// permission gets in; the handler checks each rule's kind
SetRequestMethod::MailRule(_) => {
if self.has_permission(Permission::SysMailRuleUpdate)
|| self.has_permission(Permission::SysDlpPolicyUpdate)
{
Ok(())
} else {
Err(trc::JmapEvent::Forbidden
.into_err()
.details("You are not authorized to change mail rules"))
}
}
// inbuxa: LH-12, exporting held data
SetRequestMethod::HoldExport(s) => validate_set(
s,
@@ -430,7 +407,6 @@ impl JmapAuthorization for AccessToken {
| MethodObject::AccountLock
| MethodObject::LegalHold
| MethodObject::HoldExport
| MethodObject::MailRule
| MethodObject::ProtocolPolicy
| MethodObject::TenantProtocolPolicy => Permission::JmapEmailChanges,
// inbuxa: x:MaskedEmail/changes reads what /get reads
-24
View File
@@ -279,9 +279,6 @@ impl RequestHandler for Server {
SetResponseMethod::LegalHold(set_response) => {
set_response.update_created_ids(&mut response);
}
SetResponseMethod::MailRule(set_response) => {
set_response.update_created_ids(&mut response);
}
SetResponseMethod::HoldExport(set_response) => {
set_response.update_created_ids(&mut response);
}
@@ -489,11 +486,6 @@ impl RequestHandler for Server {
resolve_account_id(&mut req.account_id, method_name.obj, access_token)?;
crate::inbuxa::legal_hold::get(self, *req).await?.into()
}
// inbuxa: DLP and mail flow rules
GetRequestMethod::MailRule(mut req) => {
resolve_account_id(&mut req.account_id, method_name.obj, access_token)?;
crate::inbuxa::mail_rule::get(self, access_token, *req).await?.into()
}
// inbuxa: the audit log (AU-9)
GetRequestMethod::AuditEvent(mut req) => {
resolve_account_id(&mut req.account_id, method_name.obj, access_token)?;
@@ -918,22 +910,6 @@ impl RequestHandler for Server {
.await?
.into()
}
SetRequestMethod::MailRule(mut req) => {
resolve_account_id(&mut req.account_id, method_name.obj, access_token)?;
let reason = req.arguments.reason.clone();
crate::inbuxa::audit::recorded(
self,
access_token,
session,
&method_name.obj.to_string(),
None,
reason,
*req,
|req| Box::pin(crate::inbuxa::mail_rule::set(self, access_token, req)),
)
.await?
.into()
}
SetRequestMethod::AuditExport(mut req) => {
resolve_account_id(&mut req.account_id, method_name.obj, access_token)?;
crate::inbuxa::audit_log::export_set(self, access_token, session, *req)
-1
View File
@@ -429,7 +429,6 @@ impl IntermediateChangesResponse {
| MethodObject::AccountLock
| MethodObject::LegalHold
| MethodObject::HoldExport
| MethodObject::MailRule
| MethodObject::ProtocolPolicy
| MethodObject::TenantProtocolPolicy
| MethodObject::Registry(_) => unreachable!(),
-327
View File
@@ -1,327 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! `inbuxa:MailRule` (dlp-and-mail-flow-rules spec, §2.2, §2.8): mail flow
//! rules and DLP rules. One object, two kinds, each with its own
//! permissions: `sysMailRuleGet`/`Update` for transport rules,
//! `sysDlpPolicyGet`/`Update` for DLP rules. Rules are the server's: nobody
//! in a tenant reaches them (settled answer 3). The request layer records
//! every change in the audit log.
use common::{Server, auth::AccessToken};
use inbuxa_features::mailflow::rules::{self, Kind, Rule};
use jmap_proto::{
error::set::SetError,
method::{
get::{GetRequest, GetResponse},
set::{SetRequest, SetResponse},
},
object::inbuxa_mail_rule::{MailRule, MailRuleProperty as P, MailRuleValue},
request::IntoValid,
types::date::UTCDate,
};
use jmap_tools::{Key, Map, Property, Value};
use registry::schema::enums::Permission;
use std::borrow::Cow;
use store::write::now;
use types::id::Id;
type RValue = Value<'static, P, MailRuleValue>;
const ALL: &[P] = &[
P::Id,
P::Name,
P::Description,
P::Kind,
P::Enabled,
P::Priority,
P::Direction,
P::Conditions,
P::Exceptions,
P::Actions,
P::StopProcessing,
P::CreatedBy,
P::CreatedAt,
P::UpdatedAt,
];
/// Properties the server sets; a client that sends them is refused.
const SERVER_SET: &[P] = &[P::Id, P::CreatedBy, P::CreatedAt, P::UpdatedAt];
fn can_see(access_token: &AccessToken, kind: Kind) -> bool {
access_token.has_permission(match kind {
Kind::Dlp => Permission::SysDlpPolicyGet,
Kind::Transport => Permission::SysMailRuleGet,
})
}
fn can_change(access_token: &AccessToken, kind: Kind) -> bool {
access_token.has_permission(match kind {
Kind::Dlp => Permission::SysDlpPolicyUpdate,
Kind::Transport => Permission::SysMailRuleUpdate,
})
}
fn server_level(access_token: &AccessToken) -> trc::Result<()> {
if access_token.tenant_id().is_some() {
Err(trc::JmapEvent::Forbidden
.into_err()
.details("Mail rules are the server's."))
} else {
Ok(())
}
}
fn json_to_value(json: serde_json::Value) -> RValue {
match json {
serde_json::Value::Null => Value::Null,
serde_json::Value::Bool(b) => Value::Bool(b),
serde_json::Value::Number(n) => {
if let Some(n) = n.as_u64() {
Value::Number(n.into())
} else if let Some(n) = n.as_i64() {
Value::Number(n.into())
} else {
Value::Number(n.as_f64().unwrap_or_default().into())
}
}
serde_json::Value::String(s) => Value::Str(Cow::Owned(s)),
serde_json::Value::Array(items) => {
Value::Array(items.into_iter().map(json_to_value).collect())
}
serde_json::Value::Object(map) => {
let mut out = Map::with_capacity(map.len());
for (key, value) in map {
out.insert_unchecked(Key::Owned(key), json_to_value(value));
}
Value::Object(out)
}
}
}
fn date(seconds: u64) -> RValue {
Value::Str(UTCDate::from_timestamp(seconds as i64).to_string().into())
}
fn to_value(rule: &Rule, properties: &[P]) -> RValue {
let json = serde_json::to_value(rule).unwrap_or_default();
let mut out = Map::with_capacity(properties.len());
for property in properties {
let value = match property {
P::Id => Value::Element(MailRuleValue::Id(Id::from(rule.id))),
P::CreatedAt => date(rule.created_at),
P::UpdatedAt => date(rule.updated_at),
other => json
.get(other.to_cow().as_ref())
.cloned()
.map_or(Value::Null, json_to_value),
};
out.insert_unchecked(Key::Property(property.clone()), value);
}
Value::Object(out)
}
/// A rule as sent: its JSON object, top-level keys only those a client may
/// set.
fn client_json(value: Value<'_, P, MailRuleValue>) -> Result<serde_json::Map<String, serde_json::Value>, SetError<P>> {
let mut map = serde_json::Map::new();
for (key, value) in value.into_expanded_object() {
match &key {
Key::Property(p) if SERVER_SET.contains(p) => {
return Err(SetError::invalid_properties()
.with_property(p.clone())
.with_description("The server sets this."));
}
Key::Property(p) => {
map.insert(p.to_cow().into_owned(), value.into());
}
_ => {
return Err(SetError::invalid_properties().with_property(key.clone().into_owned()));
}
}
}
Ok(map)
}
fn parse(json: serde_json::Map<String, serde_json::Value>) -> Result<Rule, SetError<P>> {
let rule: Rule = serde_json::from_value(serde_json::Value::Object(json)).map_err(|err| {
SetError::invalid_properties().with_description(format!("Not a valid rule: {err}"))
})?;
rule.validate().map_err(|invalid| {
let property = invalid.property.parse::<P>().unwrap_or(P::Name);
SetError::invalid_properties()
.with_property(property)
.with_description(invalid.reason)
})?;
Ok(rule)
}
fn forbidden(kind: Kind) -> SetError<P> {
SetError::forbidden().with_description(match kind {
Kind::Dlp => "Changing DLP rules needs the permission to change DLP rules.",
Kind::Transport => "Changing mail flow rules needs the permission to change them.",
})
}
fn rule_id(id: Id) -> Option<u32> {
u32::try_from(id.id()).ok()
}
/// `inbuxa:MailRule/get`: the rules the caller may see, in the order they
/// run.
pub async fn get(
server: &Server,
access_token: &AccessToken,
mut request: GetRequest<MailRule>,
) -> trc::Result<GetResponse<MailRule>> {
server_level(access_token)?;
let properties = request.unwrap_properties(ALL);
let (ids, not_found) = request.unwrap_ids(server.core.jmap.get_max_objects)?;
let mut response = GetResponse {
account_id: request.account_id.into(),
state: None,
list: Vec::new(),
not_found,
};
let visible: Vec<Rule> = rules::all(server.store())
.await?
.into_iter()
.filter(|rule| can_see(access_token, rule.kind))
.collect();
match ids {
None => {
response.list = visible
.iter()
.map(|rule| to_value(rule, &properties))
.collect()
}
Some(ids) => {
for id in ids {
match rule_id(id).and_then(|id| visible.iter().find(|r| r.id == id)) {
Some(rule) => response.list.push(to_value(rule, &properties)),
None => response.push_not_found(id),
}
}
}
}
Ok(response)
}
/// `inbuxa:MailRule/set`: create, change or delete rules, each checked
/// against the permissions for its kind (and, on a change of kind, both).
pub async fn set(
server: &Server,
access_token: &AccessToken,
mut request: SetRequest<'_, MailRule>,
) -> trc::Result<SetResponse<MailRule>> {
server_level(access_token)?;
let mut response = SetResponse::from_request(&request, server.core.jmap.set_max_objects)?;
let data = server.store();
let actor = server.audit_actor(access_token).await;
for (client_id, value) in request.unwrap_create() {
let rule = match client_json(value).and_then(parse) {
Ok(rule) => rule,
Err(error) => {
response.not_created.append(client_id, error);
continue;
}
};
if !can_change(access_token, rule.kind) {
response.not_created.append(client_id, forbidden(rule.kind));
continue;
}
let at = now();
let rule = Rule {
created_by: actor.name.clone(),
created_at: at,
updated_at: at,
..rule
};
let id = rules::create(data, &rule).await?;
let mut out = Map::with_capacity(1);
out.insert_unchecked(
Key::Property(P::Id),
Value::Element(MailRuleValue::Id(Id::from(id))),
);
response.created.insert(client_id, Value::Object(out));
}
for (id, value) in request.unwrap_update().into_valid() {
let Some(current) = (match rule_id(id) {
Some(rule_id) => rules::get(data, rule_id).await?,
None => None,
}) else {
response.not_updated.append(id, SetError::not_found());
continue;
};
if !can_see(access_token, current.kind) {
response.not_updated.append(id, SetError::not_found());
continue;
}
if !can_change(access_token, current.kind) {
response.not_updated.append(id, forbidden(current.kind));
continue;
}
// The stored rule, with each property sent replacing its own
let mut json = match serde_json::to_value(&current) {
Ok(serde_json::Value::Object(map)) => map,
_ => serde_json::Map::new(),
};
let changes = match client_json(value) {
Ok(changes) => changes,
Err(error) => {
response.not_updated.append(id, error);
continue;
}
};
json.extend(changes);
let next = match parse(json) {
Ok(next) => next,
Err(error) => {
response.not_updated.append(id, error);
continue;
}
};
if next.kind != current.kind && !can_change(access_token, next.kind) {
response.not_updated.append(id, forbidden(next.kind));
continue;
}
let next = Rule {
id: current.id,
created_by: current.created_by.clone(),
created_at: current.created_at,
updated_at: now(),
..next
};
if next != current {
rules::update(data, &next).await?;
}
response.updated.append(id, None);
}
for id in request.unwrap_destroy().into_valid() {
let Some(current) = (match rule_id(id) {
Some(rule_id) => rules::get(data, rule_id).await?,
None => None,
}) else {
response.not_destroyed.append(id, SetError::not_found());
continue;
};
if !can_see(access_token, current.kind) {
response.not_destroyed.append(id, SetError::not_found());
continue;
}
if !can_change(access_token, current.kind) {
response.not_destroyed.append(id, forbidden(current.kind));
continue;
}
rules::delete(data, current.id).await?;
response.destroyed.push(id);
}
Ok(response)
}
-1
View File
@@ -10,7 +10,6 @@
pub mod access;
pub mod account_lock;
pub mod legal_hold;
pub mod mail_rule;
pub mod hold_export;
pub mod hold_export_api;
pub mod audit;
-7
View File
@@ -1748,13 +1748,6 @@ pub enum Permission {
SysLegalHoldExport = 672,
// inbuxa: personal-data catalog, the data inventory and compliance overview
SysComplianceGet = 673,
// inbuxa: DLP and mail flow rules
SysMailRuleGet = 674,
SysMailRuleUpdate = 675,
SysDlpPolicyGet = 676,
SysDlpPolicyUpdate = 677,
SysDlpReviewGet = 678,
SysDlpReviewUpdate = 679,
SysAccountGet = 219,
SysAccountCreate = 220,
SysAccountUpdate = 221,
+1 -19
View File
@@ -7091,12 +7091,6 @@ impl EnumImpl for Permission {
b"sysLegalHoldUpdate" => Permission::SysLegalHoldUpdate,
b"sysLegalHoldExport" => Permission::SysLegalHoldExport,
b"sysComplianceGet" => Permission::SysComplianceGet,
b"sysMailRuleGet" => Permission::SysMailRuleGet,
b"sysMailRuleUpdate" => Permission::SysMailRuleUpdate,
b"sysDlpPolicyGet" => Permission::SysDlpPolicyGet,
b"sysDlpPolicyUpdate" => Permission::SysDlpPolicyUpdate,
b"sysDlpReviewGet" => Permission::SysDlpReviewGet,
b"sysDlpReviewUpdate" => Permission::SysDlpReviewUpdate,
b"sysAccountGet" => Permission::SysAccountGet,
b"sysAccountCreate" => Permission::SysAccountCreate,
b"sysAccountUpdate" => Permission::SysAccountUpdate,
@@ -7787,12 +7781,6 @@ impl EnumImpl for Permission {
Permission::SysLegalHoldUpdate => "sysLegalHoldUpdate",
Permission::SysLegalHoldExport => "sysLegalHoldExport",
Permission::SysComplianceGet => "sysComplianceGet",
Permission::SysMailRuleGet => "sysMailRuleGet",
Permission::SysMailRuleUpdate => "sysMailRuleUpdate",
Permission::SysDlpPolicyGet => "sysDlpPolicyGet",
Permission::SysDlpPolicyUpdate => "sysDlpPolicyUpdate",
Permission::SysDlpReviewGet => "sysDlpReviewGet",
Permission::SysDlpReviewUpdate => "sysDlpReviewUpdate",
Permission::SysAccountGet => "sysAccountGet",
Permission::SysAccountCreate => "sysAccountCreate",
Permission::SysAccountUpdate => "sysAccountUpdate",
@@ -8476,12 +8464,6 @@ impl EnumImpl for Permission {
671 => Some(Permission::SysLegalHoldUpdate),
672 => Some(Permission::SysLegalHoldExport),
673 => Some(Permission::SysComplianceGet),
674 => Some(Permission::SysMailRuleGet),
675 => Some(Permission::SysMailRuleUpdate),
676 => Some(Permission::SysDlpPolicyGet),
677 => Some(Permission::SysDlpPolicyUpdate),
678 => Some(Permission::SysDlpReviewGet),
679 => Some(Permission::SysDlpReviewUpdate),
219 => Some(Permission::SysAccountGet),
220 => Some(Permission::SysAccountCreate),
221 => Some(Permission::SysAccountUpdate),
@@ -8926,7 +8908,7 @@ impl EnumImpl for Permission {
}
}
const COUNT: usize = 680;
const COUNT: usize = 674;
}
impl serde::Serialize for Permission {
+1 -1
View File
@@ -81,7 +81,7 @@ fn legacy_setting(name: &str, is_set: impl Fn(&str) -> bool) -> Option<String> {
#[macro_export]
macro_rules! brand_version {
() => {
"2026.9.28.5"
"2026.9.28.6"
};
}
+2 -10
View File
@@ -139,14 +139,6 @@ DLP adds **detectors**. Each counts what it finds, and a rule sets a minimum
either side, in the languages where the identifier is used ("passport",
"Reisepass", "pasaporte"...).
A check that about one random number in ten passes (Luhn, mod 10, mod 11) is
too weak for a bare run of digits: invoice and phone numbers would match. So
a checked identifier that is only digits (SSN, SIN, NHS, TFN, Medicare…)
counts alone in the written form it's issued in (`536-22-1234`,
`130 692 544`, `943 476 5919`), and as bare digits only beside a word. ABA
routing numbers and NPIs are never written with separators, so they always
need a word. (Refinement made while building phase 2, 2026-09-28.)
The catalog (settled answer 6: the recognized, protected identifiers, not a
chosen few). Each row is one table entry and one check function in
`crates/features/src/mailflow/detectors/`:
@@ -165,10 +157,10 @@ chosen few). Each row is one table entry and one check function in
| US | Social Security number | Checked | `AAA-GG-SSSS`, or nine digits with a word; never area 000, 666 or 9xx, group 00, serial 0000 |
| US | ITIN | Checked | 9XX-GG-SSSS with the IRS's group ranges |
| US | EIN | Needs a word | a valid IRS prefix and seven digits |
| US | Bank routing number (ABA) | Needs a word | nine digits, a valid Federal Reserve prefix, the 3-7-1 checksum |
| US | Bank routing number (ABA) | Checked | nine digits, a valid Federal Reserve prefix, the 3-7-1 checksum |
| US | Driver's license | Needs a word | each state's published format |
| US | Medicare Beneficiary Identifier | Checked | CMS's 11-character pattern and excluded letters |
| US | National Provider Identifier | Needs a word | ten digits, Luhn over the `80840` prefix |
| US | National Provider Identifier | Checked | ten digits, Luhn over the `80840` prefix |
| US | DEA registration number | Checked | two letters, seven digits, DEA's check digit |
| UK | National Insurance number | Checked | two letters (HMRC's excluded prefixes), six digits, A–D |
| UK | NHS number | Checked | ten digits, mod 11 |
Binary file not shown.
-15
View File
@@ -79,21 +79,6 @@ lockedAt = ["metadata"]
lockedBy = ["identifier"]
delegates = ["identifier"]
[object."inbuxa:MailRule"]
file = "inbuxa_mail_rule.rs"
default = "none"
whose = ["administrator", "holder", "correspondent"]
where = ["data-store"]
scope = "server"
retention = "unbounded"
[object."inbuxa:MailRule".properties]
name = ["content"]
description = ["content"]
conditions = ["contact", "content"]
exceptions = ["contact", "content"]
actions = ["contact", "content"]
createdBy = ["identifier"]
[object."inbuxa:LegalHold"]
file = "inbuxa_legal_hold.rs"
default = "none"
Binary file not shown.
+1 -1
View File
@@ -1 +1 @@
yF7PlBR3UBqxlabhW5zZ5WacQWsynG1wNYEiJVAl6w4
k496pjVWlQ2p4bkZCh8agzaCKMLD4c3Z9WtoxpckYDU
-201
View File
@@ -1,201 +0,0 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! `inbuxa:MailRule` (dlp-and-mail-flow-rules spec, §2.2, §2.8): rules are
//! stored, listed in the order they run, checked when written, kept apart
//! by kind for permissions, and every change is audited.
use crate::utils::{
account::Account,
server::{TestServer, TestServerBuilder},
};
use registry::schema::{
prelude::{ObjectType, Property},
structs::{CustomRoles, Role, UserRoles},
};
use registry::types::map::Map;
use serde_json::{Value, json};
const USING: &[&str] = &[
"urn:ietf:params:jmap:core",
"urn:inbuxa:jmap",
"urn:inbuxa:jmap:registry",
];
async fn call(account: &Account, method: &str, mut arguments: Value) -> (String, Value) {
if arguments.get("accountId").is_none() {
arguments["accountId"] = account.id_string().into();
}
let response = account.jmap_request(USING, json!([[method, arguments, "0"]])).await;
let call = response
.0
.pointer("/methodResponses/0")
.cloned()
.unwrap_or_else(|| panic!("{method}: {}", response.0));
(call[0].as_str().unwrap_or_default().to_string(), call[1].clone())
}
fn dlp_rule() -> Value {
json!({
"name": "Cards leaving",
"kind": "dlp",
"direction": "outgoing",
"priority": 10,
"conditions": [
{"type": "recipientOutside"},
{"type": "detected", "detectors": [{"id": "payment-card", "atLeast": 5}]}
],
"actions": [{"type": "hold", "notice": "Held for review", "notifySender": true}]
})
}
fn transport_rule() -> Value {
json!({
"name": "Disclaimer",
"kind": "transport",
"direction": "outgoing",
"priority": 1,
"conditions": [{"type": "recipientOutside"}],
"actions": [{"type": "addDisclaimer", "text": "Sent by Example Co.", "position": "bottom"}]
})
}
async fn names(account: &Account) -> Vec<String> {
let (name, response) = call(account, "inbuxa:MailRule/get", json!({"ids": null})).await;
assert_eq!(name, "inbuxa:MailRule/get", "{response}");
response["list"]
.as_array()
.unwrap()
.iter()
.map(|r| r["name"].as_str().unwrap().to_string())
.collect()
}
pub async fn test(test: &mut TestServer) {
println!("Running mail rule tests...");
let admin = test.account("[email protected]");
// Created, then listed in the order they run
let (_, response) = call(
&admin,
"inbuxa:MailRule/set",
json!({"create": {"d": dlp_rule(), "t": transport_rule()}}),
)
.await;
let dlp_id = response["created"]["d"]["id"]
.as_str()
.unwrap_or_else(|| panic!("DLP rule created: {response}"))
.to_string();
let transport_id = response["created"]["t"]["id"].as_str().unwrap().to_string();
assert_eq!(names(&admin).await, vec!["Disclaimer", "Cards leaving"]);
let (_, response) = call(&admin, "inbuxa:MailRule/get", json!({"ids": [dlp_id]})).await;
let rule = &response["list"][0];
assert_eq!(rule["conditions"][1]["detectors"][0]["atLeast"], 5, "{rule}");
assert_eq!(rule["actions"][0]["notifySender"], true);
assert_eq!(rule["createdBy"], "[email protected]");
assert!(rule["createdAt"].as_str().is_some_and(|d| d.ends_with('Z')), "{rule}");
// Checked when written
let mut inbound = dlp_rule();
inbound["direction"] = "incoming".into();
let mut unknown = dlp_rule();
unknown["conditions"][1]["detectors"][0]["id"] = "no-such-detector".into();
let (_, response) = call(
&admin,
"inbuxa:MailRule/set",
json!({"create": {"a": inbound, "b": unknown, "c": {"name": "x", "kind": "dlp"}}}),
)
.await;
assert_eq!(response["notCreated"]["a"]["properties"][0], "direction", "{response}");
assert!(
response["notCreated"]["b"]["description"].as_str().unwrap().contains("no-such-detector"),
"{response}"
);
assert!(response["notCreated"].get("c").is_some(), "{response}");
// Changed in place; what the server sets can't be sent
let (_, response) = call(
&admin,
"inbuxa:MailRule/set",
json!({"update": {
transport_id.as_str(): {"name": "Footer", "priority": 50},
dlp_id.as_str(): {"createdBy": "someone else"}
}}),
)
.await;
assert!(response["updated"].get(transport_id.as_str()).is_some(), "{response}");
assert_eq!(response["notUpdated"][dlp_id.as_str()]["properties"][0], "createdBy", "{response}");
assert_eq!(names(&admin).await, vec!["Cards leaving", "Footer"]);
// A compliance officer sees DLP rules, not mail flow rules, and changes
// neither (settled answer 4)
let mut officer_role = None;
for id in admin
.registry_query_ids(ObjectType::Role, Vec::<(&str, &str)>::new(), Vec::<&str>::new())
.await
{
let role = admin.registry_get::<Role>(id).await;
if role.description == "Compliance Officer" && role.member_tenant_id.is_none() {
officer_role = Some(id);
}
}
let officer = admin
.create_user_account("[email protected]", "officer-secret-7731", "Officer", &[], vec![])
.await;
admin
.registry_update_object(
ObjectType::Account,
officer.id(),
json!({Property::Roles: UserRoles::Custom(CustomRoles {
role_ids: Map::new(vec![officer_role.expect("the officer role")]),
})}),
)
.await;
assert_eq!(names(&officer).await, vec!["Cards leaving"]);
let (name, response) =
call(&officer, "inbuxa:MailRule/set", json!({"create": {"d": dlp_rule()}})).await;
assert_eq!(name, "error", "the officer created a DLP rule: {response}");
// Deleted
let (_, response) = call(
&admin,
"inbuxa:MailRule/set",
json!({"destroy": [transport_id]}),
)
.await;
assert_eq!(response["destroyed"][0], transport_id.as_str(), "{response}");
assert_eq!(names(&admin).await, vec!["Cards leaving"]);
// Every change is in the audit log
let (_, response) = call(
&admin,
"inbuxa:AuditEvent/query",
json!({"filter": {"targetKind": "inbuxa:MailRule"}, "calculateTotal": true}),
)
.await;
assert!(
response["total"].as_u64().unwrap_or(0) >= 4,
"creates, update and destroy audited: {response}"
);
}
#[ignore]
#[tokio::test(flavor = "multi_thread")]
pub async fn mail_rules_tests() {
let mut test = TestServerBuilder::new("mail_rules_tests")
.await
.with_default_listeners()
.await
.build()
.await;
let admin = test.create_admin_account("[email protected]").await;
test.insert_account(admin);
self::test(&mut test).await;
if test.is_reset() {
test.temp_dir.delete();
}
}
-1
View File
@@ -14,7 +14,6 @@ pub mod ai_explain;
pub mod account_lock; // inbuxa: account lock with delegation
pub mod legal_hold; // inbuxa: legal hold
pub mod compliance; // inbuxa: the compliance roles
pub mod mail_rules; // inbuxa: DLP and mail flow rules
pub mod audit; // inbuxa: the audit log
pub mod authorization;
pub mod auto_reload; // inbuxa: registry writes apply at once