Compare commits
13
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e35fc3e6d6 | ||
|
|
5f52dad5f1 | ||
|
|
15064d6fd5 | ||
|
|
a3a36cd5d7 | ||
|
|
8afaee7d21 | ||
|
|
c8280de9c3 | ||
|
|
f8b9df6438 | ||
|
|
92d14fbd60 | ||
|
|
01f6b99631 | ||
|
|
8d5e4ee052 | ||
|
|
9e0aab6b6a | ||
|
|
3eb5a454fd | ||
|
|
dc49bf4d14 |
Generated
+4
@@ -3947,9 +3947,12 @@ name = "inbuxa-features"
|
||||
version = "0.16.22"
|
||||
dependencies = [
|
||||
"ahash",
|
||||
"aho-corasick",
|
||||
"base64 0.23.1",
|
||||
"flate2",
|
||||
"jmap_proto",
|
||||
"quick-xml 0.41.0",
|
||||
"regex",
|
||||
"registry",
|
||||
"serde",
|
||||
"serde_json",
|
||||
@@ -3961,6 +3964,7 @@ dependencies = [
|
||||
"types",
|
||||
"utils",
|
||||
"xxhash-rust",
|
||||
"zip",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
|
||||
@@ -296,6 +296,17 @@ impl Default for DefaultPermissions {
|
||||
default.superuser.push(permission);
|
||||
default.tenant.push(permission);
|
||||
}
|
||||
// inbuxa: DLP and mail flow rules, and held mail, are the
|
||||
// server's: never a tenant's (dlp-and-mail-flow-rules spec,
|
||||
// settled answer 3)
|
||||
Permission::SysMailRuleGet
|
||||
| Permission::SysMailRuleUpdate
|
||||
| Permission::SysDlpPolicyGet
|
||||
| Permission::SysDlpPolicyUpdate
|
||||
| Permission::SysDlpReviewGet
|
||||
| Permission::SysDlpReviewUpdate => {
|
||||
default.superuser.push(permission);
|
||||
}
|
||||
// inbuxa: AL-12: tenant administrators lock and delegate
|
||||
// within their tenant
|
||||
Permission::SysAccountLockGet
|
||||
|
||||
@@ -65,6 +65,10 @@ const OFFICER: &[Permission] = &[
|
||||
Permission::SysLegalHoldUpdate,
|
||||
Permission::SysLegalHoldExport,
|
||||
Permission::SysAccountLockGet,
|
||||
// dlp-and-mail-flow-rules spec, §2.8: see DLP rules, review held mail
|
||||
Permission::SysDlpPolicyGet,
|
||||
Permission::SysDlpReviewGet,
|
||||
Permission::SysDlpReviewUpdate,
|
||||
];
|
||||
|
||||
/// What a tenant's officer holds besides [`READS`].
|
||||
@@ -113,6 +117,11 @@ fn created_key(tenant: Option<Id>) -> ValueClass {
|
||||
})
|
||||
}
|
||||
|
||||
/// The server-level Compliance Officer role the server made, if it has.
|
||||
pub async fn server_role(data: &Store) -> trc::Result<Option<Id>> {
|
||||
recorded(data, None).await
|
||||
}
|
||||
|
||||
async fn recorded(data: &Store, tenant: Option<Id>) -> trc::Result<Option<Id>> {
|
||||
Ok(data
|
||||
.get_value::<u64>(ValueKey::from(created_key(tenant)))
|
||||
@@ -172,13 +181,19 @@ pub async fn ensure_compliance_roles(registry: &RegistryStore, data: &Store) ->
|
||||
|
||||
/// A new tenant gets its Compliance Officer role.
|
||||
pub async fn tenant_created(registry: &RegistryStore, data: &Store, tenant: Id) -> trc::Result<()> {
|
||||
create_once(registry, data, Some(tenant), tenant_role(tenant)).await.map(|_| ())
|
||||
create_once(registry, data, Some(tenant), tenant_role(tenant))
|
||||
.await
|
||||
.map(|_| ())
|
||||
}
|
||||
|
||||
/// Before a tenant is deleted: removes its Compliance Officer role if nobody
|
||||
/// holds it, so the role doesn't block the delete. Returns whether it did,
|
||||
/// so a delete refused for another reason can put it back.
|
||||
pub async fn tenant_deleting(registry: &RegistryStore, data: &Store, tenant: Id) -> trc::Result<bool> {
|
||||
pub async fn tenant_deleting(
|
||||
registry: &RegistryStore,
|
||||
data: &Store,
|
||||
tenant: Id,
|
||||
) -> trc::Result<bool> {
|
||||
let Some(role) = recorded(data, Some(tenant)).await? else {
|
||||
return Ok(false);
|
||||
};
|
||||
@@ -220,7 +235,9 @@ mod tests {
|
||||
// Beyond what any user holds for their own account
|
||||
for permission in all.into_iter().filter(|p| !user.contains(p)) {
|
||||
let name = permission.as_str();
|
||||
let holds = name.starts_with("sysLegalHold");
|
||||
// Placing holds and reviewing held mail are the officer's
|
||||
// job, not settings (settled answers 2 and 4)
|
||||
let holds = name.starts_with("sysLegalHold") || name.starts_with("sysDlpReview");
|
||||
assert!(
|
||||
!(name.ends_with("Update") && !holds)
|
||||
&& !(name.ends_with("Create") && !holds)
|
||||
@@ -249,7 +266,11 @@ mod tests {
|
||||
assert!(officer.contains(&hold));
|
||||
assert!(!tenant.contains(&hold));
|
||||
}
|
||||
for both in [Permission::SysComplianceGet, Permission::SysAuditGet, Permission::SysAccountGet] {
|
||||
for both in [
|
||||
Permission::SysComplianceGet,
|
||||
Permission::SysAuditGet,
|
||||
Permission::SysAccountGet,
|
||||
] {
|
||||
assert!(officer.contains(&both) && tenant.contains(&both));
|
||||
}
|
||||
assert!(!officer.contains(&Permission::SysAuditSettingsUpdate));
|
||||
@@ -257,9 +278,15 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn records_are_per_place() {
|
||||
let ValueClass::Any(server) = created_key(None) else { panic!() };
|
||||
let ValueClass::Any(a) = created_key(Some(Id::from(1u64))) else { panic!() };
|
||||
let ValueClass::Any(b) = created_key(Some(Id::from(2u64))) else { panic!() };
|
||||
let ValueClass::Any(server) = created_key(None) else {
|
||||
panic!()
|
||||
};
|
||||
let ValueClass::Any(a) = created_key(Some(Id::from(1u64))) else {
|
||||
panic!()
|
||||
};
|
||||
let ValueClass::Any(b) = created_key(Some(Id::from(2u64))) else {
|
||||
panic!()
|
||||
};
|
||||
assert_eq!(server.key, b"Pc");
|
||||
assert_ne!(a.key, b.key);
|
||||
assert!(a.key.starts_with(b"Pc"));
|
||||
|
||||
@@ -46,6 +46,21 @@ const ADMIN_GRANTS: &[Permission] = &[
|
||||
Permission::SysLegalHoldUpdate,
|
||||
Permission::SysLegalHoldExport,
|
||||
Permission::SysComplianceGet,
|
||||
Permission::SysMailRuleGet,
|
||||
Permission::SysMailRuleUpdate,
|
||||
Permission::SysDlpPolicyGet,
|
||||
Permission::SysDlpPolicyUpdate,
|
||||
Permission::SysDlpReviewGet,
|
||||
Permission::SysDlpReviewUpdate,
|
||||
];
|
||||
|
||||
/// Granted to the server-level Compliance Officer role once it exists:
|
||||
/// seeing DLP rules and reviewing held mail (dlp-and-mail-flow-rules spec,
|
||||
/// §2.8, settled answer 4). A new install's role has them from the start.
|
||||
const OFFICER_GRANTS: &[Permission] = &[
|
||||
Permission::SysDlpPolicyGet,
|
||||
Permission::SysDlpReviewGet,
|
||||
Permission::SysDlpReviewUpdate,
|
||||
];
|
||||
|
||||
/// Granted to the default tenant administrator roles: reading and exporting
|
||||
@@ -65,13 +80,16 @@ const TENANT_GRANTS: &[Permission] = &[
|
||||
enum Audience {
|
||||
Admin,
|
||||
Tenant,
|
||||
Officer,
|
||||
}
|
||||
|
||||
fn granted_key(permission: Permission, audience: Audience) -> ValueClass {
|
||||
let mut key = b"Pg".to_vec();
|
||||
// Admin grants keep the key they were first recorded under
|
||||
if audience == Audience::Tenant {
|
||||
key.extend_from_slice(b"tenant:");
|
||||
match audience {
|
||||
Audience::Admin => {}
|
||||
Audience::Tenant => key.extend_from_slice(b"tenant:"),
|
||||
Audience::Officer => key.extend_from_slice(b"officer:"),
|
||||
}
|
||||
key.extend_from_slice(permission.as_str().as_bytes());
|
||||
ValueClass::Any(AnyClass {
|
||||
@@ -82,7 +100,8 @@ fn granted_key(permission: Permission, audience: Audience) -> ValueClass {
|
||||
|
||||
pub(crate) async fn grant_new_admin_permissions(bp: &mut Bootstrap) -> trc::Result<()> {
|
||||
grant(bp, Audience::Admin, ADMIN_GRANTS).await?;
|
||||
grant(bp, Audience::Tenant, TENANT_GRANTS).await
|
||||
grant(bp, Audience::Tenant, TENANT_GRANTS).await?;
|
||||
grant(bp, Audience::Officer, OFFICER_GRANTS).await
|
||||
}
|
||||
|
||||
async fn grant(bp: &mut Bootstrap, audience: Audience, grants: &[Permission]) -> trc::Result<()> {
|
||||
@@ -101,10 +120,17 @@ async fn grant(bp: &mut Bootstrap, audience: Audience, grants: &[Permission]) ->
|
||||
if pending.is_empty() {
|
||||
return Ok(());
|
||||
}
|
||||
// The officer role is the one the server made, if it has made it yet: a
|
||||
// new install makes it after this, with the permissions already in it
|
||||
let admin_roles: Vec<Id> = if audience == Audience::Officer {
|
||||
super::compliance_roles::server_role(&bp.data_store)
|
||||
.await?
|
||||
.into_iter()
|
||||
.collect()
|
||||
} else {
|
||||
// An administrator's default roles include the plain User role, which
|
||||
// every user also holds; only roles that are the audience's alone get it
|
||||
let admin_roles: Vec<Id> = bp
|
||||
.registry
|
||||
bp.registry
|
||||
.object::<Authentication>(Id::singleton())
|
||||
.await?
|
||||
.map(|auth| {
|
||||
@@ -118,7 +144,7 @@ async fn grant(bp: &mut Bootstrap, audience: Audience, grants: &[Permission]) ->
|
||||
]
|
||||
.concat(),
|
||||
),
|
||||
Audience::Tenant => (
|
||||
Audience::Tenant | Audience::Officer => (
|
||||
auth.default_tenant_role_ids.as_slice(),
|
||||
[
|
||||
auth.default_user_role_ids.as_slice(),
|
||||
@@ -133,7 +159,8 @@ async fn grant(bp: &mut Bootstrap, audience: Audience, grants: &[Permission]) ->
|
||||
.copied()
|
||||
.collect()
|
||||
})
|
||||
.unwrap_or_default();
|
||||
.unwrap_or_default()
|
||||
};
|
||||
// Fetched by id: the registry's listing doesn't reach stored roles
|
||||
for role_id in admin_roles {
|
||||
let Some(stored) = bp
|
||||
|
||||
@@ -21,6 +21,11 @@ base64 = "0.23"
|
||||
sha2 = "0.11"
|
||||
flate2 = "1.1"
|
||||
tokio = { version = "1.53", features = ["sync", "rt"] }
|
||||
# inbuxa: DLP detectors and attachment text (dlp-and-mail-flow-rules spec)
|
||||
regex = "1.13.1"
|
||||
aho-corasick = "1.1"
|
||||
zip = "8.6"
|
||||
quick-xml = "0.41"
|
||||
|
||||
[dev-dependencies]
|
||||
tokio = { version = "1.53", features = ["macros", "rt"] }
|
||||
|
||||
@@ -23,6 +23,7 @@ pub mod audit;
|
||||
pub mod branding;
|
||||
pub mod hold;
|
||||
pub mod lock;
|
||||
pub mod mailflow;
|
||||
pub mod masked_email;
|
||||
pub mod privacy;
|
||||
pub mod security;
|
||||
|
||||
@@ -0,0 +1,53 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! The compiled rules, kept per node so a message doesn't read the store.
|
||||
//! A change made on this node applies at once; one made on another node
|
||||
//! within [`TTL`], when the copy here is next refreshed.
|
||||
|
||||
use super::{engine::Compiled, rules};
|
||||
use std::{
|
||||
sync::{Arc, RwLock},
|
||||
time::{Duration, Instant},
|
||||
};
|
||||
use store::Store;
|
||||
|
||||
/// How long a node keeps its copy before reading the rules again.
|
||||
pub const TTL: Duration = Duration::from_secs(30);
|
||||
|
||||
static CACHE: RwLock<Option<(Instant, Arc<Compiled>)>> = RwLock::new(None);
|
||||
|
||||
/// Forgets the copy, so the next message reads the rules again.
|
||||
pub fn invalidate() {
|
||||
if let Ok(mut cache) = CACHE.write() {
|
||||
*cache = None;
|
||||
}
|
||||
}
|
||||
|
||||
/// The enabled rules, compiled. A rule that no longer compiles is left out
|
||||
/// and reported, once per refresh.
|
||||
pub async fn compiled(data: &Store) -> trc::Result<Arc<Compiled>> {
|
||||
if let Ok(cache) = CACHE.read()
|
||||
&& let Some((at, compiled)) = cache.as_ref()
|
||||
&& at.elapsed() < TTL
|
||||
{
|
||||
return Ok(compiled.clone());
|
||||
}
|
||||
let (compiled, skipped) = Compiled::new(&rules::all(data).await?);
|
||||
for (id, reason) in skipped {
|
||||
trc::event!(
|
||||
Store(trc::StoreEvent::DataCorruption),
|
||||
Id = u64::from(id),
|
||||
Reason = reason,
|
||||
Details = "Mail rule skipped: it no longer compiles"
|
||||
);
|
||||
}
|
||||
let compiled = Arc::new(compiled);
|
||||
if let Ok(mut cache) = CACHE.write() {
|
||||
*cache = Some((Instant::now(), compiled.clone()));
|
||||
}
|
||||
Ok(compiled)
|
||||
}
|
||||
@@ -0,0 +1,49 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! African identifiers (§2.3): South Africa's ID number.
|
||||
|
||||
use super::{Detector, Findings, Region, Strength, checks, valid_short_date};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[Detector::new(
|
||||
"za-id",
|
||||
"South Africa: ID number",
|
||||
Region::Africa,
|
||||
Strength::Checked,
|
||||
za_id,
|
||||
)];
|
||||
|
||||
/// Birth date `YYMMDD`, four digits, citizenship (0, 1 or 2), 8 or 9, a Luhn
|
||||
/// check digit. The date and the two fixed digits make it strong enough to
|
||||
/// count alone.
|
||||
static ZA_ID: LazyLock<Regex> = LazyLock::new(|| {
|
||||
Regex::new(r"\b(\d{2})(\d{2})(\d{2})\d{4}[012][89]\d\b").expect("detector pattern")
|
||||
});
|
||||
|
||||
fn za_id(text: &str, findings: &mut Findings) {
|
||||
for c in ZA_ID.captures_iter(text) {
|
||||
let n = &c[0];
|
||||
let num = |s: &str| s.parse::<u32>().unwrap_or(0);
|
||||
if valid_short_date(num(&c[1]), num(&c[2]), num(&c[3])) && checks::luhn(n) {
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
#[test]
|
||||
fn south_africa() {
|
||||
let detector = by_id("za-id").unwrap();
|
||||
assert_eq!(detector.count("ID 8001015009087"), 1);
|
||||
assert_eq!(detector.count("8001015009088"), 0);
|
||||
assert_eq!(detector.count("8013015009087"), 0);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,172 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Identifiers from the Americas outside the US and Canada (§2.3): Brazil's
|
||||
//! CPF and CNPJ, and Mexico's CURP.
|
||||
|
||||
use super::{Detector, Findings, Region, Strength, digit_values, valid_short_date, word_near};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[
|
||||
Detector::new(
|
||||
"br-cpf",
|
||||
"Brazil: CPF",
|
||||
Region::Americas,
|
||||
Strength::Checked,
|
||||
br_cpf,
|
||||
),
|
||||
Detector::new(
|
||||
"br-cnpj",
|
||||
"Brazil: CNPJ",
|
||||
Region::Americas,
|
||||
Strength::Checked,
|
||||
br_cnpj,
|
||||
),
|
||||
Detector::new(
|
||||
"mx-curp",
|
||||
"Mexico: CURP",
|
||||
Region::Americas,
|
||||
Strength::Checked,
|
||||
mx_curp,
|
||||
),
|
||||
];
|
||||
|
||||
fn re(pattern: &str) -> Regex {
|
||||
Regex::new(pattern).expect("detector pattern")
|
||||
}
|
||||
|
||||
/// Brazil's mod 11 check digit over `digits` with `weights`.
|
||||
fn br_check(digits: &[u32], weights: &[u32]) -> u32 {
|
||||
match digits.iter().zip(weights).map(|(a, w)| a * w).sum::<u32>() % 11 {
|
||||
0 | 1 => 0,
|
||||
r => 11 - r,
|
||||
}
|
||||
}
|
||||
|
||||
/// `111.444.777-35`, or eleven bare digits.
|
||||
static CPF: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{3}(\.?)\d{3}(\.?)\d{3}(-?)\d{2}\b"));
|
||||
|
||||
pub fn cpf_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
// A run of one digit passes the arithmetic but is never issued
|
||||
d.len() == 11
|
||||
&& d.iter().any(|x| *x != d[0])
|
||||
&& br_check(&d[..9], &[10, 9, 8, 7, 6, 5, 4, 3, 2]) == d[9]
|
||||
&& br_check(&d[..10], &[11, 10, 9, 8, 7, 6, 5, 4, 3, 2]) == d[10]
|
||||
}
|
||||
|
||||
const CPF_WORDS: &[&str] = &[
|
||||
"cpf",
|
||||
"cadastro de pessoas físicas",
|
||||
"cadastro de pessoa física",
|
||||
];
|
||||
|
||||
fn br_cpf(text: &str, findings: &mut Findings) {
|
||||
for c in CPF.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let written = &c[1] == "." && &c[2] == "." && &c[3] == "-";
|
||||
let n: String = whole
|
||||
.as_str()
|
||||
.chars()
|
||||
.filter(char::is_ascii_digit)
|
||||
.collect();
|
||||
if cpf_valid(&n) && (written || word_near(text, whole.start(), whole.end(), CPF_WORDS)) {
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// `11.222.333/0001-81`, or fourteen bare digits.
|
||||
static CNPJ: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\b\d{2}(\.?)\d{3}(\.?)\d{3}(/?)\d{4}(-?)\d{2}\b"));
|
||||
|
||||
pub fn cnpj_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
d.len() == 14
|
||||
&& d.iter().any(|x| *x != d[0])
|
||||
&& br_check(&d[..12], &[5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2]) == d[12]
|
||||
&& br_check(&d[..13], &[6, 5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2]) == d[13]
|
||||
}
|
||||
|
||||
const CNPJ_WORDS: &[&str] = &["cnpj", "cadastro nacional da pessoa jurídica"];
|
||||
|
||||
fn br_cnpj(text: &str, findings: &mut Findings) {
|
||||
for c in CNPJ.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let written = &c[1] == "." && &c[2] == "." && &c[3] == "/" && &c[4] == "-";
|
||||
let n: String = whole
|
||||
.as_str()
|
||||
.chars()
|
||||
.filter(char::is_ascii_digit)
|
||||
.collect();
|
||||
if cnpj_valid(&n) && (written || word_near(text, whole.start(), whole.end(), CNPJ_WORDS)) {
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Four letters, the birth date, sex (H, M or X), the state, three
|
||||
/// consonants, a character that tells the century apart, the check digit.
|
||||
static CURP: LazyLock<Regex> = LazyLock::new(|| {
|
||||
re(r"(?i)\b[A-Z]{4}(\d{2})(\d{2})(\d{2})[HMX][A-Z]{2}[B-DF-HJ-NP-TV-Z]{3}[A-Z0-9]\d\b")
|
||||
});
|
||||
|
||||
/// RENAPO's check: each character's place in `0-9 A-N Ñ O-Z`, weighted 18
|
||||
/// down to 2; the digit is 10 minus the sum mod 10 (10 becomes 0).
|
||||
pub fn curp_valid(curp: &str) -> bool {
|
||||
const ALPHABET: &str = "0123456789ABCDEFGHIJKLMNÑOPQRSTUVWXYZ";
|
||||
let mut sum = 0u32;
|
||||
for (i, c) in curp.chars().take(17).enumerate() {
|
||||
let Some(value) = ALPHABET.chars().position(|a| a == c) else {
|
||||
return false;
|
||||
};
|
||||
sum += value as u32 * (18 - i as u32);
|
||||
}
|
||||
curp.chars().nth(17).and_then(|c| c.to_digit(10)) == Some((10 - sum % 10) % 10)
|
||||
}
|
||||
|
||||
fn mx_curp(text: &str, findings: &mut Findings) {
|
||||
for c in CURP.captures_iter(text) {
|
||||
let curp = c[0].to_ascii_uppercase();
|
||||
if valid_short_date(num(&c[1]), num(&c[2]), num(&c[3])) && curp_valid(&curp) {
|
||||
findings.insert(curp);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn num(s: &str) -> u32 {
|
||||
s.parse().unwrap_or(0)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
fn count(id: &str, text: &str) -> usize {
|
||||
by_id(id).unwrap().count(text)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn brazil() {
|
||||
assert_eq!(count("br-cpf", "CPF 111.444.777-35"), 1);
|
||||
assert_eq!(count("br-cpf", "111.444.777-36"), 0);
|
||||
assert_eq!(count("br-cpf", "pedido 11144477735"), 0);
|
||||
assert_eq!(count("br-cpf", "cpf: 11144477735"), 1);
|
||||
assert_eq!(count("br-cpf", "CPF 111.111.111-11"), 0);
|
||||
assert_eq!(count("br-cnpj", "11.222.333/0001-81"), 1);
|
||||
assert_eq!(count("br-cnpj", "11.222.333/0001-82"), 0);
|
||||
assert_eq!(count("br-cnpj", "CNPJ 11222333000181"), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn mexico() {
|
||||
// python-stdnum's documented example
|
||||
assert_eq!(count("mx-curp", "CURP BOXW310820HNERXN09"), 1);
|
||||
assert_eq!(count("mx-curp", "BOXW310820HNERXN08"), 0);
|
||||
assert_eq!(count("mx-curp", "BOXW311320HNERXN09"), 0);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,511 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Detectors that aren't tied to one country (§2.3, region "Any").
|
||||
|
||||
use super::{
|
||||
Detector, Findings, Region, Strength, checks, digits, stands_alone, valid_date, word_near,
|
||||
};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[
|
||||
Detector::new(
|
||||
"payment-card",
|
||||
"Payment card number",
|
||||
Region::Any,
|
||||
Strength::Checked,
|
||||
payment_card,
|
||||
),
|
||||
Detector::new("iban", "IBAN", Region::Any, Strength::Checked, iban),
|
||||
Detector::new(
|
||||
"swift-bic",
|
||||
"SWIFT/BIC code",
|
||||
Region::Any,
|
||||
Strength::NeedsWord,
|
||||
swift_bic,
|
||||
),
|
||||
Detector::new(
|
||||
"email-addresses",
|
||||
"Email addresses",
|
||||
Region::Any,
|
||||
Strength::Checked,
|
||||
email_addresses,
|
||||
),
|
||||
Detector::new(
|
||||
"phone-numbers",
|
||||
"Phone numbers",
|
||||
Region::Any,
|
||||
Strength::NeedsWord,
|
||||
phone_numbers,
|
||||
),
|
||||
Detector::new(
|
||||
"date-of-birth",
|
||||
"Date of birth",
|
||||
Region::Any,
|
||||
Strength::NeedsWord,
|
||||
date_of_birth,
|
||||
),
|
||||
Detector::new(
|
||||
"passport",
|
||||
"Passport number",
|
||||
Region::Any,
|
||||
Strength::NeedsWord,
|
||||
passport,
|
||||
),
|
||||
Detector::new(
|
||||
"private-key",
|
||||
"Private key",
|
||||
Region::Any,
|
||||
Strength::Checked,
|
||||
private_key,
|
||||
),
|
||||
Detector::new(
|
||||
"credentials",
|
||||
"Cloud and service credentials",
|
||||
Region::Any,
|
||||
Strength::Checked,
|
||||
credentials,
|
||||
),
|
||||
];
|
||||
|
||||
fn re(pattern: &str) -> Regex {
|
||||
Regex::new(pattern).expect("detector pattern")
|
||||
}
|
||||
|
||||
// --- Payment cards --------------------------------------------------------
|
||||
|
||||
/// Issuer prefixes (ISO/IEC 7812 IINs) and the lengths each network issues.
|
||||
fn card_network(number: &str) -> bool {
|
||||
let len = number.len();
|
||||
let prefix = |n: usize| number[..n].parse::<u32>().unwrap_or(0);
|
||||
match number.as_bytes()[0] {
|
||||
// Visa
|
||||
b'4' => matches!(len, 13 | 16 | 19),
|
||||
b'5' => {
|
||||
// Mastercard 51–55; Maestro 50, 56–58
|
||||
(51..=55).contains(&prefix(2)) && len == 16
|
||||
|| matches!(prefix(2), 50 | 56..=58) && (12..=19).contains(&len)
|
||||
}
|
||||
// Mastercard 2221–2720
|
||||
b'2' => (2221..=2720).contains(&prefix(4)) && len == 16,
|
||||
b'3' => {
|
||||
// American Express 34, 37; JCB 3528–3589; Diners 300–305, 36, 38, 39
|
||||
matches!(prefix(2), 34 | 37) && len == 15
|
||||
|| (3528..=3589).contains(&prefix(4)) && (16..=19).contains(&len)
|
||||
|| ((300..=305).contains(&prefix(3)) || matches!(prefix(2), 36 | 38 | 39))
|
||||
&& (14..=19).contains(&len)
|
||||
}
|
||||
// Discover 6011, 644–649, 65; UnionPay 62; Maestro 6x
|
||||
b'6' => (12..=19).contains(&len),
|
||||
_ => false,
|
||||
}
|
||||
}
|
||||
|
||||
fn is_card(number: &str) -> bool {
|
||||
(12..=19).contains(&number.len()) && card_network(number) && checks::luhn(number)
|
||||
}
|
||||
|
||||
static CARD: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d(?:[ -]?\d){11,18}\b"));
|
||||
|
||||
fn payment_card(text: &str, findings: &mut Findings) {
|
||||
for m in CARD.find_iter(text) {
|
||||
if !stands_alone(text, m.start(), m.end()) {
|
||||
continue;
|
||||
}
|
||||
let whole = digits(m.as_str());
|
||||
if is_card(&whole) {
|
||||
findings.insert(whole);
|
||||
continue;
|
||||
}
|
||||
// Two numbers side by side ("4242 4242 4242 4242 2031"): try each
|
||||
// run of whole groups
|
||||
let groups: Vec<String> = m.as_str().split([' ', '-']).map(digits).collect();
|
||||
'runs: for from in 0..groups.len() {
|
||||
let mut number = String::new();
|
||||
for group in &groups[from..] {
|
||||
number.push_str(group);
|
||||
if is_card(&number) {
|
||||
findings.insert(number);
|
||||
break 'runs;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- IBAN -----------------------------------------------------------------
|
||||
|
||||
static IBAN: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\b[A-Za-z]{2}\d{2}(?:[ ]?[A-Za-z0-9]){11,30}"));
|
||||
|
||||
fn iban(text: &str, findings: &mut Findings) {
|
||||
// The pattern can run on into the next words, even the next IBAN: after
|
||||
// each hit, look again from where that IBAN ended
|
||||
let mut from = 0;
|
||||
while let Some(m) = IBAN.find_at(text, from) {
|
||||
from = m.start() + 1;
|
||||
let compact = m.as_str().replace(' ', "").to_ascii_uppercase();
|
||||
let Some(len) = checks::iban_length(&compact[..2]) else {
|
||||
continue;
|
||||
};
|
||||
if compact.len() < len {
|
||||
continue;
|
||||
}
|
||||
// Where the country's length ends in the text, spaces counted
|
||||
let mut seen = 0;
|
||||
let Some(end) = m
|
||||
.as_str()
|
||||
.char_indices()
|
||||
.find(|(_, c)| {
|
||||
if *c != ' ' {
|
||||
seen += 1;
|
||||
}
|
||||
seen == len
|
||||
})
|
||||
.map(|(i, c)| m.start() + i + c.len_utf8())
|
||||
else {
|
||||
continue;
|
||||
};
|
||||
let candidate = &compact[..len];
|
||||
if stands_alone(text, m.start(), end) && checks::iban(candidate) {
|
||||
findings.insert(candidate);
|
||||
from = end;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- SWIFT/BIC ------------------------------------------------------------
|
||||
|
||||
static BIC: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\b[A-Z]{4}[A-Z]{2}[A-Z0-9]{2}(?:[A-Z0-9]{3})?\b"));
|
||||
|
||||
const BIC_WORDS: &[&str] = &[
|
||||
"swift",
|
||||
"bic",
|
||||
"swift/bic",
|
||||
"bank",
|
||||
"banque",
|
||||
"bankverbindung",
|
||||
];
|
||||
|
||||
fn swift_bic(text: &str, findings: &mut Findings) {
|
||||
for m in BIC.find_iter(text) {
|
||||
let code = m.as_str();
|
||||
if checks::is_country(&code[4..6]) && word_near(text, m.start(), m.end(), BIC_WORDS) {
|
||||
findings.insert(code);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Contact lists --------------------------------------------------------
|
||||
|
||||
static EMAIL: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"(?i)\b[a-z0-9._%+-]+@[a-z0-9-]+(?:\.[a-z0-9-]+)*\.[a-z]{2,}\b"));
|
||||
|
||||
fn email_addresses(text: &str, findings: &mut Findings) {
|
||||
for m in EMAIL.find_iter(text) {
|
||||
findings.insert(m.as_str().to_lowercase());
|
||||
}
|
||||
}
|
||||
|
||||
/// International form: found alone. National form: only with a word.
|
||||
static PHONE_INTL: LazyLock<Regex> = LazyLock::new(|| re(r"\+\d{1,3}(?:[ .-]?\(?\d{1,4}\)?){2,5}"));
|
||||
static PHONE_NATIONAL: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\(?\d{2,4}\)?[ .-]\d{3,4}[ .-]\d{3,4}"));
|
||||
|
||||
const PHONE_WORDS: &[&str] = &[
|
||||
"phone",
|
||||
"tel",
|
||||
"telephone",
|
||||
"mobile",
|
||||
"cell",
|
||||
"fax",
|
||||
"telefon",
|
||||
"téléphone",
|
||||
"teléfono",
|
||||
"telefono",
|
||||
"handy",
|
||||
"portable",
|
||||
"móvil",
|
||||
"cellulare",
|
||||
"mobiel",
|
||||
];
|
||||
|
||||
fn phone_numbers(text: &str, findings: &mut Findings) {
|
||||
let mut international = Vec::new();
|
||||
for m in PHONE_INTL.find_iter(text) {
|
||||
let number = digits(m.as_str());
|
||||
if (8..=15).contains(&number.len()) && stands_alone(text, m.start() + 1, m.end()) {
|
||||
findings.insert(number);
|
||||
international.push(m.range());
|
||||
}
|
||||
}
|
||||
for m in PHONE_NATIONAL.find_iter(text) {
|
||||
let number = digits(m.as_str());
|
||||
// Not the tail of an international number already counted
|
||||
if international.iter().any(|r| r.contains(&m.start())) {
|
||||
continue;
|
||||
}
|
||||
if (9..=11).contains(&number.len())
|
||||
&& stands_alone(text, m.start(), m.end())
|
||||
&& !text[..m.start()].ends_with('+')
|
||||
&& word_near(text, m.start(), m.end(), PHONE_WORDS)
|
||||
{
|
||||
findings.insert(number);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Date of birth --------------------------------------------------------
|
||||
|
||||
static DATE_ISO: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{4})-(\d{2})-(\d{2})\b"));
|
||||
static DATE_NUMERIC: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\b(\d{1,2})[./-](\d{1,2})[./-](\d{4})\b"));
|
||||
static DATE_WORDS: LazyLock<Regex> = LazyLock::new(|| {
|
||||
re(
|
||||
r"(?i)\b(?:(\d{1,2})\s+(jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)[a-z]*\.?,?\s+(\d{4})|(jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)[a-z]*\.?\s+(\d{1,2}),?\s+(\d{4}))\b",
|
||||
)
|
||||
});
|
||||
|
||||
const BIRTH_WORDS: &[&str] = &[
|
||||
"born",
|
||||
"birth",
|
||||
"dob",
|
||||
"d.o.b",
|
||||
"birthday",
|
||||
"birthdate",
|
||||
"geburtsdatum",
|
||||
"geboren",
|
||||
"naissance",
|
||||
"né le",
|
||||
"née le",
|
||||
"nacimiento",
|
||||
"nacido",
|
||||
"nacida",
|
||||
"nascita",
|
||||
"nato il",
|
||||
"nata il",
|
||||
"geboortedatum",
|
||||
"födelsedatum",
|
||||
"fødselsdato",
|
||||
"syntymäaika",
|
||||
"urodzenia",
|
||||
"nascimento",
|
||||
];
|
||||
|
||||
fn month_number(name: &str) -> u32 {
|
||||
const MONTHS: [&str; 12] = [
|
||||
"jan", "feb", "mar", "apr", "may", "jun", "jul", "aug", "sep", "oct", "nov", "dec",
|
||||
];
|
||||
let name = name.to_lowercase();
|
||||
MONTHS
|
||||
.iter()
|
||||
.position(|m| *m == name)
|
||||
.map_or(0, |i| i as u32 + 1)
|
||||
}
|
||||
|
||||
fn date_of_birth(text: &str, findings: &mut Findings) {
|
||||
let mut add = |start: usize, end: usize, key: String| {
|
||||
if word_near(text, start, end, BIRTH_WORDS) {
|
||||
findings.insert(key);
|
||||
}
|
||||
};
|
||||
let num = |s: &str| s.parse::<u32>().unwrap_or(0);
|
||||
for c in DATE_ISO.captures_iter(text) {
|
||||
let (y, m, d) = (num(&c[1]), num(&c[2]), num(&c[3]));
|
||||
let whole = c.get(0).unwrap();
|
||||
if valid_date(y, m, d) {
|
||||
add(whole.start(), whole.end(), format!("{y:04}{m:02}{d:02}"));
|
||||
}
|
||||
}
|
||||
for c in DATE_NUMERIC.captures_iter(text) {
|
||||
let (a, b, y) = (num(&c[1]), num(&c[2]), num(&c[3]));
|
||||
let whole = c.get(0).unwrap();
|
||||
// Day first or month first: either reading that is a real date
|
||||
if valid_date(y, b, a) || valid_date(y, a, b) {
|
||||
add(whole.start(), whole.end(), whole.as_str().to_string());
|
||||
}
|
||||
}
|
||||
for c in DATE_WORDS.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let (d, m, y) = match (c.get(1), c.get(4)) {
|
||||
(Some(d), _) => (num(d.as_str()), month_number(&c[2]), num(&c[3])),
|
||||
(_, Some(m)) => (num(&c[5]), month_number(m.as_str()), num(&c[6])),
|
||||
_ => continue,
|
||||
};
|
||||
if valid_date(y, m, d) {
|
||||
add(whole.start(), whole.end(), format!("{y:04}{m:02}{d:02}"));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Passport -------------------------------------------------------------
|
||||
|
||||
static PASSPORT: LazyLock<Regex> = LazyLock::new(|| re(r"\b[A-Z0-9]{6,9}\b"));
|
||||
|
||||
const PASSPORT_WORDS: &[&str] = &[
|
||||
"passport",
|
||||
"passeport",
|
||||
"reisepass",
|
||||
"pasaporte",
|
||||
"passaporto",
|
||||
"paspoort",
|
||||
"passnummer",
|
||||
"pass-nr",
|
||||
"passport no",
|
||||
"pasaporte n.º",
|
||||
"passaporte",
|
||||
];
|
||||
|
||||
fn passport(text: &str, findings: &mut Findings) {
|
||||
for m in PASSPORT.find_iter(text) {
|
||||
let value = m.as_str();
|
||||
if value.bytes().filter(u8::is_ascii_digit).count() >= 5
|
||||
&& word_near(text, m.start(), m.end(), PASSPORT_WORDS)
|
||||
{
|
||||
findings.insert(value);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Keys and credentials -------------------------------------------------
|
||||
|
||||
static PRIVATE_KEY: LazyLock<Regex> = LazyLock::new(|| {
|
||||
re(
|
||||
r"-----BEGIN (?:(?:RSA|EC|DSA|OPENSSH|ENCRYPTED|PGP) )?PRIVATE KEY(?: BLOCK)?-----\s*([A-Za-z0-9+/=:\s-]{0,64})",
|
||||
)
|
||||
});
|
||||
|
||||
fn private_key(text: &str, findings: &mut Findings) {
|
||||
for c in PRIVATE_KEY.captures_iter(text) {
|
||||
// Each key once, by the start of its body
|
||||
let body: String = c[1].chars().filter(|c| !c.is_whitespace()).collect();
|
||||
let whole = c.get(0).unwrap();
|
||||
findings.insert(if body.is_empty() {
|
||||
format!("@{}", whole.start())
|
||||
} else {
|
||||
body
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
/// Published token formats: AWS access key IDs, GitHub tokens, Slack
|
||||
/// tokens, Stripe live secret and restricted keys, Google API keys.
|
||||
static CREDENTIAL: LazyLock<Regex> = LazyLock::new(|| {
|
||||
re(concat!(
|
||||
r"\b(?:",
|
||||
r"(?:AKIA|ASIA|ABIA|ACCA)[A-Z0-9]{16}",
|
||||
r"|gh[pousr]_[A-Za-z0-9]{36}",
|
||||
r"|github_pat_[A-Za-z0-9_]{82}",
|
||||
r"|xox[abposr]-[A-Za-z0-9-]{10,72}",
|
||||
r"|(?:sk|rk)_live_[A-Za-z0-9]{24,99}",
|
||||
r"|AIza[0-9A-Za-z_-]{35}",
|
||||
r")\b"
|
||||
))
|
||||
});
|
||||
|
||||
fn credentials(text: &str, findings: &mut Findings) {
|
||||
for m in CREDENTIAL.find_iter(text) {
|
||||
findings.insert(m.as_str());
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
fn count(id: &str, text: &str) -> usize {
|
||||
by_id(id).unwrap().count(text)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn payment_cards() {
|
||||
// Networks' and processors' published test numbers
|
||||
let text = "Visa 4242 4242 4242 4242, MC 5555-5555-5555-4444, Amex 378282246310005, \
|
||||
Discover 6011111111111117, JCB 3566002020360505, Diners 30569309025904, \
|
||||
UnionPay 6200000000000005, Mastercard 2-series 2223003122003222";
|
||||
assert_eq!(count("payment-card", text), 8);
|
||||
// Luhn fails, wrong network length, inside a longer number
|
||||
assert_eq!(count("payment-card", "4242424242424241"), 0);
|
||||
assert_eq!(count("payment-card", "378282246310005 0"), 1);
|
||||
assert_eq!(count("payment-card", "order 94242424242424242 shipped"), 0);
|
||||
// The same number twice counts once
|
||||
assert_eq!(
|
||||
count("payment-card", "4242424242424242 and 4242-4242-4242-4242"),
|
||||
1
|
||||
);
|
||||
// A card followed by a year
|
||||
assert_eq!(count("payment-card", "card 4242 4242 4242 4242 2031"), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ibans() {
|
||||
let text =
|
||||
"Pay GB29 NWBK 6016 1331 9268 19 or de89370400440532013000 (NL91ABNA0417164300).";
|
||||
assert_eq!(count("iban", text), 3);
|
||||
assert_eq!(count("iban", "GB29 NWBK 6016 1331 9268 18"), 0);
|
||||
// Runs into the next word: still found at the country's length
|
||||
assert_eq!(count("iban", "IBAN NL91ABNA0417164300 BIC ABNANL2A"), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn swift_codes_need_a_word() {
|
||||
assert_eq!(count("swift-bic", "SWIFT: DEUTDEFF500"), 1);
|
||||
assert_eq!(count("swift-bic", "BIC NWBKGB2L"), 1);
|
||||
assert_eq!(count("swift-bic", "HAPPYDAYS DEUTDEFF"), 0);
|
||||
// Not a country in positions 5–6
|
||||
assert_eq!(count("swift-bic", "BIC DEUTZZFF"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn email_and_phone_lists() {
|
||||
let list = "[email protected], [email protected], [email protected], [email protected]";
|
||||
assert_eq!(count("email-addresses", list), 3);
|
||||
assert_eq!(
|
||||
count("phone-numbers", "+44 20 7946 0958, +1 (415) 555-2671"),
|
||||
2
|
||||
);
|
||||
assert_eq!(count("phone-numbers", "call 020 7946 0958"), 0);
|
||||
assert_eq!(count("phone-numbers", "Tel: 020 7946 0958"), 1);
|
||||
assert_eq!(count("phone-numbers", "invoice 020 7946 0958"), 0);
|
||||
// One number, not also its national tail
|
||||
assert_eq!(count("phone-numbers", "Tel: +44 20 7946 0958"), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dates_of_birth() {
|
||||
assert_eq!(count("date-of-birth", "DOB: 1984-02-29"), 1);
|
||||
assert_eq!(count("date-of-birth", "Geburtsdatum 31.12.1970"), 1);
|
||||
assert_eq!(count("date-of-birth", "born on March 3, 1962"), 1);
|
||||
assert_eq!(count("date-of-birth", "date of birth 3 Mar 1962"), 1);
|
||||
// Not a real date, no word, a meeting
|
||||
assert_eq!(count("date-of-birth", "DOB: 1985-02-29"), 0);
|
||||
assert_eq!(count("date-of-birth", "invoice 1984-02-29"), 0);
|
||||
assert_eq!(count("date-of-birth", "Meeting on 12/05/2026"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn passports_need_a_word() {
|
||||
assert_eq!(count("passport", "Passport number: 533380006"), 1);
|
||||
assert_eq!(count("passport", "Reisepass C01X00T47"), 1);
|
||||
assert_eq!(count("passport", "Order 533380006 shipped"), 0);
|
||||
// Mostly letters: a word, not a number
|
||||
assert_eq!(count("passport", "passport PASSWORD"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn keys_and_credentials() {
|
||||
let key = "-----BEGIN OPENSSH PRIVATE KEY-----\nb3BlbnNzaC1rZXktdjEAAAAABG5vbmUAAAAEbm9uZQ\n-----END OPENSSH PRIVATE KEY-----";
|
||||
assert_eq!(count("private-key", key), 1);
|
||||
assert_eq!(count("private-key", "-----BEGIN PUBLIC KEY-----\nMFkw"), 0);
|
||||
// Documentation examples of each format
|
||||
let tokens = "AKIAIOSFODNN7EXAMPLE ghp_0123456789abcdefghijklmnopqrstuvwxyz \
|
||||
AIzaSyA-0123456789abcdefghijklmnopqrstu";
|
||||
assert_eq!(count("credentials", tokens), 3);
|
||||
assert_eq!(count("credentials", "AKIA123 ghp_short"), 0);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,281 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Asian identifiers (§2.3): India's Aadhaar and PAN, China's resident ID,
|
||||
//! Japan's My Number, Singapore's NRIC and FIN, and South Korea's resident
|
||||
//! registration number.
|
||||
|
||||
use super::{
|
||||
Detector, Findings, Region, Strength, digit_values, valid_date, valid_short_date, word_near,
|
||||
};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[
|
||||
Detector::new(
|
||||
"in-aadhaar",
|
||||
"India: Aadhaar",
|
||||
Region::Asia,
|
||||
Strength::Checked,
|
||||
in_aadhaar,
|
||||
),
|
||||
Detector::new(
|
||||
"in-pan",
|
||||
"India: PAN",
|
||||
Region::Asia,
|
||||
Strength::NeedsWord,
|
||||
in_pan,
|
||||
),
|
||||
Detector::new(
|
||||
"cn-resident-id",
|
||||
"China: resident ID",
|
||||
Region::Asia,
|
||||
Strength::Checked,
|
||||
cn_resident_id,
|
||||
),
|
||||
Detector::new(
|
||||
"jp-my-number",
|
||||
"Japan: My Number",
|
||||
Region::Asia,
|
||||
Strength::Checked,
|
||||
jp_my_number,
|
||||
),
|
||||
Detector::new(
|
||||
"sg-nric",
|
||||
"Singapore: NRIC and FIN",
|
||||
Region::Asia,
|
||||
Strength::Checked,
|
||||
sg_nric,
|
||||
),
|
||||
Detector::new(
|
||||
"kr-rrn",
|
||||
"South Korea: resident registration number",
|
||||
Region::Asia,
|
||||
Strength::NeedsWord,
|
||||
kr_rrn,
|
||||
),
|
||||
];
|
||||
|
||||
fn re(pattern: &str) -> Regex {
|
||||
Regex::new(pattern).expect("detector pattern")
|
||||
}
|
||||
|
||||
/// Twelve digits written in fours, or bare.
|
||||
static TWELVE: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{4})( ?)(\d{4})( ?)(\d{4})\b"));
|
||||
|
||||
const VERHOEFF_D: [[u8; 10]; 10] = [
|
||||
[0, 1, 2, 3, 4, 5, 6, 7, 8, 9],
|
||||
[1, 2, 3, 4, 0, 6, 7, 8, 9, 5],
|
||||
[2, 3, 4, 0, 1, 7, 8, 9, 5, 6],
|
||||
[3, 4, 0, 1, 2, 8, 9, 5, 6, 7],
|
||||
[4, 0, 1, 2, 3, 9, 5, 6, 7, 8],
|
||||
[5, 9, 8, 7, 6, 0, 4, 3, 2, 1],
|
||||
[6, 5, 9, 8, 7, 1, 0, 4, 3, 2],
|
||||
[7, 6, 5, 9, 8, 2, 1, 0, 4, 3],
|
||||
[8, 7, 6, 5, 9, 3, 2, 1, 0, 4],
|
||||
[9, 8, 7, 6, 5, 4, 3, 2, 1, 0],
|
||||
];
|
||||
const VERHOEFF_P: [[u8; 10]; 8] = [
|
||||
[0, 1, 2, 3, 4, 5, 6, 7, 8, 9],
|
||||
[1, 5, 7, 6, 2, 8, 3, 0, 9, 4],
|
||||
[5, 8, 0, 3, 7, 9, 6, 1, 4, 2],
|
||||
[8, 9, 1, 6, 0, 4, 3, 5, 2, 7],
|
||||
[9, 4, 5, 3, 1, 2, 6, 8, 7, 0],
|
||||
[4, 2, 8, 6, 5, 7, 3, 9, 0, 1],
|
||||
[2, 7, 9, 3, 8, 0, 6, 4, 1, 5],
|
||||
[7, 0, 4, 6, 9, 1, 3, 2, 5, 8],
|
||||
];
|
||||
|
||||
/// The Verhoeff check (dihedral group D5).
|
||||
pub fn verhoeff(n: &str) -> bool {
|
||||
let mut c = 0u8;
|
||||
for (i, b) in n.bytes().rev().enumerate() {
|
||||
c = VERHOEFF_D[c as usize][VERHOEFF_P[i % 8][(b - b'0') as usize] as usize];
|
||||
}
|
||||
c == 0
|
||||
}
|
||||
|
||||
const AADHAAR_WORDS: &[&str] = &["aadhaar", "aadhar", "uidai", "uid"];
|
||||
|
||||
fn in_aadhaar(text: &str, findings: &mut Findings) {
|
||||
for c in TWELVE.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
|
||||
let written = &c[2] == " " && &c[4] == " ";
|
||||
// Never starts with 0 or 1
|
||||
if !n.starts_with(['0', '1'])
|
||||
&& verhoeff(&n)
|
||||
&& (written || word_near(text, whole.start(), whole.end(), AADHAAR_WORDS))
|
||||
{
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Five letters (the fourth names the holder's type), four digits, a letter.
|
||||
static PAN: LazyLock<Regex> = LazyLock::new(|| re(r"\b[A-Z]{3}[ABCFGHLJPTK][A-Z]\d{4}[A-Z]\b"));
|
||||
|
||||
const PAN_WORDS: &[&str] = &["pan", "pan card", "permanent account number", "income tax"];
|
||||
|
||||
fn in_pan(text: &str, findings: &mut Findings) {
|
||||
for m in PAN.find_iter(text) {
|
||||
if word_near(text, m.start(), m.end(), PAN_WORDS) {
|
||||
findings.insert(m.as_str());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Region, birth date `YYYYMMDD`, sequence, then the ISO 7064 MOD 11-2
|
||||
/// check (0–9 or X).
|
||||
static CN_ID: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"(?i)\b[1-8]\d{5}(\d{4})(\d{2})(\d{2})\d{3}[\dX]\b"));
|
||||
|
||||
pub fn cn_id_valid(id: &str) -> bool {
|
||||
const WEIGHTS: [u32; 17] = [7, 9, 10, 5, 8, 4, 2, 1, 6, 3, 7, 9, 10, 5, 8, 4, 2];
|
||||
const CHECKS: &[u8] = b"10X98765432";
|
||||
let sum: u32 = digit_values(&id[..17])
|
||||
.iter()
|
||||
.zip(WEIGHTS)
|
||||
.map(|(a, w)| a * w)
|
||||
.sum();
|
||||
CHECKS[(sum % 11) as usize] == id.as_bytes()[17].to_ascii_uppercase()
|
||||
}
|
||||
|
||||
fn cn_resident_id(text: &str, findings: &mut Findings) {
|
||||
for c in CN_ID.captures_iter(text) {
|
||||
let id = c[0].to_ascii_uppercase();
|
||||
let (y, m, d) = (num(&c[1]), num(&c[2]), num(&c[3]));
|
||||
if valid_date(y, m, d) && cn_id_valid(&id) {
|
||||
findings.insert(id);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// My Number: weights 2–7 then 2–6 from the right; a remainder of 0 or 1
|
||||
/// gives 0, else 11 minus it.
|
||||
pub fn my_number_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
let sum: u32 = (1..=11)
|
||||
.map(|i| d[11 - i] * if i <= 6 { i as u32 + 1 } else { i as u32 - 5 })
|
||||
.sum();
|
||||
let check = match sum % 11 {
|
||||
0 | 1 => 0,
|
||||
r => 11 - r,
|
||||
};
|
||||
check == d[11]
|
||||
}
|
||||
|
||||
const MY_NUMBER_WORDS: &[&str] = &[
|
||||
"my number",
|
||||
"mynumber",
|
||||
"マイナンバー",
|
||||
"個人番号",
|
||||
"kojin bango",
|
||||
];
|
||||
|
||||
fn jp_my_number(text: &str, findings: &mut Findings) {
|
||||
for c in TWELVE.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
|
||||
let written = &c[2] == " " && &c[4] == " ";
|
||||
if my_number_valid(&n)
|
||||
&& (written || word_near(text, whole.start(), whole.end(), MY_NUMBER_WORDS))
|
||||
{
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static NRIC: LazyLock<Regex> = LazyLock::new(|| re(r"(?i)\b([STFGM])(\d{7})([A-Z])\b"));
|
||||
|
||||
/// Weights 2, 7, 6, 5, 4, 3, 2; T and G add 4, M adds 3; each series has its
|
||||
/// own table of check letters.
|
||||
fn nric_valid(prefix: u8, digits: &str, check: u8) -> bool {
|
||||
let sum: u32 = digit_values(digits)
|
||||
.iter()
|
||||
.zip([2, 7, 6, 5, 4, 3, 2])
|
||||
.map(|(a, w)| a * w)
|
||||
.sum::<u32>()
|
||||
+ match prefix {
|
||||
b'T' | b'G' => 4,
|
||||
b'M' => 3,
|
||||
_ => 0,
|
||||
};
|
||||
let table: &[u8] = match prefix {
|
||||
b'S' | b'T' => b"JZIHGFEDCBA",
|
||||
b'F' | b'G' => b"XWUTRQPNMLK",
|
||||
_ => b"KLJNPQRTUWX",
|
||||
};
|
||||
table[(sum % 11) as usize] == check
|
||||
}
|
||||
|
||||
fn sg_nric(text: &str, findings: &mut Findings) {
|
||||
for c in NRIC.captures_iter(text) {
|
||||
let id = c[0].to_ascii_uppercase();
|
||||
let bytes = id.as_bytes();
|
||||
if nric_valid(bytes[0], &c[2], bytes[8]) {
|
||||
findings.insert(id);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// `YYMMDD-GNNNNNN`, the seventh digit giving sex and century.
|
||||
static RRN: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{2})(\d{2})(\d{2})-?([1-8])\d{6}\b"));
|
||||
|
||||
const RRN_WORDS: &[&str] = &["주민등록번호", "주민번호", "resident registration", "rrn"];
|
||||
|
||||
fn kr_rrn(text: &str, findings: &mut Findings) {
|
||||
for c in RRN.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
if valid_short_date(num(&c[1]), num(&c[2]), num(&c[3]))
|
||||
&& word_near(text, whole.start(), whole.end(), RRN_WORDS)
|
||||
{
|
||||
findings.insert(whole.as_str().replace('-', ""));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn num(s: &str) -> u32 {
|
||||
s.parse().unwrap_or(0)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
fn count(id: &str, text: &str) -> usize {
|
||||
by_id(id).unwrap().count(text)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn india() {
|
||||
assert_eq!(count("in-aadhaar", "2345 6789 0124"), 1);
|
||||
assert_eq!(count("in-aadhaar", "2345 6789 0125"), 0);
|
||||
assert_eq!(count("in-aadhaar", "order 234567890124"), 0);
|
||||
assert_eq!(count("in-aadhaar", "Aadhaar 234567890124"), 1);
|
||||
assert_eq!(count("in-pan", "PAN: ABCPE1234F"), 1);
|
||||
assert_eq!(count("in-pan", "ABCPE1234F"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn china_japan() {
|
||||
assert_eq!(count("cn-resident-id", "11010519491231002X"), 1);
|
||||
assert_eq!(count("cn-resident-id", "110105194912310021"), 0);
|
||||
assert_eq!(count("cn-resident-id", "11010519491331002X"), 0);
|
||||
assert_eq!(count("jp-my-number", "1234 5678 9018"), 1);
|
||||
assert_eq!(count("jp-my-number", "1234 5678 9017"), 0);
|
||||
assert_eq!(count("jp-my-number", "マイナンバー 123456789018"), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn singapore_korea() {
|
||||
assert_eq!(count("sg-nric", "S1234567D and T1234567J"), 2);
|
||||
assert_eq!(count("sg-nric", "S1234567E"), 0);
|
||||
assert_eq!(count("kr-rrn", "주민등록번호 800101-1234567"), 1);
|
||||
assert_eq!(count("kr-rrn", "800101-1234567"), 0);
|
||||
assert_eq!(count("kr-rrn", "RRN 801301-1234567"), 0);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,128 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Australian identifiers (§2.3): the ATO's Tax File Number and the Medicare
|
||||
//! card number.
|
||||
|
||||
use super::{Detector, Findings, Region, Strength, word_near};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[
|
||||
Detector::new(
|
||||
"au-tfn",
|
||||
"Australian Tax File Number",
|
||||
Region::Australia,
|
||||
Strength::Checked,
|
||||
tfn,
|
||||
),
|
||||
Detector::new(
|
||||
"au-medicare",
|
||||
"Australian Medicare number",
|
||||
Region::Australia,
|
||||
Strength::Checked,
|
||||
medicare,
|
||||
),
|
||||
];
|
||||
|
||||
fn re(pattern: &str) -> Regex {
|
||||
Regex::new(pattern).expect("detector pattern")
|
||||
}
|
||||
|
||||
/// `NNN NNN NNN` stands alone; bare digits (eight or nine) need a word.
|
||||
static TFN: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{3})( ?)(\d{3})( ?)(\d{2,3})\b"));
|
||||
|
||||
/// Weighted sum mod 11, with the ATO's weights for 9- and 8-digit numbers.
|
||||
pub fn tfn_valid(n: &str) -> bool {
|
||||
let weights: &[u32] = match n.len() {
|
||||
9 => &[1, 4, 3, 7, 5, 8, 6, 9, 10],
|
||||
8 => &[10, 7, 8, 4, 6, 3, 5, 1],
|
||||
_ => return false,
|
||||
};
|
||||
n.bytes()
|
||||
.zip(weights)
|
||||
.map(|(b, w)| u32::from(b - b'0') * w)
|
||||
.sum::<u32>()
|
||||
% 11
|
||||
== 0
|
||||
}
|
||||
|
||||
const TFN_WORDS: &[&str] = &["tfn", "tax file number", "tax file no"];
|
||||
|
||||
fn tfn(text: &str, findings: &mut Findings) {
|
||||
for c in TFN.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
|
||||
let written = n.len() == 9 && c[2] == *" " && c[4] == *" ";
|
||||
if tfn_valid(&n) && (written || word_near(text, whole.start(), whole.end(), TFN_WORDS)) {
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// `NNNN NNNNN N` (and an optional issue number) stands alone; bare digits
|
||||
/// need a word.
|
||||
static MEDICARE: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\b([2-6]\d{3})( ?)(\d{5})( ?)(\d)(?:[ -]?\d)?\b"));
|
||||
|
||||
/// The ninth digit is the weighted sum (1, 3, 7, 9, 1, 3, 7, 9) of the first
|
||||
/// eight, mod 10.
|
||||
pub fn medicare_valid(n: &str) -> bool {
|
||||
let d: Vec<u32> = n.bytes().map(|b| u32::from(b - b'0')).collect();
|
||||
d.len() >= 9
|
||||
&& d[..8]
|
||||
.iter()
|
||||
.zip([1, 3, 7, 9, 1, 3, 7, 9])
|
||||
.map(|(a, w)| a * w)
|
||||
.sum::<u32>()
|
||||
% 10
|
||||
== d[8]
|
||||
}
|
||||
|
||||
const MEDICARE_WORDS: &[&str] = &[
|
||||
"medicare",
|
||||
"medicare card",
|
||||
"medicare no",
|
||||
"medicare number",
|
||||
];
|
||||
|
||||
fn medicare(text: &str, findings: &mut Findings) {
|
||||
for c in MEDICARE.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
|
||||
let written = c[2] == *" " && c[4] == *" ";
|
||||
if medicare_valid(&n)
|
||||
&& (written || word_near(text, whole.start(), whole.end(), MEDICARE_WORDS))
|
||||
{
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
fn count(id: &str, text: &str) -> usize {
|
||||
by_id(id).unwrap().count(text)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tax_file_numbers() {
|
||||
assert_eq!(count("au-tfn", "TFN 123 456 782"), 1);
|
||||
assert_eq!(count("au-tfn", "123 456 789"), 0);
|
||||
assert_eq!(count("au-tfn", "order 123456782"), 0);
|
||||
assert_eq!(count("au-tfn", "tax file number 123456782"), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn medicare_numbers() {
|
||||
assert_eq!(count("au-medicare", "2123 45670 1"), 1);
|
||||
assert_eq!(count("au-medicare", "2123 45671 1"), 0);
|
||||
assert_eq!(count("au-medicare", "ref 2123456701"), 0);
|
||||
assert_eq!(count("au-medicare", "Medicare 2123456701"), 1);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,66 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Canadian identifiers (§2.3): the Social Insurance Number.
|
||||
|
||||
use super::{Detector, Findings, Region, Strength, checks, word_near};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[Detector::new(
|
||||
"ca-sin",
|
||||
"Canadian Social Insurance Number",
|
||||
Region::Canada,
|
||||
Strength::Checked,
|
||||
sin,
|
||||
)];
|
||||
|
||||
/// `NNN NNN NNN` or `NNN-NNN-NNN` stands alone; nine bare digits need a word.
|
||||
static SIN: LazyLock<Regex> = LazyLock::new(|| {
|
||||
Regex::new(r"\b(\d{3})([ -]?)(\d{3})([ -]?)(\d{3})\b").expect("detector pattern")
|
||||
});
|
||||
|
||||
const SIN_WORDS: &[&str] = &[
|
||||
"sin",
|
||||
"social insurance",
|
||||
"nas",
|
||||
"numéro d'assurance sociale",
|
||||
"assurance sociale",
|
||||
];
|
||||
|
||||
fn sin(text: &str, findings: &mut Findings) {
|
||||
for c in SIN.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
|
||||
let written = !c[2].is_empty() && c[2] == c[4];
|
||||
// 0 and 8 are never issued as a first digit
|
||||
if !n.starts_with(['0', '8'])
|
||||
&& checks::luhn(&n)
|
||||
&& (written || word_near(text, whole.start(), whole.end(), SIN_WORDS))
|
||||
{
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
fn count(text: &str) -> usize {
|
||||
by_id("ca-sin").unwrap().count(text)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn social_insurance_numbers() {
|
||||
assert_eq!(count("130 692 544 and 193-456-787"), 2);
|
||||
assert_eq!(count("130 692 545"), 0);
|
||||
// The government's printed example starts with 0, never issued
|
||||
assert_eq!(count("046 454 286"), 0);
|
||||
assert_eq!(count("order 130692544"), 0);
|
||||
assert_eq!(count("SIN: 130692544"), 1);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,227 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Check-digit algorithms, each from its public definition.
|
||||
|
||||
/// The Luhn check (ISO/IEC 7812-1, Annex B) over a string of ASCII digits.
|
||||
pub fn luhn(digits: &str) -> bool {
|
||||
if digits.len() < 2 || !digits.bytes().all(|b| b.is_ascii_digit()) {
|
||||
return false;
|
||||
}
|
||||
let sum: u32 = digits
|
||||
.bytes()
|
||||
.rev()
|
||||
.enumerate()
|
||||
.map(|(i, b)| {
|
||||
let d = u32::from(b - b'0');
|
||||
if i % 2 == 1 {
|
||||
let d = d * 2;
|
||||
if d > 9 { d - 9 } else { d }
|
||||
} else {
|
||||
d
|
||||
}
|
||||
})
|
||||
.sum();
|
||||
sum.is_multiple_of(10)
|
||||
}
|
||||
|
||||
/// ISO 13616 IBAN lengths, by country, from the IBAN registry.
|
||||
const IBAN_LENGTHS: &[(&str, usize)] = &[
|
||||
("AD", 24),
|
||||
("AE", 23),
|
||||
("AL", 28),
|
||||
("AT", 20),
|
||||
("AZ", 28),
|
||||
("BA", 20),
|
||||
("BE", 16),
|
||||
("BG", 22),
|
||||
("BH", 22),
|
||||
("BI", 27),
|
||||
("BR", 29),
|
||||
("BY", 28),
|
||||
("CH", 21),
|
||||
("CR", 22),
|
||||
("CY", 28),
|
||||
("CZ", 24),
|
||||
("DE", 22),
|
||||
("DJ", 27),
|
||||
("DK", 18),
|
||||
("DO", 28),
|
||||
("EE", 20),
|
||||
("EG", 29),
|
||||
("ES", 24),
|
||||
("FI", 18),
|
||||
("FK", 18),
|
||||
("FO", 18),
|
||||
("FR", 27),
|
||||
("GB", 22),
|
||||
("GE", 22),
|
||||
("GI", 23),
|
||||
("GL", 18),
|
||||
("GR", 27),
|
||||
("GT", 28),
|
||||
("HN", 28),
|
||||
("HR", 21),
|
||||
("HU", 28),
|
||||
("IE", 22),
|
||||
("IL", 23),
|
||||
("IQ", 23),
|
||||
("IS", 26),
|
||||
("IT", 27),
|
||||
("JO", 30),
|
||||
("KW", 30),
|
||||
("KZ", 20),
|
||||
("LB", 28),
|
||||
("LC", 32),
|
||||
("LI", 21),
|
||||
("LT", 20),
|
||||
("LU", 20),
|
||||
("LV", 21),
|
||||
("LY", 25),
|
||||
("MC", 27),
|
||||
("MD", 24),
|
||||
("ME", 22),
|
||||
("MK", 19),
|
||||
("MN", 20),
|
||||
("MR", 27),
|
||||
("MT", 31),
|
||||
("MU", 30),
|
||||
("NI", 28),
|
||||
("NL", 18),
|
||||
("NO", 15),
|
||||
("OM", 23),
|
||||
("PK", 24),
|
||||
("PL", 28),
|
||||
("PS", 29),
|
||||
("PT", 25),
|
||||
("QA", 29),
|
||||
("RO", 24),
|
||||
("RS", 22),
|
||||
("RU", 33),
|
||||
("SA", 24),
|
||||
("SC", 31),
|
||||
("SD", 18),
|
||||
("SE", 24),
|
||||
("SI", 19),
|
||||
("SK", 24),
|
||||
("SM", 27),
|
||||
("SO", 23),
|
||||
("ST", 25),
|
||||
("SV", 28),
|
||||
("TL", 23),
|
||||
("TN", 24),
|
||||
("TR", 26),
|
||||
("UA", 29),
|
||||
("VA", 22),
|
||||
("VG", 24),
|
||||
("XK", 20),
|
||||
("YE", 30),
|
||||
];
|
||||
|
||||
/// The IBAN length for a country code, if the country uses IBANs.
|
||||
pub fn iban_length(country: &str) -> Option<usize> {
|
||||
IBAN_LENGTHS
|
||||
.iter()
|
||||
.find(|(code, _)| *code == country)
|
||||
.map(|(_, len)| *len)
|
||||
}
|
||||
|
||||
/// ISO 13616 / ISO 7064 MOD 97-10 over an IBAN with no spaces, upper case:
|
||||
/// move the first four characters to the end, turn letters into 10–35, and
|
||||
/// the number mod 97 must be 1. Also checks the country's length.
|
||||
pub fn iban(iban: &str) -> bool {
|
||||
if iban.len() < 5
|
||||
|| !iban
|
||||
.bytes()
|
||||
.all(|b| b.is_ascii_uppercase() || b.is_ascii_digit())
|
||||
{
|
||||
return false;
|
||||
}
|
||||
if iban_length(&iban[..2]) != Some(iban.len())
|
||||
|| !iban[2..4].bytes().all(|b| b.is_ascii_digit())
|
||||
{
|
||||
return false;
|
||||
}
|
||||
let mut remainder: u32 = 0;
|
||||
for b in iban[4..].bytes().chain(iban[..4].bytes()) {
|
||||
let value = if b.is_ascii_digit() {
|
||||
u32::from(b - b'0')
|
||||
} else {
|
||||
u32::from(b - b'A') + 10
|
||||
};
|
||||
remainder = if value >= 10 {
|
||||
(remainder * 100 + value) % 97
|
||||
} else {
|
||||
(remainder * 10 + value) % 97
|
||||
};
|
||||
}
|
||||
remainder == 1
|
||||
}
|
||||
|
||||
/// ISO 3166-1 alpha-2 country codes, for SWIFT/BIC positions 5–6.
|
||||
const COUNTRIES: &str = "AD AE AF AG AI AL AM AO AQ AR AS AT AU AW AX AZ BA BB BD BE BF BG BH BI BJ \
|
||||
BL BM BN BO BQ BR BS BT BV BW BY BZ CA CC CD CF CG CH CI CK CL CM CN CO CR CU CV CW CX CY CZ DE DJ \
|
||||
DK DM DO DZ EC EE EG EH ER ES ET FI FJ FK FM FO FR GA GB GD GE GF GG GH GI GL GM GN GP GQ GR GS GT \
|
||||
GU GW GY HK HM HN HR HT HU ID IE IL IM IN IO IQ IR IS IT JE JM JO JP KE KG KH KI KM KN KP KR KW KY \
|
||||
KZ LA LB LC LI LK LR LS LT LU LV LY MA MC MD ME MF MG MH MK ML MM MN MO MP MQ MR MS MT MU MV MW MX \
|
||||
MY MZ NA NC NE NF NG NI NL NO NP NR NU NZ OM PA PE PF PG PH PK PL PM PN PR PS PT PW PY QA RE RO RS \
|
||||
RU RW SA SB SC SD SE SG SH SI SJ SK SL SM SN SO SR SS ST SV SX SY SZ TC TD TF TG TH TJ TK TL TM TN \
|
||||
TO TR TT TV TW TZ UA UG UM US UY UZ VA VC VE VG VI VN VU WF WS XK YE YT ZA ZM ZW";
|
||||
|
||||
pub fn is_country(code: &str) -> bool {
|
||||
code.len() == 2 && COUNTRIES.split(' ').any(|c| c == code)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn luhn_known_numbers() {
|
||||
// Published test card numbers
|
||||
for good in [
|
||||
"4242424242424242",
|
||||
"5555555555554444",
|
||||
"378282246310005",
|
||||
"79927398713",
|
||||
] {
|
||||
assert!(luhn(good), "{good}");
|
||||
}
|
||||
for bad in ["4242424242424241", "79927398710", "1", "12a4"] {
|
||||
assert!(!luhn(bad), "{bad}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn iban_registry_examples() {
|
||||
// The IBAN registry's own examples
|
||||
for good in [
|
||||
"GB29NWBK60161331926819",
|
||||
"DE89370400440532013000",
|
||||
"FR1420041010050500013M02606",
|
||||
"NL91ABNA0417164300",
|
||||
"BE68539007547034",
|
||||
"NO9386011117947",
|
||||
"CH9300762011623852957",
|
||||
] {
|
||||
assert!(iban(good), "{good}");
|
||||
}
|
||||
for bad in [
|
||||
"GB29NWBK60161331926818", // check fails
|
||||
"GB29NWBK6016133192681", // too short for GB
|
||||
"ZZ29NWBK60161331926819", // no such country
|
||||
"DE8937040044053201300A", // letters where DE has none still fail mod 97
|
||||
] {
|
||||
assert!(!iban(bad), "{bad}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn countries() {
|
||||
assert!(is_country("DE") && is_country("US") && is_country("XK"));
|
||||
assert!(!is_country("ZZ") && !is_country("D"));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,646 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! European Union national identifiers (§2.3), each from its issuer's
|
||||
//! published rules. An identifier that is only digits and whose check a
|
||||
//! random number passes often (mod 10, mod 11) counts alone only in its
|
||||
//! written form, and as bare digits only beside a word.
|
||||
|
||||
use super::{
|
||||
Detector, Findings, Region, Strength, checks, digit_values, stands_alone, valid_short_date,
|
||||
word_near,
|
||||
};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[
|
||||
Detector::new(
|
||||
"de-tax-id",
|
||||
"Germany: tax ID (Steuer-ID)",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
de_tax_id,
|
||||
),
|
||||
Detector::new(
|
||||
"de-id-card",
|
||||
"Germany: ID card number",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
de_id_card,
|
||||
),
|
||||
Detector::new(
|
||||
"fr-nir",
|
||||
"France: social security number (NIR)",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
fr_nir,
|
||||
),
|
||||
Detector::new(
|
||||
"es-dni-nie",
|
||||
"Spain: DNI and NIE",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
es_dni_nie,
|
||||
),
|
||||
Detector::new(
|
||||
"it-codice-fiscale",
|
||||
"Italy: codice fiscale",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
it_codice_fiscale,
|
||||
),
|
||||
Detector::new(
|
||||
"nl-bsn",
|
||||
"Netherlands: BSN",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
nl_bsn,
|
||||
),
|
||||
Detector::new(
|
||||
"be-national-number",
|
||||
"Belgium: national number",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
be_national_number,
|
||||
),
|
||||
Detector::new(
|
||||
"pl-pesel",
|
||||
"Poland: PESEL",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
pl_pesel,
|
||||
),
|
||||
Detector::new(
|
||||
"se-personnummer",
|
||||
"Sweden: personnummer",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
se_personnummer,
|
||||
),
|
||||
Detector::new(
|
||||
"dk-cpr",
|
||||
"Denmark: CPR number",
|
||||
Region::Eu,
|
||||
Strength::NeedsWord,
|
||||
dk_cpr,
|
||||
),
|
||||
Detector::new(
|
||||
"fi-hetu",
|
||||
"Finland: personal identity code",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
fi_hetu,
|
||||
),
|
||||
Detector::new(
|
||||
"ie-pps",
|
||||
"Ireland: PPS number",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
ie_pps,
|
||||
),
|
||||
Detector::new(
|
||||
"pt-nif",
|
||||
"Portugal: NIF",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
pt_nif,
|
||||
),
|
||||
Detector::new(
|
||||
"at-svnr",
|
||||
"Austria: social insurance number",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
at_svnr,
|
||||
),
|
||||
];
|
||||
|
||||
fn re(pattern: &str) -> Regex {
|
||||
Regex::new(pattern).expect("detector pattern")
|
||||
}
|
||||
|
||||
fn num(s: &str) -> u32 {
|
||||
s.parse().unwrap_or(0)
|
||||
}
|
||||
|
||||
// --- Germany --------------------------------------------------------------
|
||||
|
||||
/// Eleven digits, written `86 095 742 719` on the BZSt's letters.
|
||||
static DE_TAX: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{2}( ?)\d{3}( ?)\d{3}( ?)\d{3}\b"));
|
||||
|
||||
/// ISO 7064 MOD 11,10; no leading zero; in the first ten digits one digit
|
||||
/// appears two or three times and every other at most once.
|
||||
pub fn de_tax_id_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
if d.len() != 11 || d[0] == 0 {
|
||||
return false;
|
||||
}
|
||||
let mut counts = [0u8; 10];
|
||||
for &x in &d[..10] {
|
||||
counts[x as usize] += 1;
|
||||
}
|
||||
let repeated = counts.iter().filter(|&&c| c >= 2).count();
|
||||
if repeated != 1 || counts.iter().any(|&c| c > 3) {
|
||||
return false;
|
||||
}
|
||||
let mut product = 10;
|
||||
for &x in &d[..10] {
|
||||
let mut sum = (x + product) % 10;
|
||||
if sum == 0 {
|
||||
sum = 10;
|
||||
}
|
||||
product = (2 * sum) % 11;
|
||||
}
|
||||
let check = match 11 - product {
|
||||
10 => 0,
|
||||
c => c,
|
||||
};
|
||||
check == d[10]
|
||||
}
|
||||
|
||||
const DE_TAX_WORDS: &[&str] = &[
|
||||
"steuer-id",
|
||||
"steueridentifikationsnummer",
|
||||
"steuerliche identifikationsnummer",
|
||||
"idnr",
|
||||
"identifikationsnummer",
|
||||
"tax id",
|
||||
];
|
||||
|
||||
fn de_tax_id(text: &str, findings: &mut Findings) {
|
||||
for c in DE_TAX.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let written = [&c[1], &c[2], &c[3]].iter().all(|s| *s == " ");
|
||||
let n: String = whole.as_str().replace(' ', "");
|
||||
if de_tax_id_valid(&n)
|
||||
&& (written || word_near(text, whole.start(), whole.end(), DE_TAX_WORDS))
|
||||
{
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The ID card's document number: a letter from the card's alphabet, eight
|
||||
/// more characters from it, then the check digit.
|
||||
static DE_ID: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\b[CFGHJKLMNPRTVWXYZ][CFGHJKLMNPRTVWXYZ0-9]{8}\d\b"));
|
||||
|
||||
/// ICAO 9303 check digit: weights 7, 3, 1; letters A=10 … Z=35.
|
||||
pub fn icao_check(chars: &str, check: u32) -> bool {
|
||||
let value = |c: char| c.to_digit(10).unwrap_or_else(|| c as u32 - 'A' as u32 + 10);
|
||||
let sum: u32 = chars
|
||||
.chars()
|
||||
.zip([7, 3, 1].iter().cycle())
|
||||
.map(|(c, w)| value(c) * w)
|
||||
.sum();
|
||||
sum % 10 == check
|
||||
}
|
||||
|
||||
fn de_id_card(text: &str, findings: &mut Findings) {
|
||||
for m in DE_ID.find_iter(text) {
|
||||
let s = m.as_str();
|
||||
if icao_check(&s[..9], num(&s[9..])) {
|
||||
findings.insert(s);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- France ---------------------------------------------------------------
|
||||
|
||||
/// Sex, year, month, department (with Corsica's 2A and 2B), commune, order,
|
||||
/// then the two-digit key, spaces allowed between groups.
|
||||
static FR_NIR: LazyLock<Regex> = LazyLock::new(|| {
|
||||
re(r"\b([1-478]) ?(\d{2}) ?(\d{2}) ?(\d{2}|2[AB]) ?(\d{3}) ?(\d{3}) ?(\d{2})\b")
|
||||
});
|
||||
|
||||
fn fr_nir(text: &str, findings: &mut Findings) {
|
||||
for c in FR_NIR.captures_iter(text) {
|
||||
let month = num(&c[3]);
|
||||
if !(matches!(month, 1..=12 | 20..=42 | 50..=99)) {
|
||||
continue;
|
||||
}
|
||||
let department = match &c[4] {
|
||||
"2A" => "19",
|
||||
"2B" => "18",
|
||||
d => d,
|
||||
};
|
||||
let body = format!(
|
||||
"{}{}{}{}{}{}",
|
||||
&c[1], &c[2], &c[3], department, &c[5], &c[6]
|
||||
);
|
||||
let Ok(value) = body.parse::<u64>() else {
|
||||
continue;
|
||||
};
|
||||
if 97 - value % 97 == u64::from(num(&c[7])) {
|
||||
findings.insert(format!(
|
||||
"{}{}{}{}{}{}{}",
|
||||
&c[1], &c[2], &c[3], &c[4], &c[5], &c[6], &c[7]
|
||||
));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Spain ----------------------------------------------------------------
|
||||
|
||||
static ES_ID: LazyLock<Regex> = LazyLock::new(|| re(r"(?i)\b([XYZ]?)[ -]?(\d{7,8})[ -]?([A-Z])\b"));
|
||||
|
||||
const DNI_LETTERS: &[u8] = b"TRWAGMYFPDXBNJZSQVHLCKE";
|
||||
|
||||
fn es_dni_nie(text: &str, findings: &mut Findings) {
|
||||
for c in ES_ID.captures_iter(text) {
|
||||
let prefix = c[1].to_ascii_uppercase();
|
||||
let digits = &c[2];
|
||||
// DNI: eight digits; NIE: X, Y or Z and seven digits
|
||||
let number = match (prefix.as_str(), digits.len()) {
|
||||
("", 8) => digits.to_string(),
|
||||
("X", 7) => format!("0{digits}"),
|
||||
("Y", 7) => format!("1{digits}"),
|
||||
("Z", 7) => format!("2{digits}"),
|
||||
_ => continue,
|
||||
};
|
||||
let letter = c[3].to_ascii_uppercase();
|
||||
if DNI_LETTERS[(num(&number) % 23) as usize] == letter.as_bytes()[0] {
|
||||
findings.insert(format!("{prefix}{digits}{letter}"));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Italy ----------------------------------------------------------------
|
||||
|
||||
/// Surname and name letters, year, month letter, day, place code, check
|
||||
/// letter; digits may be replaced by letters (omocodia).
|
||||
static IT_CF: LazyLock<Regex> = LazyLock::new(|| {
|
||||
let d = "[0-9LMNPQRSTUV]";
|
||||
re(&format!(
|
||||
r"(?i)\b[A-Z]{{6}}{d}{{2}}[ABCDEHLMPRST]{d}{{2}}[A-Z]{d}{{3}}[A-Z]\b"
|
||||
))
|
||||
});
|
||||
|
||||
/// The Ministry's odd-position values for 0–9 and A–Z.
|
||||
const CF_ODD: [u32; 36] = [
|
||||
1, 0, 5, 7, 9, 13, 15, 17, 19, 21, // 0-9
|
||||
1, 0, 5, 7, 9, 13, 15, 17, 19, 21, 2, 4, 18, 20, 11, 3, 6, 8, 12, 14, 16, 10, 22, 25, 24,
|
||||
23, // A-Z
|
||||
];
|
||||
|
||||
pub fn codice_fiscale_valid(cf: &str) -> bool {
|
||||
let index = |c: u8| {
|
||||
if c.is_ascii_digit() {
|
||||
(c - b'0') as usize
|
||||
} else {
|
||||
(c - b'A') as usize + 10
|
||||
}
|
||||
};
|
||||
let even = |c: u8| {
|
||||
if c.is_ascii_digit() {
|
||||
u32::from(c - b'0')
|
||||
} else {
|
||||
u32::from(c - b'A')
|
||||
}
|
||||
};
|
||||
let bytes = cf.as_bytes();
|
||||
let sum: u32 = bytes[..15]
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, &c)| {
|
||||
if i % 2 == 0 {
|
||||
CF_ODD[index(c)]
|
||||
} else {
|
||||
even(c)
|
||||
}
|
||||
})
|
||||
.sum();
|
||||
u32::from(bytes[15] - b'A') == sum % 26
|
||||
}
|
||||
|
||||
fn it_codice_fiscale(text: &str, findings: &mut Findings) {
|
||||
for m in IT_CF.find_iter(text) {
|
||||
let cf = m.as_str().to_ascii_uppercase();
|
||||
if codice_fiscale_valid(&cf) {
|
||||
findings.insert(cf);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Netherlands ----------------------------------------------------------
|
||||
|
||||
/// Nine digits, sometimes written `1112.22.333`.
|
||||
static NL_BSN: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{4})(\.?)(\d{2})(\.?)(\d{3})\b"));
|
||||
|
||||
/// The eleven test: weights 9 down to 2, and −1 for the last digit.
|
||||
pub fn bsn_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
let sum: i64 = d[..8]
|
||||
.iter()
|
||||
.zip((2..=9).rev())
|
||||
.map(|(a, w)| i64::from(a * w))
|
||||
.sum::<i64>()
|
||||
- i64::from(d[8]);
|
||||
sum != 0 && sum % 11 == 0
|
||||
}
|
||||
|
||||
const BSN_WORDS: &[&str] = &[
|
||||
"bsn",
|
||||
"burgerservicenummer",
|
||||
"sofinummer",
|
||||
"sofi-nummer",
|
||||
"citizen service number",
|
||||
];
|
||||
|
||||
fn nl_bsn(text: &str, findings: &mut Findings) {
|
||||
for c in NL_BSN.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
|
||||
let written = &c[2] == "." && &c[4] == ".";
|
||||
if bsn_valid(&n) && (written || word_near(text, whole.start(), whole.end(), BSN_WORDS)) {
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Belgium --------------------------------------------------------------
|
||||
|
||||
/// `YY.MM.DD-XXX.CC` or eleven digits.
|
||||
static BE_NN: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\b(\d{2})\.?(\d{2})\.?(\d{2})-?(\d{3})\.?(\d{2})\b"));
|
||||
|
||||
fn be_national_number(text: &str, findings: &mut Findings) {
|
||||
for c in BE_NN.captures_iter(text) {
|
||||
let (month, day) = (num(&c[2]), num(&c[3]));
|
||||
// Month 0 and day 0 mean unknown; bis numbers add 20 or 40 to the month
|
||||
if !(month <= 12 || (20..=32).contains(&month) || (40..=52).contains(&month)) || day > 31 {
|
||||
continue;
|
||||
}
|
||||
let body = format!("{}{}{}{}", &c[1], &c[2], &c[3], &c[4]);
|
||||
let check = u64::from(num(&c[5]));
|
||||
let before_2000 = 97 - body.parse::<u64>().unwrap_or(0) % 97;
|
||||
let since_2000 = 97 - format!("2{body}").parse::<u64>().unwrap_or(0) % 97;
|
||||
if check == before_2000 || check == since_2000 {
|
||||
findings.insert(format!("{body}{}", &c[5]));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Poland ---------------------------------------------------------------
|
||||
|
||||
static ELEVEN: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{11}\b"));
|
||||
|
||||
/// Weights 1, 3, 7, 9 repeating; the birth date encodes the century in the
|
||||
/// month (+80 for the 1800s, +20 for the 2000s, and so on).
|
||||
pub fn pesel_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
let sum: u32 = d[..10]
|
||||
.iter()
|
||||
.zip([1, 3, 7, 9].iter().cycle())
|
||||
.map(|(a, w)| a * w)
|
||||
.sum();
|
||||
let month = d[2] * 10 + d[3];
|
||||
let (century, month) = match month {
|
||||
81..=92 => (1800, month - 80),
|
||||
1..=12 => (1900, month),
|
||||
21..=32 => (2000, month - 20),
|
||||
41..=52 => (2100, month - 40),
|
||||
_ => return false,
|
||||
};
|
||||
let year = century + d[0] * 10 + d[1];
|
||||
(10 - sum % 10) % 10 == d[10] && (1..=super::days_in(year, month)).contains(&(d[4] * 10 + d[5]))
|
||||
}
|
||||
|
||||
const PESEL_WORDS: &[&str] = &["pesel", "numer pesel", "nr pesel"];
|
||||
|
||||
fn pl_pesel(text: &str, findings: &mut Findings) {
|
||||
for m in ELEVEN.find_iter(text) {
|
||||
if pesel_valid(m.as_str()) && word_near(text, m.start(), m.end(), PESEL_WORDS) {
|
||||
findings.insert(m.as_str());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Sweden ---------------------------------------------------------------
|
||||
|
||||
/// `YYMMDD-NNNN`, `YYYYMMDD-NNNN` (`+` after 100), or the bare digits.
|
||||
static SE_PNR: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\b(?:\d{2})?(\d{2})(\d{2})(\d{2})([-+]?)(\d{4})\b"));
|
||||
|
||||
const SE_WORDS: &[&str] = &[
|
||||
"personnummer",
|
||||
"personnr",
|
||||
"person nr",
|
||||
"samordningsnummer",
|
||||
"pnr",
|
||||
];
|
||||
|
||||
fn se_personnummer(text: &str, findings: &mut Findings) {
|
||||
for c in SE_PNR.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let (yy, month, day) = (num(&c[1]), num(&c[2]), num(&c[3]));
|
||||
// Coordination numbers add 60 to the day
|
||||
let day = if day > 60 { day - 60 } else { day };
|
||||
let ten = format!("{}{}{}{}", &c[1], &c[2], &c[3], &c[5]);
|
||||
let written = !c[4].is_empty();
|
||||
if valid_short_date(yy, month, day)
|
||||
&& checks::luhn(&ten)
|
||||
&& (written || word_near(text, whole.start(), whole.end(), SE_WORDS))
|
||||
{
|
||||
findings.insert(ten);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Denmark --------------------------------------------------------------
|
||||
|
||||
static DK_CPR: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{2})(\d{2})(\d{2})-?(\d{4})\b"));
|
||||
|
||||
const CPR_WORDS: &[&str] = &["cpr", "cpr-nr", "cpr nr", "cpr-nummer", "personnummer"];
|
||||
|
||||
fn dk_cpr(text: &str, findings: &mut Findings) {
|
||||
for c in DK_CPR.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
if valid_short_date(num(&c[3]), num(&c[2]), num(&c[1]))
|
||||
&& word_near(text, whole.start(), whole.end(), CPR_WORDS)
|
||||
{
|
||||
findings.insert(whole.as_str().replace('-', ""));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Finland --------------------------------------------------------------
|
||||
|
||||
static FI_HETU: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"(?i)\b(\d{2})(\d{2})(\d{2})[-+ABCDEFYXWVU](\d{3})([0-9A-Y])\b"));
|
||||
|
||||
const HETU_CHECK: &[u8] = b"0123456789ABCDEFHJKLMNPRSTUVWXY";
|
||||
|
||||
fn fi_hetu(text: &str, findings: &mut Findings) {
|
||||
for c in FI_HETU.captures_iter(text) {
|
||||
let (day, month, yy) = (num(&c[1]), num(&c[2]), num(&c[3]));
|
||||
let n: u64 = format!("{}{}{}{}", &c[1], &c[2], &c[3], &c[4])
|
||||
.parse()
|
||||
.unwrap_or(0);
|
||||
let check = c[5].to_ascii_uppercase().as_bytes()[0];
|
||||
if valid_short_date(yy, month, day) && HETU_CHECK[(n % 31) as usize] == check {
|
||||
findings.insert(c[0].to_ascii_uppercase());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Ireland --------------------------------------------------------------
|
||||
|
||||
static IE_PPS: LazyLock<Regex> = LazyLock::new(|| re(r"(?i)\b(\d{7})([A-W])([ABHW]?)\b"));
|
||||
|
||||
const PPS_CHECK: &[u8] = b"WABCDEFGHIJKLMNOPQRSTUV";
|
||||
|
||||
fn ie_pps(text: &str, findings: &mut Findings) {
|
||||
for c in IE_PPS.captures_iter(text) {
|
||||
let mut sum: u32 = digit_values(&c[1])
|
||||
.iter()
|
||||
.zip((2..=8).rev())
|
||||
.map(|(a, w)| a * w)
|
||||
.sum();
|
||||
// The second letter counts, times 9; W (the old form) counts as 0
|
||||
let second = c[3].to_ascii_uppercase();
|
||||
if let Some(&letter) = second.as_bytes().first()
|
||||
&& letter != b'W'
|
||||
{
|
||||
sum += u32::from(letter - b'A' + 1) * 9;
|
||||
}
|
||||
let check = c[2].to_ascii_uppercase().as_bytes()[0];
|
||||
if PPS_CHECK[(sum % 23) as usize] == check {
|
||||
findings.insert(c[0].to_ascii_uppercase());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Portugal -------------------------------------------------------------
|
||||
|
||||
static NINE: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{9}\b"));
|
||||
|
||||
/// Mod 11 over weights 9 down to 2; a check of 10 or 11 becomes 0.
|
||||
pub fn nif_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
let sum: u32 = d[..8].iter().zip((2..=9).rev()).map(|(a, w)| a * w).sum();
|
||||
let check = match 11 - sum % 11 {
|
||||
10 | 11 => 0,
|
||||
c => c,
|
||||
};
|
||||
matches!(d[0], 1 | 2 | 3 | 5 | 6 | 8 | 9) && check == d[8]
|
||||
}
|
||||
|
||||
const NIF_WORDS: &[&str] = &[
|
||||
"nif",
|
||||
"contribuinte",
|
||||
"número de identificação fiscal",
|
||||
"numero de contribuinte",
|
||||
];
|
||||
|
||||
fn pt_nif(text: &str, findings: &mut Findings) {
|
||||
for m in NINE.find_iter(text) {
|
||||
if nif_valid(m.as_str()) && word_near(text, m.start(), m.end(), NIF_WORDS) {
|
||||
findings.insert(m.as_str());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Austria --------------------------------------------------------------
|
||||
|
||||
/// A serial and check digit, then the birth date: `1237 010180`.
|
||||
static AT_SVNR: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{3})(\d)( ?)(\d{2})(\d{2})(\d{2})\b"));
|
||||
|
||||
const SVNR_WORDS: &[&str] = &[
|
||||
"sozialversicherungsnummer",
|
||||
"svnr",
|
||||
"sv-nr",
|
||||
"sv-nummer",
|
||||
"versicherungsnummer",
|
||||
];
|
||||
|
||||
fn at_svnr(text: &str, findings: &mut Findings) {
|
||||
for c in AT_SVNR.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let n = format!("{}{}{}{}{}", &c[1], &c[2], &c[4], &c[5], &c[6]);
|
||||
let d = digit_values(&n);
|
||||
let sum: u32 = d
|
||||
.iter()
|
||||
.zip([3, 7, 9, 0, 5, 8, 4, 2, 1, 6])
|
||||
.map(|(a, w)| a * w)
|
||||
.sum();
|
||||
let written = &c[3] == " ";
|
||||
if d[0] != 0
|
||||
&& sum % 11 == d[3]
|
||||
&& valid_short_date(num(&c[6]), num(&c[5]), num(&c[4]))
|
||||
&& (written || word_near(text, whole.start(), whole.end(), SVNR_WORDS))
|
||||
&& stands_alone(text, whole.start(), whole.end())
|
||||
{
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
fn count(id: &str, text: &str) -> usize {
|
||||
by_id(id).unwrap().count(text)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn germany() {
|
||||
assert_eq!(count("de-tax-id", "86 095 742 719"), 1);
|
||||
assert_eq!(count("de-tax-id", "Steuer-ID: 86095742719"), 1);
|
||||
assert_eq!(count("de-tax-id", "Rechnung 86095742719"), 0);
|
||||
assert_eq!(count("de-tax-id", "86 095 742 718"), 0);
|
||||
// ICAO 9303's German specimen card
|
||||
assert_eq!(count("de-id-card", "Ausweis T220001293"), 1);
|
||||
assert_eq!(count("de-id-card", "T220001294"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn france_spain_italy() {
|
||||
assert_eq!(count("fr-nir", "2 55 08 14 168 025 38"), 1);
|
||||
assert_eq!(count("fr-nir", "255081416802539"), 0);
|
||||
assert_eq!(count("es-dni-nie", "DNI 12345678Z, NIE X-1234567-L"), 2);
|
||||
assert_eq!(count("es-dni-nie", "12345678A"), 0);
|
||||
assert_eq!(count("it-codice-fiscale", "CF: RSSMRA85T10A562S"), 1);
|
||||
assert_eq!(count("it-codice-fiscale", "RSSMRA85T10A562T"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn benelux() {
|
||||
assert_eq!(count("nl-bsn", "1112.22.333"), 1);
|
||||
assert_eq!(count("nl-bsn", "BSN 111222333"), 1);
|
||||
assert_eq!(count("nl-bsn", "order 111222333"), 0);
|
||||
assert_eq!(count("nl-bsn", "BSN 111222334"), 0);
|
||||
assert_eq!(count("be-national-number", "85.07.30-033.28"), 1);
|
||||
assert_eq!(count("be-national-number", "85073003329"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn nordics() {
|
||||
assert_eq!(count("se-personnummer", "811218-9876"), 1);
|
||||
assert_eq!(count("se-personnummer", "811218-9875"), 0);
|
||||
assert_eq!(count("se-personnummer", "order 8112189876"), 0);
|
||||
assert_eq!(count("se-personnummer", "personnummer 198112189876"), 1);
|
||||
assert_eq!(count("dk-cpr", "CPR-nr: 010170-1234"), 1);
|
||||
assert_eq!(count("dk-cpr", "010170-1234"), 0);
|
||||
assert_eq!(count("dk-cpr", "CPR 320170-1234"), 0);
|
||||
assert_eq!(count("fi-hetu", "131052-308T"), 1);
|
||||
assert_eq!(count("fi-hetu", "131052-308U"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn poland_ireland_portugal_austria() {
|
||||
assert_eq!(count("pl-pesel", "PESEL 44051401359, pesel 02070803628"), 2);
|
||||
assert_eq!(count("pl-pesel", "PESEL 44051401358"), 0);
|
||||
assert_eq!(count("pl-pesel", "44051401359"), 0);
|
||||
assert_eq!(count("ie-pps", "PPS 1234567T and 1234567FA"), 2);
|
||||
assert_eq!(count("ie-pps", "1234567U"), 0);
|
||||
assert_eq!(count("pt-nif", "NIF 123456789"), 1);
|
||||
assert_eq!(count("pt-nif", "NIF 123456788"), 0);
|
||||
assert_eq!(count("at-svnr", "1237 010180"), 1);
|
||||
assert_eq!(count("at-svnr", "SVNR 1237010180"), 1);
|
||||
assert_eq!(count("at-svnr", "1238 010180"), 0);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,116 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! European identifiers outside the EU (§2.3): Norway's national identity
|
||||
//! number and Switzerland's AHV number.
|
||||
|
||||
use super::{Detector, Findings, Region, Strength, digit_values, valid_short_date};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[
|
||||
Detector::new(
|
||||
"no-fnr",
|
||||
"Norway: national identity number",
|
||||
Region::Europe,
|
||||
Strength::Checked,
|
||||
no_fnr,
|
||||
),
|
||||
Detector::new(
|
||||
"ch-ahv",
|
||||
"Switzerland: AHV number",
|
||||
Region::Europe,
|
||||
Strength::Checked,
|
||||
ch_ahv,
|
||||
),
|
||||
];
|
||||
|
||||
static ELEVEN: LazyLock<Regex> =
|
||||
LazyLock::new(|| Regex::new(r"\b\d{6} ?\d{5}\b").expect("detector pattern"));
|
||||
|
||||
/// Two mod 11 check digits over a birth date (D-numbers add 40 to the day,
|
||||
/// H-numbers 40 to the month): strong enough to count alone.
|
||||
pub fn fnr_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
if d.len() != 11 {
|
||||
return false;
|
||||
}
|
||||
let check =
|
||||
|weights: &[u32]| match 11 - d.iter().zip(weights).map(|(a, w)| a * w).sum::<u32>() % 11 {
|
||||
11 => Some(0),
|
||||
10 => None,
|
||||
c => Some(c),
|
||||
};
|
||||
let day = d[0] * 10 + d[1];
|
||||
let month = d[2] * 10 + d[3];
|
||||
let day = if day > 40 { day - 40 } else { day };
|
||||
let month = if month > 40 { month - 40 } else { month };
|
||||
valid_short_date(d[4] * 10 + d[5], month, day)
|
||||
&& check(&[3, 7, 6, 1, 8, 9, 4, 5, 2]) == Some(d[9])
|
||||
&& check(&[5, 4, 3, 2, 7, 6, 5, 4, 3, 2]) == Some(d[10])
|
||||
}
|
||||
|
||||
fn no_fnr(text: &str, findings: &mut Findings) {
|
||||
for m in ELEVEN.find_iter(text) {
|
||||
let n = m.as_str().replace(' ', "");
|
||||
if fnr_valid(&n) {
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// `756.1234.5678.97`: the country prefix, then an EAN-13 check digit.
|
||||
static AHV: LazyLock<Regex> = LazyLock::new(|| {
|
||||
Regex::new(r"\b756[. ]?\d{4}[. ]?\d{4}[. ]?\d{2}\b").expect("detector pattern")
|
||||
});
|
||||
|
||||
pub fn ean13_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
if d.len() != 13 {
|
||||
return false;
|
||||
}
|
||||
let sum: u32 = d[..12]
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, x)| if i % 2 == 0 { *x } else { x * 3 })
|
||||
.sum();
|
||||
(10 - sum % 10) % 10 == d[12]
|
||||
}
|
||||
|
||||
fn ch_ahv(text: &str, findings: &mut Findings) {
|
||||
for m in AHV.find_iter(text) {
|
||||
let n: String = m.as_str().chars().filter(char::is_ascii_digit).collect();
|
||||
if ean13_valid(&n) {
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
fn count(id: &str, text: &str) -> usize {
|
||||
by_id(id).unwrap().count(text)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn norway() {
|
||||
assert_eq!(count("no-fnr", "01019000083"), 1);
|
||||
assert_eq!(count("no-fnr", "010190 00083"), 1);
|
||||
assert_eq!(count("no-fnr", "01019000084"), 0);
|
||||
// Not a date
|
||||
assert_eq!(count("no-fnr", "32019000083"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn switzerland() {
|
||||
// The federal example
|
||||
assert_eq!(count("ch-ahv", "AHV 756.9217.0769.85"), 1);
|
||||
assert_eq!(count("ch-ahv", "7569217076985"), 1);
|
||||
assert_eq!(count("ch-ahv", "756.9217.0769.86"), 0);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,278 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Detectors (dlp-and-mail-flow-rules spec, §2.3): each finds one kind of
|
||||
//! identifier in text and reports the distinct ones it found.
|
||||
//!
|
||||
//! A detector is one of two strengths:
|
||||
//!
|
||||
//! - **Checked**: the identifier carries a published check digit or
|
||||
//! checksum, so a random number rarely passes; found on its own.
|
||||
//! - **Needs a word**: the format alone is too common, so a candidate counts
|
||||
//! only with a corroborating word within [`WINDOW`] characters either
|
||||
//! side.
|
||||
//!
|
||||
//! Findings are distinct normalized values (digits only, upper case), so the
|
||||
//! same card number pasted twice counts once. They stay in memory: callers
|
||||
//! read only [`Findings::len`].
|
||||
|
||||
pub mod africa;
|
||||
pub mod americas;
|
||||
pub mod any;
|
||||
pub mod asia;
|
||||
pub mod australia;
|
||||
pub mod canada;
|
||||
pub mod checks;
|
||||
pub mod eu;
|
||||
pub mod europe;
|
||||
pub mod templates;
|
||||
pub mod uk;
|
||||
pub mod us;
|
||||
|
||||
use ahash::AHashSet;
|
||||
|
||||
/// How far, in characters, a corroborating word may be from a candidate.
|
||||
pub const WINDOW: usize = 50;
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum Strength {
|
||||
Checked,
|
||||
NeedsWord,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum Region {
|
||||
Any,
|
||||
Us,
|
||||
Uk,
|
||||
Canada,
|
||||
Australia,
|
||||
Eu,
|
||||
Europe,
|
||||
Asia,
|
||||
Americas,
|
||||
Africa,
|
||||
}
|
||||
|
||||
/// The distinct values one detector found.
|
||||
#[derive(Debug, Default)]
|
||||
pub struct Findings(AHashSet<String>);
|
||||
|
||||
impl Findings {
|
||||
pub fn insert(&mut self, value: impl Into<String>) {
|
||||
self.0.insert(value.into());
|
||||
}
|
||||
|
||||
pub fn len(&self) -> usize {
|
||||
self.0.len()
|
||||
}
|
||||
|
||||
pub fn is_empty(&self) -> bool {
|
||||
self.0.is_empty()
|
||||
}
|
||||
}
|
||||
|
||||
pub struct Detector {
|
||||
/// Stable id, stored in rules: `payment-card`, `iban`, `us-ssn`.
|
||||
pub id: &'static str,
|
||||
pub name: &'static str,
|
||||
pub region: Region,
|
||||
pub strength: Strength,
|
||||
find: fn(&str, &mut Findings),
|
||||
}
|
||||
|
||||
impl Detector {
|
||||
pub const fn new(
|
||||
id: &'static str,
|
||||
name: &'static str,
|
||||
region: Region,
|
||||
strength: Strength,
|
||||
find: fn(&str, &mut Findings),
|
||||
) -> Self {
|
||||
Self {
|
||||
id,
|
||||
name,
|
||||
region,
|
||||
strength,
|
||||
find,
|
||||
}
|
||||
}
|
||||
|
||||
/// Adds what this detector finds in `text` to `findings`. Call once per
|
||||
/// piece of text (subject, each part, each attachment) with the same
|
||||
/// `findings`, then read its length.
|
||||
pub fn find(&self, text: &str, findings: &mut Findings) {
|
||||
(self.find)(text, findings)
|
||||
}
|
||||
|
||||
/// The distinct values found in one text.
|
||||
pub fn count(&self, text: &str) -> usize {
|
||||
let mut findings = Findings::default();
|
||||
self.find(text, &mut findings);
|
||||
findings.len()
|
||||
}
|
||||
}
|
||||
|
||||
/// Every detector, in the order the console lists them.
|
||||
pub fn all() -> impl Iterator<Item = &'static Detector> {
|
||||
[
|
||||
any::DETECTORS,
|
||||
us::DETECTORS,
|
||||
uk::DETECTORS,
|
||||
canada::DETECTORS,
|
||||
australia::DETECTORS,
|
||||
eu::DETECTORS,
|
||||
europe::DETECTORS,
|
||||
asia::DETECTORS,
|
||||
americas::DETECTORS,
|
||||
africa::DETECTORS,
|
||||
]
|
||||
.into_iter()
|
||||
.flatten()
|
||||
}
|
||||
|
||||
pub fn by_id(id: &str) -> Option<&'static Detector> {
|
||||
all().find(|detector| detector.id == id)
|
||||
}
|
||||
|
||||
/// Whether one of `words` appears, as a whole word and ignoring case, within
|
||||
/// [`WINDOW`] characters before `start` or after `end` (byte offsets of the
|
||||
/// candidate in `text`). The window is widened by the longest word, so a
|
||||
/// word that reaches into it still counts whole.
|
||||
pub fn word_near(text: &str, start: usize, end: usize, words: &[&str]) -> bool {
|
||||
let reach = WINDOW + words.iter().map(|w| w.chars().count()).max().unwrap_or(0);
|
||||
let before = text[..start]
|
||||
.char_indices()
|
||||
.rev()
|
||||
.nth(reach - 1)
|
||||
.map_or(0, |(i, _)| i);
|
||||
let after = text[end..]
|
||||
.char_indices()
|
||||
.nth(reach)
|
||||
.map_or(text.len(), |(i, _)| end + i);
|
||||
let window = text[before..after].to_lowercase();
|
||||
words.iter().any(|word| contains_word(&window, word))
|
||||
}
|
||||
|
||||
/// Whether `word` (lower case) appears in `haystack` (lower case) with no
|
||||
/// letter or digit on either side.
|
||||
pub fn contains_word(haystack: &str, word: &str) -> bool {
|
||||
haystack.match_indices(word).any(|(i, _)| {
|
||||
let before_ok = haystack[..i]
|
||||
.chars()
|
||||
.next_back()
|
||||
.is_none_or(|c| !c.is_alphanumeric());
|
||||
let after_ok = haystack[i + word.len()..]
|
||||
.chars()
|
||||
.next()
|
||||
.is_none_or(|c| !c.is_alphanumeric());
|
||||
before_ok && after_ok
|
||||
})
|
||||
}
|
||||
|
||||
/// Whether the match at `start..end` stands alone: no digit or letter
|
||||
/// directly before or after it, so `123-45-6789` isn't found inside a
|
||||
/// longer run of digits.
|
||||
pub fn stands_alone(text: &str, start: usize, end: usize) -> bool {
|
||||
let before = text[..start].chars().next_back();
|
||||
let after = text[end..].chars().next();
|
||||
before.is_none_or(|c| !c.is_alphanumeric()) && after.is_none_or(|c| !c.is_alphanumeric())
|
||||
}
|
||||
|
||||
/// Days in `month` of `year` (0 for a month that doesn't exist).
|
||||
pub fn days_in(year: u32, month: u32) -> u32 {
|
||||
match month {
|
||||
1 | 3 | 5 | 7 | 8 | 10 | 12 => 31,
|
||||
4 | 6 | 9 | 11 => 30,
|
||||
2 if year.is_multiple_of(4) && (!year.is_multiple_of(100) || year.is_multiple_of(400)) => {
|
||||
29
|
||||
}
|
||||
2 => 28,
|
||||
_ => 0,
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether `year`-`month`-`day` is a real date between 1900 and 2100.
|
||||
pub fn valid_date(year: u32, month: u32, day: u32) -> bool {
|
||||
(1900..=2100).contains(&year) && (1..=days_in(year, month)).contains(&day)
|
||||
}
|
||||
|
||||
/// Whether a two-digit year, month and day make a real date in either the
|
||||
/// 1900s or the 2000s.
|
||||
pub fn valid_short_date(yy: u32, month: u32, day: u32) -> bool {
|
||||
valid_date(1900 + yy, month, day) || valid_date(2000 + yy, month, day)
|
||||
}
|
||||
|
||||
/// The value of each digit in `s`.
|
||||
pub fn digit_values(s: &str) -> Vec<u32> {
|
||||
s.bytes()
|
||||
.filter(u8::is_ascii_digit)
|
||||
.map(|b| u32::from(b - b'0'))
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// The ASCII digits of `s`.
|
||||
pub fn digits(s: &str) -> String {
|
||||
s.chars().filter(char::is_ascii_digit).collect()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn words_are_whole_and_near() {
|
||||
let text = "Your passport number is X1234567, thanks";
|
||||
let start = text.find("X123").unwrap();
|
||||
assert!(word_near(text, start, start + 8, &["passport"]));
|
||||
assert!(!word_near(text, start, start + 8, &["pass"]));
|
||||
let far = format!("passport{}X1234567", " ".repeat(60));
|
||||
let start = far.find("X123").unwrap();
|
||||
assert!(!word_near(&far, start, start + 8, &["passport"]));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn near_counts_characters_not_bytes() {
|
||||
// 45 two-byte characters between the word and the candidate: within
|
||||
// 50 characters, though over 50 bytes
|
||||
let text = format!("passport {} X1234567", "é".repeat(45));
|
||||
let start = text.find("X123").unwrap();
|
||||
assert!(word_near(&text, start, start + 8, &["passport"]));
|
||||
}
|
||||
|
||||
/// An ordinary business email: order, invoice and tracking numbers,
|
||||
/// dates, amounts, a street address. Nothing here is an identifier, so
|
||||
/// no detector may fire, except the contact ones on the signature.
|
||||
#[test]
|
||||
fn ordinary_mail_finds_nothing() {
|
||||
let text = "Hi Dana,\n\nThanks for order 4471-2290 placed 2026-09-14. Invoice INV-2026-00917 \
|
||||
for $12,480.00 is due 10/31/2026; PO 7731902 covers lines 1-14. Tracking \
|
||||
1Z999AA10123456784, parcel 3 of 5, 12.5 kg, box 40x30x20 cm. Meeting moved to \
|
||||
Tuesday 9:30-10:15 in room 2B, building 1177. Ticket #5520318, case 20260914-0042. \
|
||||
Version 2026.9.28.4, build 118822, commit 5a73a118. Serial SN-88213-X. \
|
||||
Ship to 1600 Amphitheatre Pkwy, Mountain View, CA 94043. Revenue grew 18% to \
|
||||
1,204,332 units; see figures 3.1-3.4 and table 12.\n\nBest,\nSam\n\
|
||||
Sam Rivera | +1 (415) 555-2671 | [email protected]";
|
||||
let quiet = ["email-addresses", "phone-numbers"];
|
||||
for detector in all().filter(|d| !quiet.contains(&d.id)) {
|
||||
assert_eq!(
|
||||
detector.count(text),
|
||||
0,
|
||||
"{} fired on ordinary mail",
|
||||
detector.id
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ids_are_unique() {
|
||||
let mut seen = AHashSet::new();
|
||||
for detector in all() {
|
||||
assert!(seen.insert(detector.id), "duplicate id {}", detector.id);
|
||||
assert!(by_id(detector.id).is_some());
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,97 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Templates (§2.3): named sets of detectors, so a policy doesn't pick forty
|
||||
//! one at a time. Each is named for what it finds, never for a law, and is a
|
||||
//! starting point: once added to a rule, its detectors can be changed.
|
||||
|
||||
pub struct Template {
|
||||
pub id: &'static str,
|
||||
pub name: &'static str,
|
||||
pub detectors: &'static [&'static str],
|
||||
}
|
||||
|
||||
pub static TEMPLATES: &[Template] = &[
|
||||
Template {
|
||||
id: "payment-and-bank",
|
||||
name: "Payment cards and bank accounts",
|
||||
detectors: &["payment-card", "iban", "swift-bic", "us-aba-routing"],
|
||||
},
|
||||
Template {
|
||||
id: "us-personal",
|
||||
name: "US personal identifiers",
|
||||
detectors: &[
|
||||
"us-ssn",
|
||||
"us-itin",
|
||||
"us-ein",
|
||||
"us-drivers-license",
|
||||
"passport",
|
||||
"date-of-birth",
|
||||
],
|
||||
},
|
||||
Template {
|
||||
id: "uk-personal",
|
||||
name: "UK personal identifiers",
|
||||
detectors: &["uk-nino", "uk-utr", "uk-nhs", "passport", "date-of-birth"],
|
||||
},
|
||||
Template {
|
||||
id: "eu-national",
|
||||
name: "EU national identifiers",
|
||||
detectors: &[
|
||||
"de-tax-id",
|
||||
"de-id-card",
|
||||
"fr-nir",
|
||||
"es-dni-nie",
|
||||
"it-codice-fiscale",
|
||||
"nl-bsn",
|
||||
"be-national-number",
|
||||
"pl-pesel",
|
||||
"se-personnummer",
|
||||
"dk-cpr",
|
||||
"fi-hetu",
|
||||
"ie-pps",
|
||||
"pt-nif",
|
||||
"at-svnr",
|
||||
],
|
||||
},
|
||||
Template {
|
||||
id: "health",
|
||||
name: "Health identifiers",
|
||||
detectors: &["uk-nhs", "us-mbi", "us-npi", "us-dea", "au-medicare"],
|
||||
},
|
||||
Template {
|
||||
id: "credentials",
|
||||
name: "Credentials and keys",
|
||||
detectors: &["private-key", "credentials"],
|
||||
},
|
||||
Template {
|
||||
id: "contact-lists",
|
||||
name: "Contact lists",
|
||||
detectors: &["email-addresses", "phone-numbers"],
|
||||
},
|
||||
];
|
||||
|
||||
pub fn by_id(id: &str) -> Option<&'static Template> {
|
||||
TEMPLATES.iter().find(|template| template.id == id)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn every_template_names_real_detectors() {
|
||||
for template in TEMPLATES {
|
||||
for id in template.detectors {
|
||||
assert!(
|
||||
super::super::by_id(id).is_some(),
|
||||
"{}: no detector {id}",
|
||||
template.id
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,155 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! United Kingdom identifiers (§2.3): HMRC's National Insurance number and
|
||||
//! Unique Taxpayer Reference, and the NHS number.
|
||||
|
||||
use super::{Detector, Findings, Region, Strength, word_near};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[
|
||||
Detector::new(
|
||||
"uk-nino",
|
||||
"UK National Insurance number",
|
||||
Region::Uk,
|
||||
Strength::Checked,
|
||||
nino,
|
||||
),
|
||||
Detector::new(
|
||||
"uk-nhs",
|
||||
"UK NHS number",
|
||||
Region::Uk,
|
||||
Strength::Checked,
|
||||
nhs,
|
||||
),
|
||||
Detector::new(
|
||||
"uk-utr",
|
||||
"UK Unique Taxpayer Reference",
|
||||
Region::Uk,
|
||||
Strength::NeedsWord,
|
||||
utr,
|
||||
),
|
||||
];
|
||||
|
||||
fn re(pattern: &str) -> Regex {
|
||||
Regex::new(pattern).expect("detector pattern")
|
||||
}
|
||||
|
||||
/// Two letters, six digits (often in pairs), a suffix A–D.
|
||||
static NINO: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"(?i)\b([A-Z])([A-Z]) ?(\d{2}) ?(\d{2}) ?(\d{2}) ?([A-D])\b"));
|
||||
|
||||
/// HMRC's rules: D, F, I, Q, U and V are never used; O never second; and
|
||||
/// BG, GB, KN, NK, NT, TN and ZZ are never allocated.
|
||||
fn nino_prefix(first: char, second: char) -> bool {
|
||||
const NEVER: &str = "DFIQUV";
|
||||
let pair: String = [first, second].iter().collect();
|
||||
!NEVER.contains(first)
|
||||
&& !NEVER.contains(second)
|
||||
&& second != 'O'
|
||||
&& !["BG", "GB", "KN", "NK", "NT", "TN", "ZZ"].contains(&pair.as_str())
|
||||
}
|
||||
|
||||
fn nino(text: &str, findings: &mut Findings) {
|
||||
for c in NINO.captures_iter(text) {
|
||||
let first = c[1].to_ascii_uppercase().chars().next().unwrap();
|
||||
let second = c[2].to_ascii_uppercase().chars().next().unwrap();
|
||||
if nino_prefix(first, second) {
|
||||
findings.insert(format!(
|
||||
"{first}{second}{}{}{}{}",
|
||||
&c[3],
|
||||
&c[4],
|
||||
&c[5],
|
||||
c[6].to_ascii_uppercase()
|
||||
));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// `NNN NNN NNNN` stands alone; ten bare digits need a word.
|
||||
static NHS: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{3})([ -]?)(\d{3})([ -]?)(\d{4})\b"));
|
||||
|
||||
/// Mod 11: weights 10 down to 2 over the first nine digits; the check digit
|
||||
/// is 11 minus the remainder (11 becomes 0; 10 is never issued).
|
||||
pub fn nhs_valid(n: &str) -> bool {
|
||||
let d: Vec<u32> = n.bytes().map(|b| u32::from(b - b'0')).collect();
|
||||
let sum: u32 = d[..9].iter().zip((2..=10).rev()).map(|(a, w)| a * w).sum();
|
||||
match 11 - sum % 11 {
|
||||
11 => d[9] == 0,
|
||||
10 => false,
|
||||
check => d[9] == check,
|
||||
}
|
||||
}
|
||||
|
||||
const NHS_WORDS: &[&str] = &["nhs", "nhs number", "nhs no"];
|
||||
|
||||
fn nhs(text: &str, findings: &mut Findings) {
|
||||
for c in NHS.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
|
||||
let written = !c[2].is_empty() && c[2] == c[4];
|
||||
if nhs_valid(&n) && (written || word_near(text, whole.start(), whole.end(), NHS_WORDS)) {
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static UTR: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{5} ?\d{5}\b"));
|
||||
|
||||
const UTR_WORDS: &[&str] = &[
|
||||
"utr",
|
||||
"unique taxpayer reference",
|
||||
"tax reference",
|
||||
"self assessment",
|
||||
];
|
||||
|
||||
fn utr(text: &str, findings: &mut Findings) {
|
||||
for m in UTR.find_iter(text) {
|
||||
if word_near(text, m.start(), m.end(), UTR_WORDS) {
|
||||
findings.insert(m.as_str().replace(' ', ""));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
fn count(id: &str, text: &str) -> usize {
|
||||
by_id(id).unwrap().count(text)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn national_insurance() {
|
||||
assert_eq!(count("uk-nino", "NI: AB 12 34 56 C, ce123456d"), 2);
|
||||
// Letters never used, pairs never allocated, a suffix past D
|
||||
for bad in [
|
||||
"QQ123456C",
|
||||
"AO123456C",
|
||||
"GB123456A",
|
||||
"AB123456E",
|
||||
"DA123456A",
|
||||
] {
|
||||
assert_eq!(count("uk-nino", bad), 0, "{bad}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn nhs_numbers() {
|
||||
// The NHS's own example
|
||||
assert_eq!(count("uk-nhs", "943 476 5919"), 1);
|
||||
assert_eq!(count("uk-nhs", "943 476 5918"), 0);
|
||||
assert_eq!(count("uk-nhs", "order 9434765919"), 0);
|
||||
assert_eq!(count("uk-nhs", "NHS number 9434765919"), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn utr() {
|
||||
assert_eq!(count("uk-utr", "UTR 12345 67890"), 1);
|
||||
assert_eq!(count("uk-utr", "order 1234567890"), 0);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,304 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! United States identifiers (§2.3), each from its issuer's published rules:
|
||||
//! the SSA (SSN), the IRS (ITIN, EIN), the ABA (routing numbers), CMS (MBI,
|
||||
//! NPI) and the DEA.
|
||||
|
||||
use super::{Detector, Findings, Region, Strength, checks, digits, stands_alone, word_near};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[
|
||||
Detector::new(
|
||||
"us-ssn",
|
||||
"US Social Security number",
|
||||
Region::Us,
|
||||
Strength::Checked,
|
||||
ssn,
|
||||
),
|
||||
Detector::new("us-itin", "US ITIN", Region::Us, Strength::Checked, itin),
|
||||
Detector::new("us-ein", "US EIN", Region::Us, Strength::NeedsWord, ein),
|
||||
Detector::new(
|
||||
"us-aba-routing",
|
||||
"US bank routing number",
|
||||
Region::Us,
|
||||
Strength::NeedsWord,
|
||||
aba_routing,
|
||||
),
|
||||
Detector::new(
|
||||
"us-drivers-license",
|
||||
"US driver's license",
|
||||
Region::Us,
|
||||
Strength::NeedsWord,
|
||||
drivers_license,
|
||||
),
|
||||
Detector::new(
|
||||
"us-mbi",
|
||||
"US Medicare Beneficiary Identifier",
|
||||
Region::Us,
|
||||
Strength::Checked,
|
||||
mbi,
|
||||
),
|
||||
Detector::new(
|
||||
"us-npi",
|
||||
"US National Provider Identifier",
|
||||
Region::Us,
|
||||
Strength::NeedsWord,
|
||||
npi,
|
||||
),
|
||||
Detector::new(
|
||||
"us-dea",
|
||||
"US DEA registration number",
|
||||
Region::Us,
|
||||
Strength::Checked,
|
||||
dea,
|
||||
),
|
||||
];
|
||||
|
||||
fn re(pattern: &str) -> Regex {
|
||||
Regex::new(pattern).expect("detector pattern")
|
||||
}
|
||||
|
||||
/// `AAA-GG-SSSS` (dashes or spaces), or nine bare digits.
|
||||
static NINE: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{3})([ -]?)(\d{2})([ -]?)(\d{4})\b"));
|
||||
|
||||
/// Numbers the SSA has published as never valid: widely printed examples.
|
||||
const SSN_EXAMPLES: &[&str] = &["078051120", "219099999"];
|
||||
|
||||
fn ssn_rules(area: u32, group: u32, serial: u32) -> bool {
|
||||
area != 0 && area != 666 && area < 900 && group != 0 && serial != 0
|
||||
}
|
||||
|
||||
const SSN_WORDS: &[&str] = &["ssn", "social security", "soc sec", "ss#", "ss no"];
|
||||
|
||||
fn ssn(text: &str, findings: &mut Findings) {
|
||||
for c in NINE.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let (area, group, serial) = (num(&c[1]), num(&c[3]), num(&c[5]));
|
||||
let number = format!("{}{}{}", &c[1], &c[3], &c[5]);
|
||||
// Written form (with both separators, the same one) stands alone;
|
||||
// nine bare digits need a word
|
||||
let written = !c[2].is_empty() && c[2] == c[4];
|
||||
if ssn_rules(area, group, serial)
|
||||
&& !SSN_EXAMPLES.contains(&number.as_str())
|
||||
&& (written || word_near(text, whole.start(), whole.end(), SSN_WORDS))
|
||||
{
|
||||
findings.insert(number);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// ITINs: 9XX, then a group in the IRS's ranges.
|
||||
fn itin_group(group: u32) -> bool {
|
||||
matches!(group, 50..=65 | 70..=88 | 90..=92 | 94..=99)
|
||||
}
|
||||
|
||||
const ITIN_WORDS: &[&str] = &["itin", "taxpayer identification", "tax id"];
|
||||
|
||||
fn itin(text: &str, findings: &mut Findings) {
|
||||
for c in NINE.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let written = !c[2].is_empty() && c[2] == c[4];
|
||||
if c[1].starts_with('9')
|
||||
&& itin_group(num(&c[3]))
|
||||
&& (written || word_near(text, whole.start(), whole.end(), ITIN_WORDS))
|
||||
{
|
||||
findings.insert(format!("{}{}{}", &c[1], &c[3], &c[5]));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static EIN: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{2})-?(\d{7})\b"));
|
||||
|
||||
/// The prefixes the IRS assigns to its campuses and internet EINs.
|
||||
fn ein_prefix(prefix: u32) -> bool {
|
||||
matches!(prefix, 1..=6 | 10..=16 | 20..=27 | 30..=48 | 50..=68 | 71..=77 | 80..=88 | 90..=95 | 98 | 99)
|
||||
}
|
||||
|
||||
const EIN_WORDS: &[&str] = &[
|
||||
"ein",
|
||||
"fein",
|
||||
"employer identification",
|
||||
"tax id",
|
||||
"tin",
|
||||
"federal tax",
|
||||
];
|
||||
|
||||
fn ein(text: &str, findings: &mut Findings) {
|
||||
for c in EIN.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
if ein_prefix(num(&c[1])) && word_near(text, whole.start(), whole.end(), EIN_WORDS) {
|
||||
findings.insert(format!("{}{}", &c[1], &c[2]));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static ROUTING: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{9}\b"));
|
||||
|
||||
/// The ABA check: 3, 7 and 1 weights, mod 10; and a Federal Reserve prefix.
|
||||
pub fn aba_valid(n: &str) -> bool {
|
||||
let d: Vec<u32> = n.bytes().map(|b| u32::from(b - b'0')).collect();
|
||||
let prefix = d[0] * 10 + d[1];
|
||||
matches!(prefix, 0..=12 | 21..=32 | 61..=72 | 80)
|
||||
&& (3 * (d[0] + d[3] + d[6]) + 7 * (d[1] + d[4] + d[7]) + (d[2] + d[5] + d[8]))
|
||||
.is_multiple_of(10)
|
||||
}
|
||||
|
||||
const ROUTING_WORDS: &[&str] = &["routing", "aba", "rtn", "routing number", "transit"];
|
||||
|
||||
fn aba_routing(text: &str, findings: &mut Findings) {
|
||||
for m in ROUTING.find_iter(text) {
|
||||
// One random number in ten passes the check: always needs a word
|
||||
if aba_valid(m.as_str()) && word_near(text, m.start(), m.end(), ROUTING_WORDS) {
|
||||
findings.insert(m.as_str());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The shapes states issue: up to two letters, then 5–14 digits, dashes
|
||||
/// allowed (Florida and Illinois print them).
|
||||
static LICENSE: LazyLock<Regex> = LazyLock::new(|| re(r"\b[A-Z]{0,2}\d[\d-]{3,16}\d\b"));
|
||||
|
||||
const LICENSE_WORDS: &[&str] = &[
|
||||
"driver's license",
|
||||
"drivers license",
|
||||
"driver license",
|
||||
"driver's licence",
|
||||
"dl",
|
||||
"dl#",
|
||||
"license number",
|
||||
"lic no",
|
||||
"dmv",
|
||||
];
|
||||
|
||||
fn drivers_license(text: &str, findings: &mut Findings) {
|
||||
for m in LICENSE.find_iter(text) {
|
||||
let n = digits(m.as_str());
|
||||
if (5..=14).contains(&n.len()) && word_near(text, m.start(), m.end(), LICENSE_WORDS) {
|
||||
findings.insert(m.as_str().replace('-', ""));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// CMS's MBI: 11 characters in a fixed pattern of digits, letters and
|
||||
/// either, the letters S, L, O, I, B and Z never used; dashes may follow the
|
||||
/// 4th and 7th.
|
||||
static MBI: LazyLock<Regex> = LazyLock::new(|| {
|
||||
let c = "[AC-HJKMNP-RT-Y]";
|
||||
let an = "[AC-HJKMNP-RT-Y0-9]";
|
||||
re(&format!(
|
||||
r"\b[1-9]{c}{an}[0-9]-?{c}{an}[0-9]-?{c}{c}[0-9][0-9]\b"
|
||||
))
|
||||
});
|
||||
|
||||
fn mbi(text: &str, findings: &mut Findings) {
|
||||
for m in MBI.find_iter(text) {
|
||||
findings.insert(m.as_str().replace('-', ""));
|
||||
}
|
||||
}
|
||||
|
||||
static TEN: LazyLock<Regex> = LazyLock::new(|| re(r"\b[12]\d{9}\b"));
|
||||
|
||||
const NPI_WORDS: &[&str] = &["npi", "national provider", "provider id", "provider number"];
|
||||
|
||||
/// NPI: Luhn over the ISO card-issuer prefix 80840 and the number.
|
||||
fn npi(text: &str, findings: &mut Findings) {
|
||||
for m in TEN.find_iter(text) {
|
||||
if checks::luhn(&format!("80840{}", m.as_str()))
|
||||
&& word_near(text, m.start(), m.end(), NPI_WORDS)
|
||||
{
|
||||
findings.insert(m.as_str());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static DEA: LazyLock<Regex> = LazyLock::new(|| re(r"\b([ABCDEFGHJKLMPRSTUX][A-Z9])(\d{7})\b"));
|
||||
|
||||
/// DEA: (1st + 3rd + 5th) + 2 × (2nd + 4th + 6th) ends in the 7th digit.
|
||||
fn dea(text: &str, findings: &mut Findings) {
|
||||
for c in DEA.captures_iter(text) {
|
||||
let d: Vec<u32> = c[2].bytes().map(|b| u32::from(b - b'0')).collect();
|
||||
if ((d[0] + d[2] + d[4]) + 2 * (d[1] + d[3] + d[5])) % 10 == d[6] {
|
||||
let whole = c.get(0).unwrap();
|
||||
if stands_alone(text, whole.start(), whole.end()) {
|
||||
findings.insert(whole.as_str());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn num(s: &str) -> u32 {
|
||||
s.parse().unwrap_or(0)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
fn count(id: &str, text: &str) -> usize {
|
||||
by_id(id).unwrap().count(text)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ssn() {
|
||||
assert_eq!(count("us-ssn", "SSN 536-22-1234, also 536 22 1235"), 2);
|
||||
// Bare digits: only with a word
|
||||
assert_eq!(count("us-ssn", "ref 536221234"), 0);
|
||||
assert_eq!(count("us-ssn", "social security: 536221234"), 1);
|
||||
// Never issued, the SSA's printed examples, mixed separators
|
||||
for bad in [
|
||||
"000-12-3456",
|
||||
"666-12-3456",
|
||||
"912-12-3456",
|
||||
"123-00-4567",
|
||||
"123-45-0000",
|
||||
"078-05-1120",
|
||||
"536-22 1234",
|
||||
] {
|
||||
assert_eq!(count("us-ssn", bad), 0, "{bad}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn itin_and_ein() {
|
||||
assert_eq!(count("us-itin", "912-70-1234"), 1);
|
||||
assert_eq!(count("us-itin", "912-69-1234"), 0);
|
||||
assert_eq!(count("us-ssn", "912-70-1234"), 0);
|
||||
assert_eq!(count("us-ein", "EIN: 12-3456789"), 1);
|
||||
assert_eq!(count("us-ein", "part 12-3456789"), 0);
|
||||
assert_eq!(count("us-ein", "EIN 07-3456789"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn routing_needs_a_word() {
|
||||
assert_eq!(count("us-aba-routing", "Routing number 011000015"), 1);
|
||||
assert_eq!(count("us-aba-routing", "ABA 021000021"), 1);
|
||||
assert_eq!(count("us-aba-routing", "invoice 011000015"), 0);
|
||||
assert_eq!(count("us-aba-routing", "routing 011000016"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn licenses() {
|
||||
assert_eq!(count("us-drivers-license", "Driver's license: D1234567"), 1);
|
||||
assert_eq!(count("us-drivers-license", "DL# S123-456-78-901-0"), 1);
|
||||
assert_eq!(count("us-drivers-license", "Order D1234567"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn health_identifiers() {
|
||||
// CMS's own MBI example
|
||||
assert_eq!(count("us-mbi", "Medicare 1EG4-TE5-MK73"), 1);
|
||||
assert_eq!(count("us-mbi", "1EG4TE5MK73"), 1);
|
||||
assert_eq!(count("us-mbi", "1EG4-TE5-MK7S"), 0);
|
||||
// CMS's NPI example
|
||||
assert_eq!(count("us-npi", "NPI 1234567893"), 1);
|
||||
assert_eq!(count("us-npi", "NPI 1234567894"), 0);
|
||||
assert_eq!(count("us-npi", "call 1234567893"), 0);
|
||||
assert_eq!(count("us-dea", "DEA AB1234563"), 1);
|
||||
assert_eq!(count("us-dea", "AB1234564"), 0);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,695 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Evaluating rules against a message (§2.1–§2.4). Rules are compiled once,
|
||||
//! when they change: word lists become automata, patterns regexes. A message
|
||||
//! is then checked against every enabled rule in order; each detector runs
|
||||
//! at most once per message, and only when some rule asks for it.
|
||||
//!
|
||||
//! Pure: the caller parses the message, extracts attachment text
|
||||
//! ([`super::extract`]) and knows the sender's groups and tenant. What comes
|
||||
//! back is which rules matched, with each detector's count, and what DLP
|
||||
//! decided; the matched text itself never leaves here (§2.7).
|
||||
|
||||
use super::{
|
||||
detectors::{self, Findings},
|
||||
extract::Extracted,
|
||||
rules::{Action, Condition, Direction, Kind, Rule},
|
||||
words::{Pattern, WordList},
|
||||
};
|
||||
use ahash::AHashMap;
|
||||
use std::borrow::Cow;
|
||||
|
||||
/// Who sent a message, and to whom.
|
||||
#[derive(Debug, Clone, Default)]
|
||||
pub struct Envelope<'a> {
|
||||
/// Outgoing (an authenticated sender) or incoming.
|
||||
pub outgoing: bool,
|
||||
pub sender: &'a str,
|
||||
pub sender_groups: &'a [u32],
|
||||
pub sender_tenant: Option<u32>,
|
||||
pub recipients: Vec<Recipient<'a>>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Default)]
|
||||
pub struct Recipient<'a> {
|
||||
pub address: &'a str,
|
||||
/// At a domain this server hosts.
|
||||
pub local: bool,
|
||||
pub groups: &'a [u32],
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct Attachment<'a> {
|
||||
pub name: Option<&'a str>,
|
||||
/// Declared type, or detected where the caller knows better.
|
||||
pub content_type: &'a str,
|
||||
pub size: u64,
|
||||
pub extracted: Extracted,
|
||||
}
|
||||
|
||||
/// What rules look at.
|
||||
#[derive(Debug, Clone, Default)]
|
||||
pub struct Content<'a> {
|
||||
pub subject: &'a str,
|
||||
/// Each text and HTML part, as text.
|
||||
pub bodies: Vec<Cow<'a, str>>,
|
||||
pub headers: Vec<(&'a str, &'a str)>,
|
||||
pub attachments: Vec<Attachment<'a>>,
|
||||
pub size: u64,
|
||||
/// Text past the inspection limit wasn't read.
|
||||
pub truncated: bool,
|
||||
}
|
||||
|
||||
impl Content<'_> {
|
||||
fn texts(&self) -> impl Iterator<Item = &str> {
|
||||
std::iter::once(self.subject)
|
||||
.chain(self.bodies.iter().map(|b| b.as_ref()))
|
||||
.chain(self.attachments.iter().filter_map(|a| match &a.extracted {
|
||||
Extracted::Text(text) => Some(text.as_str()),
|
||||
_ => None,
|
||||
}))
|
||||
}
|
||||
|
||||
fn cant_be_inspected(&self) -> bool {
|
||||
self.truncated
|
||||
|| self
|
||||
.attachments
|
||||
.iter()
|
||||
.any(|a| matches!(a.extracted, Extracted::NotInspectable(_)))
|
||||
}
|
||||
}
|
||||
|
||||
enum Check {
|
||||
Plain(Condition),
|
||||
Words(WordList, u32),
|
||||
Pattern(Pattern, u32),
|
||||
Header {
|
||||
name: String,
|
||||
contains: Option<String>,
|
||||
matches: Option<Pattern>,
|
||||
},
|
||||
AttachmentName(Pattern),
|
||||
}
|
||||
|
||||
struct CompiledRule {
|
||||
rule: Rule,
|
||||
conditions: Vec<Check>,
|
||||
exceptions: Vec<Check>,
|
||||
}
|
||||
|
||||
/// The enabled rules, ready to run.
|
||||
pub struct Compiled {
|
||||
rules: Vec<CompiledRule>,
|
||||
}
|
||||
|
||||
/// A rule reference, for notices and the audit record.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct RuleRef {
|
||||
pub id: u32,
|
||||
pub name: String,
|
||||
pub notice: String,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct Match {
|
||||
pub rule_id: u32,
|
||||
pub name: String,
|
||||
pub kind: Kind,
|
||||
pub actions: Vec<Action>,
|
||||
/// Each detector (or `words`, `pattern`) that counted, and its count.
|
||||
pub counts: Vec<(String, usize)>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Default)]
|
||||
pub struct Outcome {
|
||||
pub matched: Vec<Match>,
|
||||
pub blocks: Vec<RuleRef>,
|
||||
pub holds: Vec<(RuleRef, bool)>,
|
||||
pub warns: Vec<RuleRef>,
|
||||
}
|
||||
|
||||
/// What DLP decided, strictest first (§2.4).
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub enum Decision {
|
||||
Pass,
|
||||
Block(Vec<RuleRef>),
|
||||
Hold {
|
||||
rules: Vec<RuleRef>,
|
||||
notify_sender: bool,
|
||||
},
|
||||
Warn(Vec<RuleRef>),
|
||||
}
|
||||
|
||||
impl Outcome {
|
||||
/// Block beats hold beats warn. An override (§2.5) answers the warnings
|
||||
/// only: a block or hold still applies.
|
||||
pub fn decision(&self, overridden: bool) -> Decision {
|
||||
if !self.blocks.is_empty() {
|
||||
Decision::Block(self.blocks.clone())
|
||||
} else if !self.holds.is_empty() {
|
||||
Decision::Hold {
|
||||
rules: self.holds.iter().map(|(r, _)| r.clone()).collect(),
|
||||
notify_sender: self.holds.iter().any(|(_, notify)| *notify),
|
||||
}
|
||||
} else if !self.warns.is_empty() && !overridden {
|
||||
Decision::Warn(self.warns.clone())
|
||||
} else {
|
||||
Decision::Pass
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn compile_check(condition: &Condition) -> Result<Check, String> {
|
||||
Ok(match condition {
|
||||
Condition::Words { words, at_least } => Check::Words(WordList::new(words)?, *at_least),
|
||||
Condition::Pattern { pattern, at_least } => {
|
||||
Check::Pattern(Pattern::new(pattern)?, *at_least)
|
||||
}
|
||||
Condition::Header {
|
||||
name,
|
||||
contains,
|
||||
matches,
|
||||
} => Check::Header {
|
||||
name: name.to_ascii_lowercase(),
|
||||
contains: contains.as_ref().map(|c| c.to_lowercase()),
|
||||
matches: matches.as_deref().map(Pattern::new).transpose()?,
|
||||
},
|
||||
Condition::AttachmentName { pattern } => Check::AttachmentName(Pattern::new(pattern)?),
|
||||
other => Check::Plain(other.clone()),
|
||||
})
|
||||
}
|
||||
|
||||
impl Compiled {
|
||||
/// Compiles the enabled rules; one that no longer compiles (a detector
|
||||
/// renamed since it was saved) is skipped and named in the second list.
|
||||
pub fn new(rules: &[Rule]) -> (Self, Vec<(u32, String)>) {
|
||||
let mut compiled = Vec::new();
|
||||
let mut skipped = Vec::new();
|
||||
for rule in rules.iter().filter(|r| r.enabled) {
|
||||
let result = rule.validate().map_err(|e| e.reason).and_then(|_| {
|
||||
Ok(CompiledRule {
|
||||
rule: rule.clone(),
|
||||
conditions: rule
|
||||
.conditions
|
||||
.iter()
|
||||
.map(compile_check)
|
||||
.collect::<Result<_, _>>()?,
|
||||
exceptions: rule
|
||||
.exceptions
|
||||
.iter()
|
||||
.map(compile_check)
|
||||
.collect::<Result<_, _>>()?,
|
||||
})
|
||||
});
|
||||
match result {
|
||||
Ok(c) => compiled.push(c),
|
||||
Err(reason) => skipped.push((rule.id, reason)),
|
||||
}
|
||||
}
|
||||
compiled.sort_by_key(|c| (c.rule.priority, c.rule.id));
|
||||
(Self { rules: compiled }, skipped)
|
||||
}
|
||||
|
||||
pub fn is_empty(&self) -> bool {
|
||||
self.rules.is_empty()
|
||||
}
|
||||
|
||||
/// Whether any rule could apply to mail going this way, so a caller can
|
||||
/// skip parsing when none can.
|
||||
pub fn applies_to(&self, outgoing: bool) -> bool {
|
||||
self.rules
|
||||
.iter()
|
||||
.any(|c| direction_matches(c.rule.direction, outgoing))
|
||||
}
|
||||
|
||||
pub fn evaluate(&self, envelope: &Envelope<'_>, content: &Content<'_>) -> Outcome {
|
||||
let mut state = State {
|
||||
content,
|
||||
detected: AHashMap::new(),
|
||||
};
|
||||
let mut outcome = Outcome::default();
|
||||
for compiled in &self.rules {
|
||||
let rule = &compiled.rule;
|
||||
if !direction_matches(rule.direction, envelope.outgoing) {
|
||||
continue;
|
||||
}
|
||||
let mut counts = Vec::new();
|
||||
let all_match = compiled
|
||||
.conditions
|
||||
.iter()
|
||||
.all(|check| state.check(check, envelope, &mut counts));
|
||||
if !all_match {
|
||||
continue;
|
||||
}
|
||||
let mut ignored = Vec::new();
|
||||
if compiled
|
||||
.exceptions
|
||||
.iter()
|
||||
.any(|check| state.check(check, envelope, &mut ignored))
|
||||
{
|
||||
continue;
|
||||
}
|
||||
for action in &rule.actions {
|
||||
let reference = |notice: &str| RuleRef {
|
||||
id: rule.id,
|
||||
name: rule.name.clone(),
|
||||
notice: notice.to_string(),
|
||||
};
|
||||
match action {
|
||||
Action::Block { notice } => outcome.blocks.push(reference(notice)),
|
||||
Action::Hold {
|
||||
notice,
|
||||
notify_sender,
|
||||
} => outcome.holds.push((reference(notice), *notify_sender)),
|
||||
Action::Warn { notice } => outcome.warns.push(reference(notice)),
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
outcome.matched.push(Match {
|
||||
rule_id: rule.id,
|
||||
name: rule.name.clone(),
|
||||
kind: rule.kind,
|
||||
actions: rule.actions.clone(),
|
||||
counts,
|
||||
});
|
||||
if rule.stop_processing {
|
||||
break;
|
||||
}
|
||||
}
|
||||
outcome
|
||||
}
|
||||
}
|
||||
|
||||
fn direction_matches(direction: Direction, outgoing: bool) -> bool {
|
||||
match direction {
|
||||
Direction::Any => true,
|
||||
Direction::Outgoing => outgoing,
|
||||
Direction::Incoming => !outgoing,
|
||||
}
|
||||
}
|
||||
|
||||
fn domain_of(address: &str) -> &str {
|
||||
address.rsplit_once('@').map_or("", |(_, d)| d)
|
||||
}
|
||||
|
||||
fn in_list(value: &str, list: &[String]) -> bool {
|
||||
list.iter().any(|v| v.eq_ignore_ascii_case(value))
|
||||
}
|
||||
|
||||
struct State<'c, 'a> {
|
||||
content: &'c Content<'a>,
|
||||
/// Each detector's count, run once per message.
|
||||
detected: AHashMap<&'static str, usize>,
|
||||
}
|
||||
|
||||
impl State<'_, '_> {
|
||||
fn detector_count(&mut self, id: &str) -> usize {
|
||||
let Some(detector) = detectors::by_id(id) else {
|
||||
return 0;
|
||||
};
|
||||
if let Some(count) = self.detected.get(detector.id) {
|
||||
return *count;
|
||||
}
|
||||
let mut findings = Findings::default();
|
||||
for text in self.content.texts() {
|
||||
detector.find(text, &mut findings);
|
||||
}
|
||||
self.detected.insert(detector.id, findings.len());
|
||||
findings.len()
|
||||
}
|
||||
|
||||
fn check(
|
||||
&mut self,
|
||||
check: &Check,
|
||||
envelope: &Envelope<'_>,
|
||||
counts: &mut Vec<(String, usize)>,
|
||||
) -> bool {
|
||||
let content = self.content;
|
||||
match check {
|
||||
Check::Words(list, at_least) => {
|
||||
let n: usize = content.texts().map(|t| list.count(t)).sum();
|
||||
counts.push(("words".into(), n));
|
||||
n >= *at_least as usize
|
||||
}
|
||||
Check::Pattern(pattern, at_least) => {
|
||||
let n: usize = content.texts().map(|t| pattern.count(t)).sum();
|
||||
counts.push(("pattern".into(), n));
|
||||
n >= *at_least as usize
|
||||
}
|
||||
Check::Header {
|
||||
name,
|
||||
contains,
|
||||
matches,
|
||||
} => content
|
||||
.headers
|
||||
.iter()
|
||||
.filter(|(n, _)| n.eq_ignore_ascii_case(name))
|
||||
.any(|(_, value)| match (contains, matches) {
|
||||
(Some(needle), _) => value.to_lowercase().contains(needle.as_str()),
|
||||
(_, Some(pattern)) => pattern.count(value) > 0,
|
||||
_ => true,
|
||||
}),
|
||||
Check::AttachmentName(pattern) => content
|
||||
.attachments
|
||||
.iter()
|
||||
.any(|a| a.name.is_some_and(|n| pattern.count(n) > 0)),
|
||||
Check::Plain(condition) => match condition {
|
||||
Condition::SenderAddress { addresses } => in_list(envelope.sender, addresses),
|
||||
Condition::SenderDomain { domains } => in_list(domain_of(envelope.sender), domains),
|
||||
Condition::SenderGroup { groups } => {
|
||||
envelope.sender_groups.iter().any(|g| groups.contains(g))
|
||||
}
|
||||
Condition::SenderTenant { tenants } => {
|
||||
envelope.sender_tenant.is_some_and(|t| tenants.contains(&t))
|
||||
}
|
||||
Condition::RecipientAddress { addresses } => envelope
|
||||
.recipients
|
||||
.iter()
|
||||
.any(|r| in_list(r.address, addresses)),
|
||||
Condition::RecipientDomain { domains } => envelope
|
||||
.recipients
|
||||
.iter()
|
||||
.any(|r| in_list(domain_of(r.address), domains)),
|
||||
Condition::RecipientGroup { groups } => envelope
|
||||
.recipients
|
||||
.iter()
|
||||
.any(|r| r.groups.iter().any(|g| groups.contains(g))),
|
||||
Condition::RecipientOutside => envelope.recipients.iter().any(|r| !r.local),
|
||||
Condition::AttachmentType { types } => content.attachments.iter().any(|a| {
|
||||
let ct = a.content_type.to_ascii_lowercase();
|
||||
types
|
||||
.iter()
|
||||
.any(|t| ct.starts_with(&t.to_ascii_lowercase()))
|
||||
}),
|
||||
Condition::AttachmentExtension { extensions } => {
|
||||
content.attachments.iter().any(|a| {
|
||||
a.name
|
||||
.and_then(|n| n.rsplit_once('.'))
|
||||
.is_some_and(|(_, ext)| {
|
||||
extensions
|
||||
.iter()
|
||||
.any(|e| e.trim_start_matches('.').eq_ignore_ascii_case(ext))
|
||||
})
|
||||
})
|
||||
}
|
||||
Condition::AttachmentSizeOver { bytes } => {
|
||||
content.attachments.iter().any(|a| a.size > *bytes)
|
||||
}
|
||||
Condition::AttachmentCountOver { count } => {
|
||||
content.attachments.len() > *count as usize
|
||||
}
|
||||
Condition::CantBeInspected => content.cant_be_inspected(),
|
||||
Condition::MessageSizeOver { bytes } => content.size > *bytes,
|
||||
Condition::Detected { detectors } => {
|
||||
let mut any = false;
|
||||
for d in detectors {
|
||||
let n = self.detector_count(&d.id);
|
||||
counts.push((d.id.clone(), n));
|
||||
any |= n >= d.at_least as usize;
|
||||
}
|
||||
any
|
||||
}
|
||||
// Compiled into their own checks
|
||||
Condition::Words { .. }
|
||||
| Condition::Pattern { .. }
|
||||
| Condition::Header { .. }
|
||||
| Condition::AttachmentName { .. } => false,
|
||||
},
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::mailflow::{
|
||||
extract::Why,
|
||||
rules::{DetectorMin, Position},
|
||||
};
|
||||
|
||||
fn rule(id: u32, kind: Kind, conditions: Vec<Condition>, action: Action) -> Rule {
|
||||
Rule {
|
||||
id,
|
||||
name: format!("rule {id}"),
|
||||
description: String::new(),
|
||||
kind,
|
||||
enabled: true,
|
||||
priority: id as i32,
|
||||
direction: if kind == Kind::Dlp {
|
||||
Direction::Outgoing
|
||||
} else {
|
||||
Direction::Any
|
||||
},
|
||||
conditions,
|
||||
exceptions: vec![],
|
||||
actions: vec![action],
|
||||
stop_processing: false,
|
||||
created_by: String::new(),
|
||||
created_at: 0,
|
||||
updated_at: 0,
|
||||
}
|
||||
}
|
||||
|
||||
fn envelope(outside: bool) -> Envelope<'static> {
|
||||
Envelope {
|
||||
outgoing: true,
|
||||
sender: "[email protected]",
|
||||
sender_groups: &[7],
|
||||
sender_tenant: None,
|
||||
recipients: vec![Recipient {
|
||||
address: if outside {
|
||||
"[email protected]"
|
||||
} else {
|
||||
"[email protected]"
|
||||
},
|
||||
local: !outside,
|
||||
groups: &[],
|
||||
}],
|
||||
}
|
||||
}
|
||||
|
||||
fn cards(n: usize) -> Content<'static> {
|
||||
let body: String = [
|
||||
"4242 4242 4242 4242",
|
||||
"5555-5555-5555-4444",
|
||||
"378282246310005",
|
||||
"6011111111111117",
|
||||
"3566002020360505",
|
||||
]
|
||||
.iter()
|
||||
.take(n)
|
||||
.map(|c| format!("card {c}\n"))
|
||||
.collect();
|
||||
Content {
|
||||
subject: "Numbers",
|
||||
bodies: vec![body.into()],
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
|
||||
fn five_cards_outside(action: Action) -> Rule {
|
||||
rule(
|
||||
1,
|
||||
Kind::Dlp,
|
||||
vec![
|
||||
Condition::RecipientOutside,
|
||||
Condition::Detected {
|
||||
detectors: vec![DetectorMin {
|
||||
id: "payment-card".into(),
|
||||
at_least: 5,
|
||||
}],
|
||||
},
|
||||
],
|
||||
action,
|
||||
)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn detector_threshold_and_recipients() {
|
||||
let (rules, skipped) = Compiled::new(&[five_cards_outside(Action::Hold {
|
||||
notice: "Held".into(),
|
||||
notify_sender: true,
|
||||
})]);
|
||||
assert!(skipped.is_empty());
|
||||
let outcome = rules.evaluate(&envelope(true), &cards(5));
|
||||
assert_eq!(
|
||||
outcome.matched[0].counts,
|
||||
vec![("payment-card".to_string(), 5)]
|
||||
);
|
||||
assert!(matches!(
|
||||
outcome.decision(false),
|
||||
Decision::Hold {
|
||||
notify_sender: true,
|
||||
..
|
||||
}
|
||||
));
|
||||
// Four cards, or everyone inside: nothing
|
||||
assert_eq!(
|
||||
rules.evaluate(&envelope(true), &cards(4)).decision(false),
|
||||
Decision::Pass
|
||||
);
|
||||
assert_eq!(
|
||||
rules.evaluate(&envelope(false), &cards(5)).decision(false),
|
||||
Decision::Pass
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn strictest_wins_and_override_answers_warnings_only() {
|
||||
let warn = five_cards_outside(Action::Warn {
|
||||
notice: "Sure?".into(),
|
||||
});
|
||||
let mut block = five_cards_outside(Action::Block {
|
||||
notice: "No".into(),
|
||||
});
|
||||
block.id = 2;
|
||||
let (rules, _) = Compiled::new(&[warn.clone(), block]);
|
||||
let outcome = rules.evaluate(&envelope(true), &cards(5));
|
||||
assert!(matches!(outcome.decision(true), Decision::Block(_)));
|
||||
let (rules, _) = Compiled::new(&[warn]);
|
||||
let outcome = rules.evaluate(&envelope(true), &cards(5));
|
||||
assert!(matches!(outcome.decision(false), Decision::Warn(ref w) if w[0].notice == "Sure?"));
|
||||
assert_eq!(outcome.decision(true), Decision::Pass);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn exceptions_order_and_stop_processing() {
|
||||
let disclaimer = |id| {
|
||||
rule(
|
||||
id,
|
||||
Kind::Transport,
|
||||
vec![Condition::RecipientOutside],
|
||||
Action::AddDisclaimer {
|
||||
text: "t".into(),
|
||||
html: None,
|
||||
position: Position::Bottom,
|
||||
},
|
||||
)
|
||||
};
|
||||
let mut first = disclaimer(1);
|
||||
first.stop_processing = true;
|
||||
let (rules, _) = Compiled::new(&[disclaimer(2), first.clone()]);
|
||||
let outcome = rules.evaluate(&envelope(true), &cards(0));
|
||||
assert_eq!(
|
||||
outcome
|
||||
.matched
|
||||
.iter()
|
||||
.map(|m| m.rule_id)
|
||||
.collect::<Vec<_>>(),
|
||||
vec![1]
|
||||
);
|
||||
|
||||
first.stop_processing = false;
|
||||
first.exceptions = vec![Condition::SenderGroup { groups: vec![7] }];
|
||||
let (rules, _) = Compiled::new(&[disclaimer(2), first]);
|
||||
let outcome = rules.evaluate(&envelope(true), &cards(0));
|
||||
assert_eq!(
|
||||
outcome
|
||||
.matched
|
||||
.iter()
|
||||
.map(|m| m.rule_id)
|
||||
.collect::<Vec<_>>(),
|
||||
vec![2]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn content_conditions() {
|
||||
let content = Content {
|
||||
subject: "Project Falcon",
|
||||
bodies: vec!["see attached".into()],
|
||||
headers: vec![("X-Class", "Internal only")],
|
||||
attachments: vec![
|
||||
Attachment {
|
||||
name: Some("plan.docx"),
|
||||
content_type: "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
||||
size: 40_000,
|
||||
extracted: Extracted::Text("IBAN GB29 NWBK 6016 1331 9268 19".into()),
|
||||
},
|
||||
Attachment {
|
||||
name: Some("scan.pdf"),
|
||||
content_type: "application/pdf",
|
||||
size: 900_000,
|
||||
extracted: Extracted::NotInspectable(Why::Pdf),
|
||||
},
|
||||
],
|
||||
size: 1_000_000,
|
||||
truncated: false,
|
||||
};
|
||||
let block = || Action::Block { notice: "n".into() };
|
||||
let checks = [
|
||||
(
|
||||
Condition::Words {
|
||||
words: vec!["project falcon".into()],
|
||||
at_least: 1,
|
||||
},
|
||||
true,
|
||||
),
|
||||
(
|
||||
Condition::Header {
|
||||
name: "x-class".into(),
|
||||
contains: Some("internal".into()),
|
||||
matches: None,
|
||||
},
|
||||
true,
|
||||
),
|
||||
(
|
||||
Condition::AttachmentExtension {
|
||||
extensions: vec![".PDF".into()],
|
||||
},
|
||||
true,
|
||||
),
|
||||
(
|
||||
Condition::AttachmentType {
|
||||
types: vec!["image/".into()],
|
||||
},
|
||||
false,
|
||||
),
|
||||
(Condition::AttachmentSizeOver { bytes: 500_000 }, true),
|
||||
(Condition::AttachmentCountOver { count: 2 }, false),
|
||||
(Condition::CantBeInspected, true),
|
||||
(Condition::MessageSizeOver { bytes: 2_000_000 }, false),
|
||||
(
|
||||
Condition::Detected {
|
||||
detectors: vec![DetectorMin {
|
||||
id: "iban".into(),
|
||||
at_least: 1,
|
||||
}],
|
||||
},
|
||||
true,
|
||||
),
|
||||
(
|
||||
Condition::SenderDomain {
|
||||
domains: vec!["EXAMPLE.com".into()],
|
||||
},
|
||||
true,
|
||||
),
|
||||
];
|
||||
for (condition, expected) in checks {
|
||||
let (rules, skipped) =
|
||||
Compiled::new(&[rule(1, Kind::Dlp, vec![condition.clone()], block())]);
|
||||
assert!(skipped.is_empty(), "{condition:?}");
|
||||
let matched = !rules.evaluate(&envelope(true), &content).matched.is_empty();
|
||||
assert_eq!(matched, expected, "{condition:?}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn direction_and_disabled_rules() {
|
||||
let mut r = five_cards_outside(Action::Block { notice: "n".into() });
|
||||
let (rules, _) = Compiled::new(std::slice::from_ref(&r));
|
||||
assert!(rules.applies_to(true) && !rules.applies_to(false));
|
||||
let mut incoming = envelope(true);
|
||||
incoming.outgoing = false;
|
||||
assert_eq!(
|
||||
rules.evaluate(&incoming, &cards(5)).decision(false),
|
||||
Decision::Pass
|
||||
);
|
||||
r.enabled = false;
|
||||
assert!(Compiled::new(&[r]).0.is_empty());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,694 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! The text of an attachment, for the detectors (§2.3), or why there isn't
|
||||
//! one.
|
||||
//!
|
||||
//! Read: text files (plain, CSV, JSON, XML, HTML), Office Open XML (DOCX,
|
||||
//! XLSX, PPTX) and OpenDocument (ODT, ODS, ODP) documents, and ZIP archives
|
||||
//! one level deep. **Can't be inspected**: encrypted or password-protected
|
||||
//! files, PDF (settled answer 2), the older binary Office formats, archives
|
||||
//! inside archives, and anything past the limits. Everything else (images,
|
||||
//! audio, programs) has no text to read and is neither.
|
||||
//!
|
||||
//! Office files are ZIP archives of XML, read here with the `zip` and
|
||||
//! `quick-xml` crates the server already uses: no outside converter runs.
|
||||
|
||||
use quick_xml::{Reader, XmlVersion, events::Event};
|
||||
use std::io::{Cursor, Read};
|
||||
|
||||
/// How much may be unpacked from one attachment, and from how many entries.
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub struct Limits {
|
||||
pub max_unpacked: u64,
|
||||
pub max_entries: usize,
|
||||
}
|
||||
|
||||
impl Default for Limits {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
max_unpacked: 50 * 1024 * 1024,
|
||||
max_entries: 10_000,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub enum Extracted {
|
||||
/// The text to check.
|
||||
Text(String),
|
||||
/// A kind of file with no text in it: nothing to check, nothing missed.
|
||||
NoText,
|
||||
/// A file that may hold text the detectors couldn't read.
|
||||
NotInspectable(Why),
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum Why {
|
||||
Encrypted,
|
||||
Pdf,
|
||||
LegacyOffice,
|
||||
NestedArchive,
|
||||
TooLarge,
|
||||
Damaged,
|
||||
}
|
||||
|
||||
impl Why {
|
||||
pub fn as_str(&self) -> &'static str {
|
||||
match self {
|
||||
Why::Encrypted => "encrypted",
|
||||
Why::Pdf => "pdf",
|
||||
Why::LegacyOffice => "legacy-office",
|
||||
Why::NestedArchive => "nested-archive",
|
||||
Why::TooLarge => "too-large",
|
||||
Why::Damaged => "damaged",
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const OLE_MAGIC: &[u8] = &[0xD0, 0xCF, 0x11, 0xE0, 0xA1, 0xB1, 0x1A, 0xE1];
|
||||
const ZIP_MAGIC: &[u8] = b"PK\x03\x04";
|
||||
|
||||
/// What an attachment says, from its declared type, its file name and, above
|
||||
/// all, its first bytes.
|
||||
pub fn extract(
|
||||
content_type: &str,
|
||||
file_name: Option<&str>,
|
||||
data: &[u8],
|
||||
limits: &Limits,
|
||||
) -> Extracted {
|
||||
extract_at(content_type, file_name, data, limits, 0)
|
||||
}
|
||||
|
||||
fn extract_at(
|
||||
content_type: &str,
|
||||
file_name: Option<&str>,
|
||||
data: &[u8],
|
||||
limits: &Limits,
|
||||
depth: u8,
|
||||
) -> Extracted {
|
||||
let content_type = content_type.to_ascii_lowercase();
|
||||
let extension = file_name
|
||||
.and_then(|name| name.rsplit_once('.'))
|
||||
.map(|(_, ext)| ext.to_ascii_lowercase())
|
||||
.unwrap_or_default();
|
||||
|
||||
if data.len() as u64 > limits.max_unpacked {
|
||||
return Extracted::NotInspectable(Why::TooLarge);
|
||||
}
|
||||
if data.starts_with(b"%PDF-") || content_type == "application/pdf" || extension == "pdf" {
|
||||
return Extracted::NotInspectable(Why::Pdf);
|
||||
}
|
||||
if data.starts_with(OLE_MAGIC) {
|
||||
// An encrypted OOXML file is an OLE container holding the encrypted
|
||||
// package; any other OLE file is a legacy .doc, .xls or .ppt
|
||||
return Extracted::NotInspectable(if has_utf16(data, "EncryptedPackage") {
|
||||
Why::Encrypted
|
||||
} else {
|
||||
Why::LegacyOffice
|
||||
});
|
||||
}
|
||||
if data.starts_with(ZIP_MAGIC) {
|
||||
if depth > 0 {
|
||||
return Extracted::NotInspectable(Why::NestedArchive);
|
||||
}
|
||||
return zip(data, limits);
|
||||
}
|
||||
if is_text(&content_type, &extension) {
|
||||
let text = decode_text(data);
|
||||
return Extracted::Text(
|
||||
if content_type == "text/html" || matches!(extension.as_str(), "html" | "htm") {
|
||||
strip_html(&text)
|
||||
} else {
|
||||
text
|
||||
},
|
||||
);
|
||||
}
|
||||
Extracted::NoText
|
||||
}
|
||||
|
||||
fn is_text(content_type: &str, extension: &str) -> bool {
|
||||
content_type.starts_with("text/")
|
||||
|| matches!(
|
||||
content_type,
|
||||
"application/json"
|
||||
| "application/xml"
|
||||
| "application/csv"
|
||||
| "application/x-csv"
|
||||
| "message/rfc822"
|
||||
)
|
||||
|| matches!(
|
||||
extension,
|
||||
"txt"
|
||||
| "csv"
|
||||
| "tsv"
|
||||
| "json"
|
||||
| "xml"
|
||||
| "md"
|
||||
| "log"
|
||||
| "html"
|
||||
| "htm"
|
||||
| "eml"
|
||||
| "ics"
|
||||
| "vcf"
|
||||
)
|
||||
}
|
||||
|
||||
/// UTF-16 with a byte order mark, else UTF-8 (lossy).
|
||||
fn decode_text(data: &[u8]) -> String {
|
||||
let utf16 = |bytes: &[u8], big: bool| {
|
||||
let units: Vec<u16> = bytes
|
||||
.as_chunks::<2>()
|
||||
.0
|
||||
.iter()
|
||||
.map(|&c| {
|
||||
if big {
|
||||
u16::from_be_bytes(c)
|
||||
} else {
|
||||
u16::from_le_bytes(c)
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
String::from_utf16_lossy(&units)
|
||||
};
|
||||
match data {
|
||||
[0xFF, 0xFE, rest @ ..] => utf16(rest, false),
|
||||
[0xFE, 0xFF, rest @ ..] => utf16(rest, true),
|
||||
[0xEF, 0xBB, 0xBF, rest @ ..] => String::from_utf8_lossy(rest).into_owned(),
|
||||
_ => String::from_utf8_lossy(data).into_owned(),
|
||||
}
|
||||
}
|
||||
|
||||
fn has_utf16(data: &[u8], needle: &str) -> bool {
|
||||
let needle: Vec<u8> = needle.encode_utf16().flat_map(u16::to_le_bytes).collect();
|
||||
data.windows(needle.len()).any(|w| w == needle.as_slice())
|
||||
}
|
||||
|
||||
/// Tags out, the common entities decoded, block ends as new lines.
|
||||
fn strip_html(html: &str) -> String {
|
||||
let mut out = String::with_capacity(html.len());
|
||||
let mut in_tag = false;
|
||||
let mut skip_until: Option<&str> = None;
|
||||
let lower = html.to_ascii_lowercase();
|
||||
let mut i = 0;
|
||||
let bytes = html.as_bytes();
|
||||
while i < bytes.len() {
|
||||
if let Some(end) = skip_until {
|
||||
match lower[i..].find(end) {
|
||||
Some(at) => {
|
||||
i += at + end.len();
|
||||
skip_until = None;
|
||||
}
|
||||
None => break,
|
||||
}
|
||||
continue;
|
||||
}
|
||||
let c = bytes[i];
|
||||
if in_tag {
|
||||
if c == b'>' {
|
||||
in_tag = false;
|
||||
}
|
||||
i += 1;
|
||||
continue;
|
||||
}
|
||||
if c == b'<' {
|
||||
if lower[i..].starts_with("<script") {
|
||||
skip_until = Some("</script>");
|
||||
} else if lower[i..].starts_with("<style") {
|
||||
skip_until = Some("</style>");
|
||||
} else {
|
||||
if [
|
||||
"<br", "<p", "</p", "<div", "</div", "<tr", "<li", "<td", "<th",
|
||||
]
|
||||
.iter()
|
||||
.any(|t| lower[i..].starts_with(t))
|
||||
{
|
||||
out.push(
|
||||
if lower[i..].starts_with("<td") || lower[i..].starts_with("<th") {
|
||||
'\t'
|
||||
} else {
|
||||
'\n'
|
||||
},
|
||||
);
|
||||
}
|
||||
in_tag = true;
|
||||
}
|
||||
i += 1;
|
||||
continue;
|
||||
}
|
||||
// Copy up to the next tag
|
||||
let next = html[i..].find('<').map_or(html.len(), |at| i + at);
|
||||
out.push_str(&html[i..next]);
|
||||
i = next;
|
||||
}
|
||||
for (entity, text) in [
|
||||
(" ", " "),
|
||||
("<", "<"),
|
||||
(">", ">"),
|
||||
(""", "\""),
|
||||
("'", "'"),
|
||||
("&", "&"),
|
||||
] {
|
||||
out = out.replace(entity, text);
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
/// A ZIP file: an Office document, an OpenDocument, or an archive.
|
||||
fn zip(data: &[u8], limits: &Limits) -> Extracted {
|
||||
let Ok(mut archive) = zip::ZipArchive::new(Cursor::new(data)) else {
|
||||
return Extracted::NotInspectable(Why::Damaged);
|
||||
};
|
||||
if archive.len() > limits.max_entries {
|
||||
return Extracted::NotInspectable(Why::TooLarge);
|
||||
}
|
||||
let mut names = Vec::with_capacity(archive.len());
|
||||
let mut declared: u64 = 0;
|
||||
for i in 0..archive.len() {
|
||||
let Ok(entry) = archive.by_index_raw(i) else {
|
||||
return Extracted::NotInspectable(Why::Damaged);
|
||||
};
|
||||
if entry.encrypted() {
|
||||
return Extracted::NotInspectable(Why::Encrypted);
|
||||
}
|
||||
declared = declared.saturating_add(entry.size());
|
||||
names.push(entry.name().to_string());
|
||||
}
|
||||
if declared > limits.max_unpacked {
|
||||
return Extracted::NotInspectable(Why::TooLarge);
|
||||
}
|
||||
let mut budget = limits.max_unpacked;
|
||||
let mut read =
|
||||
|archive: &mut zip::ZipArchive<Cursor<&[u8]>>, name: &str| -> Result<Vec<u8>, Why> {
|
||||
let entry = archive.by_name(name).map_err(|_| Why::Damaged)?;
|
||||
let mut bytes = Vec::new();
|
||||
// Declared sizes can lie: stop at the budget whatever they say
|
||||
entry
|
||||
.take(budget + 1)
|
||||
.read_to_end(&mut bytes)
|
||||
.map_err(|_| Why::Damaged)?;
|
||||
if bytes.len() as u64 > budget {
|
||||
return Err(Why::TooLarge);
|
||||
}
|
||||
budget -= bytes.len() as u64;
|
||||
Ok(bytes)
|
||||
};
|
||||
|
||||
let has = |name: &str| names.iter().any(|n| n == name);
|
||||
let mut text = String::new();
|
||||
let result: Result<(), Why> = (|| {
|
||||
if has("[Content_Types].xml") {
|
||||
// Office Open XML: the parts that hold what a person wrote
|
||||
let mut shared = Vec::new();
|
||||
if has("xl/sharedStrings.xml") {
|
||||
shared = xml_strings(&read(&mut archive, "xl/sharedStrings.xml")?, "si");
|
||||
}
|
||||
for name in names.iter().filter(|n| ooxml_text_part(n)) {
|
||||
let xml = read(&mut archive, name)?;
|
||||
if name.starts_with("xl/worksheets/") {
|
||||
xlsx_sheet(&xml, &mut text);
|
||||
} else {
|
||||
xml_text(&xml, &mut text);
|
||||
}
|
||||
text.push('\n');
|
||||
}
|
||||
text.extend(shared.iter().map(|s| format!("{s}\n")));
|
||||
} else if names.first().is_some_and(|n| n == "mimetype")
|
||||
&& read(&mut archive, "mimetype")?.starts_with(b"application/vnd.oasis.opendocument")
|
||||
{
|
||||
// OpenDocument: an encrypted one says so in its manifest
|
||||
if has("META-INF/manifest.xml")
|
||||
&& contains(
|
||||
&read(&mut archive, "META-INF/manifest.xml")?,
|
||||
b"encryption-data",
|
||||
)
|
||||
{
|
||||
return Err(Why::Encrypted);
|
||||
}
|
||||
for name in ["content.xml", "styles.xml"] {
|
||||
if has(name) {
|
||||
xml_text(&read(&mut archive, name)?, &mut text);
|
||||
text.push('\n');
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// An archive: each file inside, one level deep
|
||||
for name in names.iter().filter(|n| !n.ends_with('/')) {
|
||||
let bytes = read(&mut archive, name)?;
|
||||
match extract_at("", Some(name), &bytes, limits, 1) {
|
||||
Extracted::Text(inner) => {
|
||||
text.push_str(&inner);
|
||||
text.push('\n');
|
||||
}
|
||||
Extracted::NoText => {}
|
||||
Extracted::NotInspectable(why) => return Err(why),
|
||||
}
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
})();
|
||||
match result {
|
||||
Ok(()) => Extracted::Text(text),
|
||||
Err(why) => Extracted::NotInspectable(why),
|
||||
}
|
||||
}
|
||||
|
||||
fn ooxml_text_part(name: &str) -> bool {
|
||||
let xml = name.ends_with(".xml");
|
||||
xml && (name == "word/document.xml"
|
||||
|| [
|
||||
"word/header",
|
||||
"word/footer",
|
||||
"word/footnotes",
|
||||
"word/endnotes",
|
||||
"word/comments",
|
||||
]
|
||||
.iter()
|
||||
.any(|p| name.starts_with(p))
|
||||
|| name.starts_with("xl/worksheets/sheet")
|
||||
|| name.starts_with("ppt/slides/slide")
|
||||
|| name.starts_with("ppt/notesSlides/"))
|
||||
}
|
||||
|
||||
fn contains(haystack: &[u8], needle: &[u8]) -> bool {
|
||||
haystack.windows(needle.len()).any(|w| w == needle)
|
||||
}
|
||||
|
||||
/// The local name of a tag, without its namespace prefix.
|
||||
fn local(name: &[u8]) -> &[u8] {
|
||||
name.rsplit(|b| *b == b':').next().unwrap_or(name)
|
||||
}
|
||||
|
||||
fn push_entity(entity: &[u8], out: &mut String) {
|
||||
match entity {
|
||||
b"lt" => out.push('<'),
|
||||
b"gt" => out.push('>'),
|
||||
b"amp" => out.push('&'),
|
||||
b"apos" => out.push('\''),
|
||||
b"quot" => out.push('"'),
|
||||
_ => {
|
||||
let code = match entity {
|
||||
[b'#', b'x' | b'X', hex @ ..] => std::str::from_utf8(hex)
|
||||
.ok()
|
||||
.and_then(|h| u32::from_str_radix(h, 16).ok()),
|
||||
[b'#', dec @ ..] => std::str::from_utf8(dec).ok().and_then(|d| d.parse().ok()),
|
||||
_ => None,
|
||||
};
|
||||
if let Some(c) = code.and_then(char::from_u32) {
|
||||
out.push(c);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Every text node, runs joined as written, a new line after each paragraph
|
||||
/// or row and a tab after each cell, so a number split across runs is whole
|
||||
/// again.
|
||||
fn xml_text(xml: &[u8], out: &mut String) {
|
||||
let mut reader = Reader::from_reader(xml);
|
||||
let mut buf = Vec::new();
|
||||
loop {
|
||||
match reader.read_event_into(&mut buf) {
|
||||
Ok(Event::Text(t)) => {
|
||||
if let Ok(text) = t.xml_content(XmlVersion::Implicit1_0) {
|
||||
out.push_str(&text);
|
||||
}
|
||||
}
|
||||
Ok(Event::CData(t)) => out.push_str(&String::from_utf8_lossy(&t)),
|
||||
Ok(Event::GeneralRef(entity)) => push_entity(&entity, out),
|
||||
Ok(Event::End(e)) => match local(e.name().as_ref()) {
|
||||
b"p" | b"h" | b"tr" | b"row" | b"table-row" | b"br" => out.push('\n'),
|
||||
b"tc" | b"c" | b"table-cell" | b"tab" => out.push('\t'),
|
||||
_ => {}
|
||||
},
|
||||
Ok(Event::Empty(e)) => match local(e.name().as_ref()) {
|
||||
b"br" | b"line-break" => out.push('\n'),
|
||||
b"tab" | b"s" => out.push(' '),
|
||||
_ => {}
|
||||
},
|
||||
Ok(Event::Eof) | Err(_) => break,
|
||||
_ => {}
|
||||
}
|
||||
buf.clear();
|
||||
}
|
||||
}
|
||||
|
||||
/// The text of each `item` element (a shared string in XLSX).
|
||||
fn xml_strings(xml: &[u8], item: &str) -> Vec<String> {
|
||||
let mut reader = Reader::from_reader(xml);
|
||||
let mut buf = Vec::new();
|
||||
let mut items = Vec::new();
|
||||
let mut current: Option<String> = None;
|
||||
loop {
|
||||
match reader.read_event_into(&mut buf) {
|
||||
Ok(Event::Start(e)) if local(e.name().as_ref()) == item.as_bytes() => {
|
||||
current = Some(String::new())
|
||||
}
|
||||
Ok(Event::End(e)) if local(e.name().as_ref()) == item.as_bytes() => {
|
||||
items.extend(current.take());
|
||||
}
|
||||
Ok(Event::Text(t)) => {
|
||||
if let (Some(s), Ok(text)) =
|
||||
(current.as_mut(), t.xml_content(XmlVersion::Implicit1_0))
|
||||
{
|
||||
s.push_str(&text);
|
||||
}
|
||||
}
|
||||
Ok(Event::GeneralRef(entity)) => {
|
||||
if let Some(s) = current.as_mut() {
|
||||
push_entity(&entity, s);
|
||||
}
|
||||
}
|
||||
Ok(Event::Eof) | Err(_) => break,
|
||||
_ => {}
|
||||
}
|
||||
buf.clear();
|
||||
}
|
||||
items
|
||||
}
|
||||
|
||||
/// A worksheet's cell values: numbers and inline strings. Cells holding a
|
||||
/// shared string are skipped here; the shared strings are read whole.
|
||||
fn xlsx_sheet(xml: &[u8], out: &mut String) {
|
||||
let mut reader = Reader::from_reader(xml);
|
||||
let mut buf = Vec::new();
|
||||
let mut shared_cell = false;
|
||||
let mut in_value = false;
|
||||
loop {
|
||||
match reader.read_event_into(&mut buf) {
|
||||
Ok(Event::Start(e)) => match local(e.name().as_ref()) {
|
||||
b"c" => {
|
||||
shared_cell = e
|
||||
.attributes()
|
||||
.flatten()
|
||||
.any(|a| a.key.as_ref() == b"t" && a.value.as_ref() == b"s");
|
||||
}
|
||||
b"v" | b"t" => in_value = true,
|
||||
_ => {}
|
||||
},
|
||||
Ok(Event::End(e)) => match local(e.name().as_ref()) {
|
||||
b"v" | b"t" => in_value = false,
|
||||
b"c" => out.push('\t'),
|
||||
b"row" => out.push('\n'),
|
||||
_ => {}
|
||||
},
|
||||
// A shared string's cell holds only its index: the string itself
|
||||
// is added with the shared strings
|
||||
Ok(Event::Text(t)) if in_value && !shared_cell => {
|
||||
if let Ok(text) = t.xml_content(XmlVersion::Implicit1_0) {
|
||||
out.push_str(&text);
|
||||
}
|
||||
}
|
||||
Ok(Event::Eof) | Err(_) => break,
|
||||
_ => {}
|
||||
}
|
||||
buf.clear();
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use std::io::Write;
|
||||
use zip::{ZipWriter, write::SimpleFileOptions};
|
||||
|
||||
fn zip_of(files: &[(&str, &str)]) -> Vec<u8> {
|
||||
let mut zip = ZipWriter::new(Cursor::new(Vec::new()));
|
||||
for (name, body) in files {
|
||||
zip.start_file(*name, SimpleFileOptions::default()).unwrap();
|
||||
zip.write_all(body.as_bytes()).unwrap();
|
||||
}
|
||||
zip.finish().unwrap().into_inner()
|
||||
}
|
||||
|
||||
fn text_of(extracted: Extracted) -> String {
|
||||
match extracted {
|
||||
Extracted::Text(text) => text,
|
||||
other => panic!("expected text, got {other:?}"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn plain_text_and_html() {
|
||||
let limits = Limits::default();
|
||||
assert_eq!(
|
||||
text_of(extract("text/plain", None, b"card 4242", &limits)),
|
||||
"card 4242"
|
||||
);
|
||||
let utf16: Vec<u8> = [0xFF, 0xFE]
|
||||
.into_iter()
|
||||
.chain("héllo".encode_utf16().flat_map(u16::to_le_bytes))
|
||||
.collect();
|
||||
assert_eq!(
|
||||
text_of(extract(
|
||||
"application/octet-stream",
|
||||
Some("a.csv"),
|
||||
&utf16,
|
||||
&limits
|
||||
)),
|
||||
"héllo"
|
||||
);
|
||||
let html = "<html><style>p{}</style><p>Card 4242</p><script>x()</script><td>a</td><td>b</td></html>";
|
||||
let text = text_of(extract("text/html", None, html.as_bytes(), &limits));
|
||||
assert!(
|
||||
text.contains("Card 4242") && !text.contains("x()") && !text.contains("p{}"),
|
||||
"{text:?}"
|
||||
);
|
||||
assert_eq!(
|
||||
extract("image/png", Some("a.png"), b"\x89PNG....", &limits),
|
||||
Extracted::NoText
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn docx_joins_split_runs() {
|
||||
let doc = r#"<w:document xmlns:w="w"><w:body><w:p><w:r><w:t>Card 4242 42</w:t></w:r><w:r><w:t>42 4242 4242</w:t></w:r></w:p><w:p><w:r><w:t>A & B</w:t></w:r></w:p></w:body></w:document>"#;
|
||||
let docx = zip_of(&[
|
||||
("[Content_Types].xml", "<Types/>"),
|
||||
("word/document.xml", doc),
|
||||
]);
|
||||
let text = text_of(extract(
|
||||
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
||||
Some("a.docx"),
|
||||
&docx,
|
||||
&Limits::default(),
|
||||
));
|
||||
assert!(text.contains("Card 4242 4242 4242 4242\nA & B"), "{text:?}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn xlsx_numbers_and_shared_strings() {
|
||||
let sheet = r#"<worksheet><sheetData><row><c r="A1" t="s"><v>0</v></c><c r="B1"><v>4242424242424242</v></c></row></sheetData></worksheet>"#;
|
||||
let shared = r#"<sst><si><t>IBAN GB29 NWBK 6016 1331 9268 19</t></si></sst>"#;
|
||||
let xlsx = zip_of(&[
|
||||
("[Content_Types].xml", "<Types/>"),
|
||||
("xl/sharedStrings.xml", shared),
|
||||
("xl/worksheets/sheet1.xml", sheet),
|
||||
]);
|
||||
let text = text_of(extract("", Some("book.xlsx"), &xlsx, &Limits::default()));
|
||||
assert!(
|
||||
text.contains("4242424242424242") && text.contains("GB29 NWBK 6016 1331 9268 19"),
|
||||
"{text:?}"
|
||||
);
|
||||
// The shared string's index isn't read as a value
|
||||
assert!(
|
||||
!text.contains("\t0\t") && !text.starts_with('0'),
|
||||
"{text:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn opendocument_and_encrypted_opendocument() {
|
||||
let content = r#"<office:document-content xmlns:text="t"><text:p>SSN 078-05-1120</text:p></office:document-content>"#;
|
||||
let odt = zip_of(&[
|
||||
("mimetype", "application/vnd.oasis.opendocument.text"),
|
||||
("content.xml", content),
|
||||
]);
|
||||
assert!(
|
||||
text_of(extract("", Some("a.odt"), &odt, &Limits::default()))
|
||||
.contains("SSN 078-05-1120")
|
||||
);
|
||||
let manifest = r#"<manifest:manifest><manifest:file-entry><manifest:encryption-data/></manifest:file-entry></manifest:manifest>"#;
|
||||
let locked = zip_of(&[
|
||||
("mimetype", "application/vnd.oasis.opendocument.text"),
|
||||
("META-INF/manifest.xml", manifest),
|
||||
("content.xml", "x"),
|
||||
]);
|
||||
assert_eq!(
|
||||
extract("", Some("a.odt"), &locked, &Limits::default()),
|
||||
Extracted::NotInspectable(Why::Encrypted)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn archives() {
|
||||
let limits = Limits::default();
|
||||
let archive = zip_of(&[
|
||||
("notes/a.txt", "card 4242424242424242"),
|
||||
("b.png", "\u{89}PNG"),
|
||||
]);
|
||||
assert!(
|
||||
text_of(extract("application/zip", Some("x.zip"), &archive, &limits))
|
||||
.contains("4242424242424242")
|
||||
);
|
||||
let nested = zip_of(&[(
|
||||
"inner.zip",
|
||||
std::str::from_utf8(&[b'P', b'K', 3, 4]).unwrap(),
|
||||
)]);
|
||||
assert_eq!(
|
||||
extract("application/zip", Some("x.zip"), &nested, &limits),
|
||||
Extracted::NotInspectable(Why::NestedArchive)
|
||||
);
|
||||
|
||||
// Password-protected
|
||||
let mut zip = ZipWriter::new(Cursor::new(Vec::new()));
|
||||
zip.start_file(
|
||||
"secret.txt",
|
||||
SimpleFileOptions::default().with_aes_encryption(zip::AesMode::Aes256, "pw"),
|
||||
)
|
||||
.unwrap();
|
||||
zip.write_all(b"4242424242424242").unwrap();
|
||||
let locked = zip.finish().unwrap().into_inner();
|
||||
assert_eq!(
|
||||
extract("application/zip", Some("x.zip"), &locked, &limits),
|
||||
Extracted::NotInspectable(Why::Encrypted)
|
||||
);
|
||||
|
||||
// Past the limits
|
||||
let small = Limits {
|
||||
max_unpacked: 10,
|
||||
max_entries: 1,
|
||||
};
|
||||
assert_eq!(
|
||||
extract("application/zip", Some("x.zip"), &archive, &small),
|
||||
Extracted::NotInspectable(Why::TooLarge)
|
||||
);
|
||||
assert_eq!(
|
||||
extract("application/zip", None, b"PK\x03\x04garbage", &limits),
|
||||
Extracted::NotInspectable(Why::Damaged)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn not_inspectable_kinds() {
|
||||
let limits = Limits::default();
|
||||
assert_eq!(
|
||||
extract("application/octet-stream", None, b"%PDF-1.7 ...", &limits),
|
||||
Extracted::NotInspectable(Why::Pdf)
|
||||
);
|
||||
let mut ole = OLE_MAGIC.to_vec();
|
||||
ole.extend(std::iter::repeat_n(0, 64));
|
||||
assert_eq!(
|
||||
extract("", Some("old.doc"), &ole, &limits),
|
||||
Extracted::NotInspectable(Why::LegacyOffice)
|
||||
);
|
||||
ole.extend("EncryptedPackage".encode_utf16().flat_map(u16::to_le_bytes));
|
||||
assert_eq!(
|
||||
extract("", Some("new.docx"), &ole, &limits),
|
||||
Extracted::NotInspectable(Why::Encrypted)
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,29 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Data loss prevention and mail flow rules (dlp-and-mail-flow-rules spec).
|
||||
//!
|
||||
//! Mostly pure functions over text and attachment bytes, unit-tested
|
||||
//! without a server:
|
||||
//!
|
||||
//! - [`detectors`]: find identifiers in text (payment cards, IBANs,
|
||||
//! national ID numbers, keys), each by its published format and check
|
||||
//! (§2.3);
|
||||
//! - [`words`]: an organization's own word lists and patterns;
|
||||
//! - [`extract`]: the text of an attachment, or why it can't be read;
|
||||
//! - [`rules`]: what a rule is, its checks, and where rules are kept;
|
||||
//! - [`engine`]: rules compiled and run against a message;
|
||||
//! - [`cache`]: each node's compiled copy.
|
||||
//!
|
||||
//! Nothing here writes what it finds anywhere: callers get counts, and the
|
||||
//! matched text never leaves the evaluation (§2.7).
|
||||
|
||||
pub mod cache;
|
||||
pub mod detectors;
|
||||
pub mod engine;
|
||||
pub mod extract;
|
||||
pub mod rules;
|
||||
pub mod words;
|
||||
@@ -0,0 +1,717 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Mail flow rules and DLP rules (dlp-and-mail-flow-rules spec, §2.2–§2.4):
|
||||
//! what a rule is, what makes one valid, and where it's kept.
|
||||
//!
|
||||
//! Kept in the fork's subspace (`store::SUBSPACE_INBUXA`), never in the
|
||||
//! registry, so an upstream schema import never touches them. Every key
|
||||
//! starts with `R`, then one byte for the kind:
|
||||
//!
|
||||
//! - `r` + rule id (u32): the rule, as JSON.
|
||||
//!
|
||||
//! Numbers are big-endian. There are few rules, so they're read whole.
|
||||
|
||||
use super::{detectors, words};
|
||||
use serde::{Deserialize as SerdeDeserialize, Serialize as SerdeSerialize, de::DeserializeOwned};
|
||||
use store::{
|
||||
Deserialize, IterateParams, SUBSPACE_INBUXA, Serialize, Store, ValueKey,
|
||||
write::{AnyClass, BatchBuilder, ValueClass, assert::AssertValue},
|
||||
};
|
||||
use trc::AddContext;
|
||||
|
||||
const FEATURE: u8 = b'R';
|
||||
const KIND_RULE: u8 = b'r';
|
||||
const CREATE_ATTEMPTS: usize = 5;
|
||||
|
||||
/// Longest text a rule may carry (a notice, a disclaimer), in bytes.
|
||||
const MAX_TEXT: usize = 16 * 1024;
|
||||
/// Most entries in one list (words, addresses, domains).
|
||||
const MAX_LIST: usize = 5_000;
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
pub enum Kind {
|
||||
Dlp,
|
||||
Transport,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
pub enum Direction {
|
||||
/// Mail an authenticated sender submits, over SMTP or JMAP.
|
||||
Outgoing,
|
||||
/// Everything else the server accepts.
|
||||
Incoming,
|
||||
Any,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
pub enum Position {
|
||||
Top,
|
||||
Bottom,
|
||||
}
|
||||
|
||||
fn one() -> u32 {
|
||||
1
|
||||
}
|
||||
|
||||
/// A detector and the least it must find.
|
||||
#[derive(Debug, Clone, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
pub struct DetectorMin {
|
||||
pub id: String,
|
||||
#[serde(default = "one")]
|
||||
pub at_least: u32,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(
|
||||
tag = "type",
|
||||
rename_all = "camelCase",
|
||||
rename_all_fields = "camelCase"
|
||||
)]
|
||||
pub enum Condition {
|
||||
SenderAddress {
|
||||
addresses: Vec<String>,
|
||||
},
|
||||
SenderDomain {
|
||||
domains: Vec<String>,
|
||||
},
|
||||
SenderGroup {
|
||||
groups: Vec<u32>,
|
||||
},
|
||||
SenderTenant {
|
||||
tenants: Vec<u32>,
|
||||
},
|
||||
/// Any recipient is one of these.
|
||||
RecipientAddress {
|
||||
addresses: Vec<String>,
|
||||
},
|
||||
RecipientDomain {
|
||||
domains: Vec<String>,
|
||||
},
|
||||
RecipientGroup {
|
||||
groups: Vec<u32>,
|
||||
},
|
||||
/// Any recipient isn't at a domain this server hosts.
|
||||
RecipientOutside,
|
||||
/// Words or phrases in the subject, body or readable attachments.
|
||||
Words {
|
||||
words: Vec<String>,
|
||||
#[serde(default = "one")]
|
||||
at_least: u32,
|
||||
},
|
||||
/// The organization's regular expression, in the same places.
|
||||
Pattern {
|
||||
pattern: String,
|
||||
#[serde(default = "one")]
|
||||
at_least: u32,
|
||||
},
|
||||
/// A header exists, or its value contains or matches.
|
||||
Header {
|
||||
name: String,
|
||||
#[serde(default)]
|
||||
contains: Option<String>,
|
||||
#[serde(default)]
|
||||
matches: Option<String>,
|
||||
},
|
||||
/// An attachment's declared or detected type starts with one of these.
|
||||
AttachmentType {
|
||||
types: Vec<String>,
|
||||
},
|
||||
AttachmentExtension {
|
||||
extensions: Vec<String>,
|
||||
},
|
||||
AttachmentName {
|
||||
pattern: String,
|
||||
},
|
||||
AttachmentSizeOver {
|
||||
bytes: u64,
|
||||
},
|
||||
AttachmentCountOver {
|
||||
count: u32,
|
||||
},
|
||||
/// An attachment is encrypted, a PDF, a legacy Office file, an archive
|
||||
/// inside an archive, or past the inspection limit.
|
||||
CantBeInspected,
|
||||
MessageSizeOver {
|
||||
bytes: u64,
|
||||
},
|
||||
/// Any of these detectors finds at least its minimum (DLP rules only).
|
||||
Detected {
|
||||
detectors: Vec<DetectorMin>,
|
||||
},
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(
|
||||
tag = "type",
|
||||
rename_all = "camelCase",
|
||||
rename_all_fields = "camelCase"
|
||||
)]
|
||||
pub enum Action {
|
||||
// Transport actions
|
||||
AddDisclaimer {
|
||||
text: String,
|
||||
#[serde(default)]
|
||||
html: Option<String>,
|
||||
position: Position,
|
||||
},
|
||||
AddHeader {
|
||||
name: String,
|
||||
value: String,
|
||||
},
|
||||
RemoveHeader {
|
||||
name: String,
|
||||
},
|
||||
PrefixSubject {
|
||||
text: String,
|
||||
},
|
||||
AddRecipient {
|
||||
address: String,
|
||||
},
|
||||
Redirect {
|
||||
addresses: Vec<String>,
|
||||
},
|
||||
Refuse {
|
||||
text: String,
|
||||
},
|
||||
Route {
|
||||
queue: String,
|
||||
},
|
||||
// DLP actions
|
||||
Block {
|
||||
notice: String,
|
||||
},
|
||||
Warn {
|
||||
notice: String,
|
||||
},
|
||||
Hold {
|
||||
notice: String,
|
||||
#[serde(default)]
|
||||
notify_sender: bool,
|
||||
},
|
||||
}
|
||||
|
||||
impl Action {
|
||||
pub fn is_dlp(&self) -> bool {
|
||||
matches!(
|
||||
self,
|
||||
Action::Block { .. } | Action::Warn { .. } | Action::Hold { .. }
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, SerdeSerialize, SerdeDeserialize)]
|
||||
#[serde(rename_all = "camelCase")]
|
||||
pub struct Rule {
|
||||
#[serde(default)]
|
||||
pub id: u32,
|
||||
pub name: String,
|
||||
#[serde(default)]
|
||||
pub description: String,
|
||||
pub kind: Kind,
|
||||
#[serde(default = "enabled")]
|
||||
pub enabled: bool,
|
||||
#[serde(default)]
|
||||
pub priority: i32,
|
||||
pub direction: Direction,
|
||||
#[serde(default)]
|
||||
pub conditions: Vec<Condition>,
|
||||
#[serde(default)]
|
||||
pub exceptions: Vec<Condition>,
|
||||
pub actions: Vec<Action>,
|
||||
#[serde(default)]
|
||||
pub stop_processing: bool,
|
||||
#[serde(default)]
|
||||
pub created_by: String,
|
||||
#[serde(default)]
|
||||
pub created_at: u64,
|
||||
#[serde(default)]
|
||||
pub updated_at: u64,
|
||||
}
|
||||
|
||||
fn enabled() -> bool {
|
||||
true
|
||||
}
|
||||
|
||||
/// Why a rule can't be saved: the property at fault, and a sentence.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct Invalid {
|
||||
pub property: &'static str,
|
||||
pub reason: String,
|
||||
}
|
||||
|
||||
fn invalid(property: &'static str, reason: impl Into<String>) -> Invalid {
|
||||
Invalid {
|
||||
property,
|
||||
reason: reason.into(),
|
||||
}
|
||||
}
|
||||
|
||||
impl Rule {
|
||||
/// Everything that can be checked without the rest of the server: the
|
||||
/// shape (§2.2, §2.4), the detectors, word lists and patterns.
|
||||
pub fn validate(&self) -> Result<(), Invalid> {
|
||||
if self.name.trim().is_empty() {
|
||||
return Err(invalid("name", "A rule needs a name."));
|
||||
}
|
||||
if self.name.len() > 200 || self.description.len() > MAX_TEXT {
|
||||
return Err(invalid("name", "The name or description is too long."));
|
||||
}
|
||||
if self.actions.is_empty() {
|
||||
return Err(invalid("actions", "A rule needs something to do."));
|
||||
}
|
||||
let dlp_actions = self.actions.iter().filter(|a| a.is_dlp()).count();
|
||||
match self.kind {
|
||||
Kind::Dlp => {
|
||||
if self.direction != Direction::Outgoing {
|
||||
return Err(invalid("direction", "DLP rules check outgoing mail only."));
|
||||
}
|
||||
if dlp_actions != 1 || self.actions.len() != 1 {
|
||||
return Err(invalid(
|
||||
"actions",
|
||||
"A DLP rule has exactly one action: block, warn or hold.",
|
||||
));
|
||||
}
|
||||
}
|
||||
Kind::Transport => {
|
||||
if dlp_actions > 0 {
|
||||
return Err(invalid(
|
||||
"actions",
|
||||
"Block, warn and hold belong to DLP rules.",
|
||||
));
|
||||
}
|
||||
if self
|
||||
.conditions
|
||||
.iter()
|
||||
.chain(&self.exceptions)
|
||||
.any(|c| matches!(c, Condition::Detected { .. }))
|
||||
{
|
||||
return Err(invalid("conditions", "Detectors belong to DLP rules."));
|
||||
}
|
||||
}
|
||||
}
|
||||
for (property, list) in [
|
||||
("conditions", &self.conditions),
|
||||
("exceptions", &self.exceptions),
|
||||
] {
|
||||
for condition in list {
|
||||
validate_condition(condition).map_err(|reason| invalid(property, reason))?;
|
||||
}
|
||||
}
|
||||
for action in &self.actions {
|
||||
validate_action(action).map_err(|reason| invalid("actions", reason))?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
fn nonempty_list<T>(list: &[T], what: &str) -> Result<(), String> {
|
||||
if list.is_empty() {
|
||||
Err(format!("The {what} list is empty."))
|
||||
} else if list.len() > MAX_LIST {
|
||||
Err(format!(
|
||||
"The {what} list is longer than {MAX_LIST} entries."
|
||||
))
|
||||
} else {
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
fn header_name(name: &str) -> Result<(), String> {
|
||||
if !name.is_empty()
|
||||
&& name.len() <= 100
|
||||
&& name.bytes().all(|b| b.is_ascii_graphic() && b != b':')
|
||||
{
|
||||
Ok(())
|
||||
} else {
|
||||
Err(format!("\"{name}\" isn't a header name."))
|
||||
}
|
||||
}
|
||||
|
||||
fn text(value: &str, what: &str) -> Result<(), String> {
|
||||
if value.trim().is_empty() {
|
||||
Err(format!("The {what} is empty."))
|
||||
} else if value.len() > MAX_TEXT {
|
||||
Err(format!("The {what} is longer than {MAX_TEXT} bytes."))
|
||||
} else {
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
fn validate_condition(condition: &Condition) -> Result<(), String> {
|
||||
match condition {
|
||||
Condition::SenderAddress { addresses } | Condition::RecipientAddress { addresses } => {
|
||||
nonempty_list(addresses, "address")
|
||||
}
|
||||
Condition::SenderDomain { domains } | Condition::RecipientDomain { domains } => {
|
||||
nonempty_list(domains, "domain")
|
||||
}
|
||||
Condition::SenderGroup { groups } | Condition::RecipientGroup { groups } => {
|
||||
nonempty_list(groups, "group")
|
||||
}
|
||||
Condition::SenderTenant { tenants } => nonempty_list(tenants, "tenant"),
|
||||
Condition::Words { words, at_least } => {
|
||||
nonempty_list(words, "word")?;
|
||||
if *at_least == 0 {
|
||||
return Err("The least number of words must be 1 or more.".into());
|
||||
}
|
||||
words::WordList::new(words).map(|_| ())
|
||||
}
|
||||
Condition::Pattern { pattern, at_least } => {
|
||||
if *at_least == 0 {
|
||||
return Err("The least number of matches must be 1 or more.".into());
|
||||
}
|
||||
words::Pattern::new(pattern).map(|_| ())
|
||||
}
|
||||
Condition::Header {
|
||||
name,
|
||||
contains,
|
||||
matches,
|
||||
} => {
|
||||
header_name(name)?;
|
||||
if let Some(pattern) = matches {
|
||||
words::Pattern::new(pattern)?;
|
||||
}
|
||||
if contains.is_some() && matches.is_some() {
|
||||
return Err("A header condition is either contains or matches.".into());
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
Condition::AttachmentType { types } => nonempty_list(types, "type"),
|
||||
Condition::AttachmentExtension { extensions } => nonempty_list(extensions, "extension"),
|
||||
Condition::AttachmentName { pattern } => words::Pattern::new(pattern).map(|_| ()),
|
||||
Condition::Detected { detectors } => {
|
||||
nonempty_list(detectors, "detector")?;
|
||||
for d in detectors {
|
||||
if detectors::by_id(&d.id).is_none() {
|
||||
return Err(format!("There is no detector \"{}\".", d.id));
|
||||
}
|
||||
if d.at_least == 0 {
|
||||
return Err("A detector's least count must be 1 or more.".into());
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
Condition::RecipientOutside
|
||||
| Condition::AttachmentSizeOver { .. }
|
||||
| Condition::AttachmentCountOver { .. }
|
||||
| Condition::CantBeInspected
|
||||
| Condition::MessageSizeOver { .. } => Ok(()),
|
||||
}
|
||||
}
|
||||
|
||||
fn validate_action(action: &Action) -> Result<(), String> {
|
||||
match action {
|
||||
Action::AddDisclaimer { text: t, html, .. } => {
|
||||
text(t, "disclaimer")?;
|
||||
html.as_deref()
|
||||
.map_or(Ok(()), |h| text(h, "disclaimer's HTML"))
|
||||
}
|
||||
Action::AddHeader { name, value } => {
|
||||
header_name(name)?;
|
||||
if value.len() > 998 || value.contains(['\r', '\n']) {
|
||||
Err("A header value is one line of at most 998 characters.".into())
|
||||
} else {
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
Action::RemoveHeader { name } => header_name(name),
|
||||
Action::PrefixSubject { text: t } => text(t, "subject prefix"),
|
||||
Action::AddRecipient { address } => {
|
||||
if address.contains('@') {
|
||||
Ok(())
|
||||
} else {
|
||||
Err(format!("\"{address}\" isn't an address."))
|
||||
}
|
||||
}
|
||||
Action::Redirect { addresses } => {
|
||||
nonempty_list(addresses, "address")?;
|
||||
match addresses.iter().find(|a| !a.contains('@')) {
|
||||
Some(a) => Err(format!("\"{a}\" isn't an address.")),
|
||||
None => Ok(()),
|
||||
}
|
||||
}
|
||||
Action::Refuse { text: t } => text(t, "refusal text"),
|
||||
Action::Route { queue } => text(queue, "queue"),
|
||||
Action::Block { notice } | Action::Warn { notice } | Action::Hold { notice, .. } => {
|
||||
text(notice, "notice")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Storage --------------------------------------------------------------
|
||||
|
||||
struct Json<T>(T);
|
||||
|
||||
impl<T: SerdeSerialize> Serialize for Json<T> {
|
||||
fn serialize(&self) -> trc::Result<Vec<u8>> {
|
||||
serde_json::to_vec(&self.0).map_err(|err| {
|
||||
trc::StoreEvent::UnexpectedError
|
||||
.into_err()
|
||||
.details("Failed to serialize mail rule")
|
||||
.reason(err)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
impl<T: DeserializeOwned + Sync + Send> Deserialize for Json<T> {
|
||||
fn deserialize(bytes: &[u8]) -> trc::Result<Self> {
|
||||
serde_json::from_slice(bytes).map(Json).map_err(|err| {
|
||||
trc::StoreEvent::DataCorruption
|
||||
.into_err()
|
||||
.details("Invalid mail rule")
|
||||
.reason(err)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
fn class(id: u32) -> ValueClass {
|
||||
let mut key = Vec::with_capacity(6);
|
||||
key.push(FEATURE);
|
||||
key.push(KIND_RULE);
|
||||
key.extend_from_slice(&id.to_be_bytes());
|
||||
ValueClass::Any(AnyClass {
|
||||
subspace: SUBSPACE_INBUXA,
|
||||
key,
|
||||
})
|
||||
}
|
||||
|
||||
fn key(id: u32) -> ValueKey<ValueClass> {
|
||||
ValueKey::from(class(id))
|
||||
}
|
||||
|
||||
pub async fn get(data: &Store, id: u32) -> trc::Result<Option<Rule>> {
|
||||
Ok(data
|
||||
.get_value::<Json<Rule>>(key(id))
|
||||
.await
|
||||
.caused_by(trc::location!())?
|
||||
.map(|Json(rule)| rule))
|
||||
}
|
||||
|
||||
/// Every rule, in the order they run: by priority, then oldest first.
|
||||
pub async fn all(data: &Store) -> trc::Result<Vec<Rule>> {
|
||||
let mut rules = Vec::new();
|
||||
data.iterate(IterateParams::new(key(0), key(u32::MAX)), |_, value| {
|
||||
if let Ok(Json(rule)) = Json::<Rule>::deserialize(value) {
|
||||
rules.push(rule);
|
||||
}
|
||||
Ok(true)
|
||||
})
|
||||
.await
|
||||
.caused_by(trc::location!())?;
|
||||
rules.sort_by_key(|rule| (rule.priority, rule.id));
|
||||
Ok(rules)
|
||||
}
|
||||
|
||||
/// Writes a new rule under the next free id, which it returns. Two nodes
|
||||
/// creating rules at once can't take the same id: the key must be absent.
|
||||
pub async fn create(data: &Store, rule: &Rule) -> trc::Result<u32> {
|
||||
let mut attempt = 0;
|
||||
loop {
|
||||
attempt += 1;
|
||||
let id = all(data).await?.iter().map(|r| r.id).max().unwrap_or(0) + 1;
|
||||
let stored = Rule { id, ..rule.clone() };
|
||||
let mut batch = BatchBuilder::new();
|
||||
batch.assert_value(class(id), AssertValue::None);
|
||||
batch.set(class(id), Json(&stored).serialize()?);
|
||||
match data.write(batch.build_all()).await {
|
||||
Ok(_) => {
|
||||
super::cache::invalidate();
|
||||
return Ok(id);
|
||||
}
|
||||
Err(err)
|
||||
if attempt < CREATE_ATTEMPTS
|
||||
&& matches!(
|
||||
err.as_ref(),
|
||||
trc::EventType::Store(trc::StoreEvent::AssertValueFailed)
|
||||
) => {}
|
||||
Err(err) => return Err(err.caused_by(trc::location!())),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Replaces a stored rule (same id).
|
||||
pub async fn update(data: &Store, rule: &Rule) -> trc::Result<()> {
|
||||
let mut batch = BatchBuilder::new();
|
||||
batch.set(class(rule.id), Json(rule).serialize()?);
|
||||
data.write(batch.build_all())
|
||||
.await
|
||||
.caused_by(trc::location!())?;
|
||||
super::cache::invalidate();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub async fn delete(data: &Store, id: u32) -> trc::Result<()> {
|
||||
let mut batch = BatchBuilder::new();
|
||||
batch.clear(class(id));
|
||||
data.write(batch.build_all())
|
||||
.await
|
||||
.caused_by(trc::location!())?;
|
||||
super::cache::invalidate();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn rule(kind: Kind, actions: Vec<Action>) -> Rule {
|
||||
Rule {
|
||||
id: 0,
|
||||
name: "Cards outside".into(),
|
||||
description: String::new(),
|
||||
kind,
|
||||
enabled: true,
|
||||
priority: 0,
|
||||
direction: Direction::Outgoing,
|
||||
conditions: vec![Condition::RecipientOutside],
|
||||
exceptions: vec![],
|
||||
actions,
|
||||
stop_processing: false,
|
||||
created_by: String::new(),
|
||||
created_at: 0,
|
||||
updated_at: 0,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn wire_format() {
|
||||
let json = r#"{"name":"Cards","kind":"dlp","direction":"outgoing",
|
||||
"conditions":[{"type":"recipientOutside"},{"type":"detected","detectors":[{"id":"payment-card","atLeast":5}]}],
|
||||
"actions":[{"type":"hold","notice":"Held for review","notifySender":true}]}"#;
|
||||
let parsed: Rule = serde_json::from_str(json).unwrap();
|
||||
assert!(parsed.enabled);
|
||||
assert_eq!(
|
||||
parsed.conditions[1],
|
||||
Condition::Detected {
|
||||
detectors: vec![DetectorMin {
|
||||
id: "payment-card".into(),
|
||||
at_least: 5
|
||||
}]
|
||||
}
|
||||
);
|
||||
assert_eq!(
|
||||
parsed.actions[0],
|
||||
Action::Hold {
|
||||
notice: "Held for review".into(),
|
||||
notify_sender: true
|
||||
}
|
||||
);
|
||||
assert!(parsed.validate().is_ok());
|
||||
let back = serde_json::to_value(&parsed).unwrap();
|
||||
assert_eq!(back["actions"][0]["notifySender"], true);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dlp_rules_have_one_dlp_action_on_outgoing_mail() {
|
||||
let block = Action::Block {
|
||||
notice: "No.".into(),
|
||||
};
|
||||
assert!(rule(Kind::Dlp, vec![block.clone()]).validate().is_ok());
|
||||
let two = rule(
|
||||
Kind::Dlp,
|
||||
vec![
|
||||
block.clone(),
|
||||
Action::Warn {
|
||||
notice: "Hm.".into(),
|
||||
},
|
||||
],
|
||||
);
|
||||
assert_eq!(two.validate().unwrap_err().property, "actions");
|
||||
let mixed = rule(
|
||||
Kind::Dlp,
|
||||
vec![block.clone(), Action::PrefixSubject { text: "[x]".into() }],
|
||||
);
|
||||
assert_eq!(mixed.validate().unwrap_err().property, "actions");
|
||||
let mut inbound = rule(Kind::Dlp, vec![block.clone()]);
|
||||
inbound.direction = Direction::Incoming;
|
||||
assert_eq!(inbound.validate().unwrap_err().property, "direction");
|
||||
assert_eq!(
|
||||
rule(Kind::Transport, vec![block])
|
||||
.validate()
|
||||
.unwrap_err()
|
||||
.property,
|
||||
"actions"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn conditions_and_actions_are_checked() {
|
||||
let disclaimer = Action::AddDisclaimer {
|
||||
text: "Sent from Example Co.".into(),
|
||||
html: None,
|
||||
position: Position::Bottom,
|
||||
};
|
||||
let mut r = rule(Kind::Transport, vec![disclaimer]);
|
||||
assert!(r.validate().is_ok());
|
||||
r.conditions.push(Condition::Detected {
|
||||
detectors: vec![DetectorMin {
|
||||
id: "iban".into(),
|
||||
at_least: 1,
|
||||
}],
|
||||
});
|
||||
assert_eq!(r.validate().unwrap_err().property, "conditions");
|
||||
|
||||
let mut r = rule(
|
||||
Kind::Dlp,
|
||||
vec![Action::Block {
|
||||
notice: "No.".into(),
|
||||
}],
|
||||
);
|
||||
r.conditions = vec![Condition::Detected {
|
||||
detectors: vec![DetectorMin {
|
||||
id: "nope".into(),
|
||||
at_least: 1,
|
||||
}],
|
||||
}];
|
||||
assert!(r.validate().unwrap_err().reason.contains("nope"));
|
||||
r.conditions = vec![Condition::Pattern {
|
||||
pattern: "(".into(),
|
||||
at_least: 1,
|
||||
}];
|
||||
assert!(r.validate().is_err());
|
||||
r.conditions = vec![Condition::Words {
|
||||
words: vec![],
|
||||
at_least: 1,
|
||||
}];
|
||||
assert!(r.validate().is_err());
|
||||
r.exceptions = vec![Condition::Header {
|
||||
name: "X-Bad: yes".into(),
|
||||
contains: None,
|
||||
matches: None,
|
||||
}];
|
||||
r.conditions = vec![];
|
||||
assert_eq!(r.validate().unwrap_err().property, "exceptions");
|
||||
|
||||
let header = rule(
|
||||
Kind::Transport,
|
||||
vec![Action::AddHeader {
|
||||
name: "X-Tag".into(),
|
||||
value: "a\r\nBcc: x@y".into(),
|
||||
}],
|
||||
);
|
||||
assert!(header.validate().is_err());
|
||||
let redirect = rule(
|
||||
Kind::Transport,
|
||||
vec![Action::Redirect {
|
||||
addresses: vec!["nobody".into()],
|
||||
}],
|
||||
);
|
||||
assert!(redirect.validate().is_err());
|
||||
let mut unnamed = rule(
|
||||
Kind::Transport,
|
||||
vec![Action::RemoveHeader {
|
||||
name: "X-Tag".into(),
|
||||
}],
|
||||
);
|
||||
unnamed.name = " ".into();
|
||||
assert_eq!(unnamed.validate().unwrap_err().property, "name");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,109 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! An organization's own word lists and patterns (§2.3). Both count
|
||||
//! occurrences, not distinct values: "confidential" three times is three.
|
||||
|
||||
use aho_corasick::{AhoCorasick, AhoCorasickBuilder, MatchKind};
|
||||
use regex::{Regex, RegexBuilder};
|
||||
|
||||
/// How large a compiled pattern may grow. Keeps a rule someone writes from
|
||||
/// making every message slow to send.
|
||||
const PATTERN_SIZE_LIMIT: usize = 1 << 20;
|
||||
|
||||
/// Words and phrases, matched whole and ignoring case.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct WordList {
|
||||
matcher: AhoCorasick,
|
||||
}
|
||||
|
||||
impl WordList {
|
||||
/// Builds a list from words or phrases; empty entries are skipped.
|
||||
pub fn new<I, S>(words: I) -> Result<Self, String>
|
||||
where
|
||||
I: IntoIterator<Item = S>,
|
||||
S: AsRef<str>,
|
||||
{
|
||||
let words: Vec<String> = words
|
||||
.into_iter()
|
||||
.map(|w| w.as_ref().trim().to_lowercase())
|
||||
.filter(|w| !w.is_empty())
|
||||
.collect();
|
||||
if words.is_empty() {
|
||||
return Err("The list has no words".into());
|
||||
}
|
||||
AhoCorasickBuilder::new()
|
||||
.match_kind(MatchKind::LeftmostLongest)
|
||||
.build(&words)
|
||||
.map(|matcher| Self { matcher })
|
||||
.map_err(|err| err.to_string())
|
||||
}
|
||||
|
||||
/// How many times any word of the list appears in `text`.
|
||||
pub fn count(&self, text: &str) -> usize {
|
||||
let text = text.to_lowercase();
|
||||
self.matcher
|
||||
.find_iter(&text)
|
||||
.filter(|m| super::detectors::stands_alone(&text, m.start(), m.end()))
|
||||
.count()
|
||||
}
|
||||
}
|
||||
|
||||
/// An organization's regular expression.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct Pattern {
|
||||
regex: Regex,
|
||||
}
|
||||
|
||||
impl Pattern {
|
||||
/// Compiles `pattern`, or says why it can't be used. Matching ignores
|
||||
/// case unless the pattern turns that off with `(?-i)`.
|
||||
pub fn new(pattern: &str) -> Result<Self, String> {
|
||||
RegexBuilder::new(pattern)
|
||||
.case_insensitive(true)
|
||||
.size_limit(PATTERN_SIZE_LIMIT)
|
||||
.build()
|
||||
.map(|regex| Self { regex })
|
||||
.map_err(|err| err.to_string())
|
||||
}
|
||||
|
||||
/// How many times the pattern matches in `text`.
|
||||
pub fn count(&self, text: &str) -> usize {
|
||||
self.regex.find_iter(text).filter(|m| !m.is_empty()).count()
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn words_whole_and_any_case() {
|
||||
let list = WordList::new(["Project Falcon", "confidential", " "]).unwrap();
|
||||
assert_eq!(
|
||||
list.count(
|
||||
"CONFIDENTIAL: project falcon notes. Not confidentiality, not projectfalcon."
|
||||
),
|
||||
2
|
||||
);
|
||||
assert_eq!(list.count("Confidential, confidential and confidential"), 3);
|
||||
// Non-ASCII case folding
|
||||
let list = WordList::new(["GEHEIM", "Straße"]).unwrap();
|
||||
assert_eq!(list.count("streng geheim, STRASSE ist nicht Straße"), 2);
|
||||
assert!(WordList::new(["", " "]).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn patterns() {
|
||||
let pattern = Pattern::new(r"\bPRJ-\d{4}\b").unwrap();
|
||||
assert_eq!(pattern.count("prj-1234 and PRJ-5678, not PRJ-12"), 2);
|
||||
assert!(Pattern::new("(unclosed").is_err());
|
||||
// Too large to compile within the limit
|
||||
assert!(Pattern::new(r"\w{1000}\w{1000}\w{1000}").is_err());
|
||||
// Empty matches don't count
|
||||
assert_eq!(Pattern::new("x*").unwrap().count("abc"), 0);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,218 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! `inbuxa:MailRule/get` and `/set` under `urn:inbuxa:jmap`: mail flow rules
|
||||
//! and DLP rules (dlp-and-mail-flow-rules spec, §2.2). `kind` says which,
|
||||
//! and which permissions reach it. The set call's `reason` argument, if
|
||||
//! given, goes into the audit log with the change.
|
||||
|
||||
use crate::{
|
||||
object::{AnyId, JmapObject, JmapObjectId},
|
||||
request::deserialize::DeserializeArguments,
|
||||
};
|
||||
use jmap_tools::{Element, Key, Property};
|
||||
use std::{borrow::Cow, str::FromStr};
|
||||
use types::id::Id;
|
||||
|
||||
#[derive(Debug, Clone, Default)]
|
||||
pub struct MailRule;
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash)]
|
||||
pub enum MailRuleProperty {
|
||||
Id,
|
||||
Name,
|
||||
Description,
|
||||
/// `dlp` or `transport`.
|
||||
Kind,
|
||||
Enabled,
|
||||
/// Lower runs first.
|
||||
Priority,
|
||||
/// `outgoing`, `incoming` or `any`.
|
||||
Direction,
|
||||
Conditions,
|
||||
Exceptions,
|
||||
Actions,
|
||||
StopProcessing,
|
||||
CreatedBy,
|
||||
CreatedAt,
|
||||
UpdatedAt,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash)]
|
||||
pub enum MailRuleValue {
|
||||
Id(Id),
|
||||
}
|
||||
|
||||
impl Property for MailRuleProperty {
|
||||
fn try_parse(parent: Option<&Key<'_, Self>>, value: &str) -> Option<Self> {
|
||||
// Keys inside conditions and actions stay plain keys
|
||||
match parent {
|
||||
None => MailRuleProperty::parse(value),
|
||||
Some(_) => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn to_cow(&self) -> Cow<'static, str> {
|
||||
match self {
|
||||
MailRuleProperty::Id => "id",
|
||||
MailRuleProperty::Name => "name",
|
||||
MailRuleProperty::Description => "description",
|
||||
MailRuleProperty::Kind => "kind",
|
||||
MailRuleProperty::Enabled => "enabled",
|
||||
MailRuleProperty::Priority => "priority",
|
||||
MailRuleProperty::Direction => "direction",
|
||||
MailRuleProperty::Conditions => "conditions",
|
||||
MailRuleProperty::Exceptions => "exceptions",
|
||||
MailRuleProperty::Actions => "actions",
|
||||
MailRuleProperty::StopProcessing => "stopProcessing",
|
||||
MailRuleProperty::CreatedBy => "createdBy",
|
||||
MailRuleProperty::CreatedAt => "createdAt",
|
||||
MailRuleProperty::UpdatedAt => "updatedAt",
|
||||
}
|
||||
.into()
|
||||
}
|
||||
}
|
||||
|
||||
impl MailRuleProperty {
|
||||
fn parse(value: &str) -> Option<Self> {
|
||||
hashify::tiny_map!(value.as_bytes(),
|
||||
b"id" => MailRuleProperty::Id,
|
||||
b"name" => MailRuleProperty::Name,
|
||||
b"description" => MailRuleProperty::Description,
|
||||
b"kind" => MailRuleProperty::Kind,
|
||||
b"enabled" => MailRuleProperty::Enabled,
|
||||
b"priority" => MailRuleProperty::Priority,
|
||||
b"direction" => MailRuleProperty::Direction,
|
||||
b"conditions" => MailRuleProperty::Conditions,
|
||||
b"exceptions" => MailRuleProperty::Exceptions,
|
||||
b"actions" => MailRuleProperty::Actions,
|
||||
b"stopProcessing" => MailRuleProperty::StopProcessing,
|
||||
b"createdBy" => MailRuleProperty::CreatedBy,
|
||||
b"createdAt" => MailRuleProperty::CreatedAt,
|
||||
b"updatedAt" => MailRuleProperty::UpdatedAt,
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
impl FromStr for MailRuleProperty {
|
||||
type Err = ();
|
||||
|
||||
fn from_str(s: &str) -> Result<Self, Self::Err> {
|
||||
MailRuleProperty::parse(s).ok_or(())
|
||||
}
|
||||
}
|
||||
|
||||
impl Element for MailRuleValue {
|
||||
type Property = MailRuleProperty;
|
||||
|
||||
fn try_parse<P>(key: &Key<'_, Self::Property>, value: &str) -> Option<Self> {
|
||||
match key {
|
||||
Key::Property(MailRuleProperty::Id) => Id::from_str(value).ok().map(MailRuleValue::Id),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn to_cow(&self) -> Cow<'static, str> {
|
||||
match self {
|
||||
MailRuleValue::Id(id) => id.to_string().into(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The set call's own argument: why, for the audit log.
|
||||
#[derive(Debug, Clone, Default)]
|
||||
pub struct MailRuleSetArguments {
|
||||
pub reason: Option<String>,
|
||||
}
|
||||
|
||||
impl<'de> DeserializeArguments<'de> for MailRuleSetArguments {
|
||||
fn deserialize_argument<A>(&mut self, key: &str, map: &mut A) -> Result<(), A::Error>
|
||||
where
|
||||
A: serde::de::MapAccess<'de>,
|
||||
{
|
||||
if key == "reason" {
|
||||
self.reason = map.next_value()?;
|
||||
} else {
|
||||
let _ = map.next_value::<serde::de::IgnoredAny>()?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
impl JmapObject for MailRule {
|
||||
type Property = MailRuleProperty;
|
||||
|
||||
type Element = MailRuleValue;
|
||||
|
||||
type Id = Id;
|
||||
|
||||
type Filter = ();
|
||||
|
||||
type Comparator = ();
|
||||
|
||||
type GetArguments = ();
|
||||
|
||||
type SetArguments<'de> = MailRuleSetArguments;
|
||||
|
||||
type QueryArguments = ();
|
||||
|
||||
type CopyArguments = ();
|
||||
|
||||
type ParseArguments = ();
|
||||
|
||||
const ID_PROPERTY: Self::Property = MailRuleProperty::Id;
|
||||
}
|
||||
|
||||
impl From<Id> for MailRuleValue {
|
||||
fn from(id: Id) -> Self {
|
||||
MailRuleValue::Id(id)
|
||||
}
|
||||
}
|
||||
|
||||
impl JmapObjectId for MailRuleValue {
|
||||
fn as_id(&self) -> Option<Id> {
|
||||
match self {
|
||||
MailRuleValue::Id(id) => Some(*id),
|
||||
}
|
||||
}
|
||||
|
||||
fn as_any_id(&self) -> Option<AnyId> {
|
||||
match self {
|
||||
MailRuleValue::Id(id) => Some(AnyId::Id(*id)),
|
||||
}
|
||||
}
|
||||
|
||||
fn as_id_ref(&self) -> Option<&str> {
|
||||
None
|
||||
}
|
||||
|
||||
fn try_set_id(&mut self, new_id: AnyId) -> bool {
|
||||
if let AnyId::Id(id) = new_id {
|
||||
*self = MailRuleValue::Id(id);
|
||||
true
|
||||
} else {
|
||||
false
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl JmapObjectId for MailRuleProperty {
|
||||
fn as_id(&self) -> Option<Id> {
|
||||
None
|
||||
}
|
||||
|
||||
fn as_any_id(&self) -> Option<AnyId> {
|
||||
None
|
||||
}
|
||||
|
||||
fn as_id_ref(&self) -> Option<&str> {
|
||||
None
|
||||
}
|
||||
|
||||
fn try_set_id(&mut self, _: AnyId) -> bool {
|
||||
false
|
||||
}
|
||||
}
|
||||
@@ -28,6 +28,7 @@ pub mod inbuxa_data_inventory; // inbuxa: personal-data catalog
|
||||
pub mod inbuxa_inventory_snapshot; // inbuxa: personal-data catalog
|
||||
pub mod inbuxa_audit; // inbuxa: the audit log
|
||||
pub mod inbuxa_legal_hold; // inbuxa: legal hold
|
||||
pub mod inbuxa_mail_rule; // inbuxa: DLP and mail flow rules
|
||||
pub mod inbuxa_hold_export; // inbuxa: legal hold exports
|
||||
pub mod inbuxa_explanation; // inbuxa: "Explain this" with the local model
|
||||
pub mod inbuxa_protocol_policy; // inbuxa: legacy protocols off
|
||||
|
||||
@@ -82,6 +82,9 @@ impl Response<'_> {
|
||||
GetResponseMethod::LegalHold(response) => {
|
||||
response.eval_jptr(path, &mut results)
|
||||
}
|
||||
GetResponseMethod::MailRule(response) => {
|
||||
response.eval_jptr(path, &mut results)
|
||||
}
|
||||
GetResponseMethod::HoldExport(response) => {
|
||||
response.eval_jptr(path, &mut results)
|
||||
}
|
||||
|
||||
@@ -53,6 +53,7 @@ impl Response<'_> {
|
||||
GetRequestMethod::AuditSettings(request) => request.resolve_references(self)?,
|
||||
GetRequestMethod::AccountLock(request) => request.resolve_references(self)?,
|
||||
GetRequestMethod::LegalHold(request) => request.resolve_references(self)?,
|
||||
GetRequestMethod::MailRule(request) => request.resolve_references(self)?,
|
||||
GetRequestMethod::HoldExport(request) => request.resolve_references(self)?,
|
||||
GetRequestMethod::ProtocolPolicy(request) => request.resolve_references(self)?,
|
||||
GetRequestMethod::TenantProtocolPolicy(request) => {
|
||||
@@ -122,6 +123,9 @@ impl Response<'_> {
|
||||
SetRequestMethod::LegalHold(request) => {
|
||||
request.resolve_references(self, 1, false)?
|
||||
}
|
||||
SetRequestMethod::MailRule(request) => {
|
||||
request.resolve_references(self, 1, false)?
|
||||
}
|
||||
SetRequestMethod::HoldExport(request) => {
|
||||
request.resolve_references(self, 1, false)?
|
||||
}
|
||||
|
||||
@@ -65,6 +65,8 @@ pub enum MethodObject {
|
||||
LegalHold,
|
||||
HoldExport,
|
||||
ProtocolPolicy,
|
||||
// inbuxa: DLP and mail flow rules
|
||||
MailRule,
|
||||
TenantProtocolPolicy,
|
||||
}
|
||||
|
||||
@@ -102,7 +104,8 @@ impl MethodObject {
|
||||
| MethodObject::AuditVerification
|
||||
| MethodObject::AccountLock
|
||||
| MethodObject::LegalHold
|
||||
| MethodObject::HoldExport => Capability::Inbuxa,
|
||||
| MethodObject::HoldExport
|
||||
| MethodObject::MailRule => Capability::Inbuxa,
|
||||
MethodObject::ProtocolPolicy => Capability::Inbuxa,
|
||||
MethodObject::TenantProtocolPolicy => Capability::Inbuxa,
|
||||
}
|
||||
@@ -296,6 +299,8 @@ impl MethodName {
|
||||
(MethodFunction::Set, MethodObject::AccountLock) => "inbuxa:AccountLock/set",
|
||||
(MethodFunction::Get, MethodObject::LegalHold) => "inbuxa:LegalHold/get",
|
||||
(MethodFunction::Set, MethodObject::LegalHold) => "inbuxa:LegalHold/set",
|
||||
(MethodFunction::Get, MethodObject::MailRule) => "inbuxa:MailRule/get",
|
||||
(MethodFunction::Set, MethodObject::MailRule) => "inbuxa:MailRule/set",
|
||||
(MethodFunction::Get, MethodObject::HoldExport) => "inbuxa:HoldExport/get",
|
||||
(MethodFunction::Set, MethodObject::HoldExport) => "inbuxa:HoldExport/set",
|
||||
(MethodFunction::Set, MethodObject::AuditVerification) => {
|
||||
@@ -448,6 +453,8 @@ impl MethodName {
|
||||
"inbuxa:AccountLock/set" => (MethodObject::AccountLock, MethodFunction::Set),
|
||||
"inbuxa:LegalHold/get" => (MethodObject::LegalHold, MethodFunction::Get),
|
||||
"inbuxa:LegalHold/set" => (MethodObject::LegalHold, MethodFunction::Set),
|
||||
"inbuxa:MailRule/get" => (MethodObject::MailRule, MethodFunction::Get),
|
||||
"inbuxa:MailRule/set" => (MethodObject::MailRule, MethodFunction::Set),
|
||||
"inbuxa:HoldExport/get" => (MethodObject::HoldExport, MethodFunction::Get),
|
||||
"inbuxa:HoldExport/set" => (MethodObject::HoldExport, MethodFunction::Set),
|
||||
"inbuxa:AuditVerification/set" => (MethodObject::AuditVerification, MethodFunction::Set),
|
||||
@@ -518,6 +525,7 @@ impl Display for MethodObject {
|
||||
MethodObject::AuditVerification => "inbuxa:AuditVerification",
|
||||
MethodObject::AccountLock => "inbuxa:AccountLock",
|
||||
MethodObject::LegalHold => "inbuxa:LegalHold",
|
||||
MethodObject::MailRule => "inbuxa:MailRule",
|
||||
MethodObject::HoldExport => "inbuxa:HoldExport",
|
||||
MethodObject::ProtocolPolicy => "inbuxa:ProtocolPolicy",
|
||||
MethodObject::TenantProtocolPolicy => "inbuxa:TenantProtocolPolicy",
|
||||
|
||||
@@ -123,6 +123,7 @@ pub enum GetRequestMethod {
|
||||
AuditSettings(Box<GetRequest<crate::object::inbuxa_audit::AuditSettings>>),
|
||||
AccountLock(Box<GetRequest<crate::object::inbuxa_account_lock::AccountLock>>),
|
||||
LegalHold(Box<GetRequest<crate::object::inbuxa_legal_hold::LegalHold>>),
|
||||
MailRule(Box<GetRequest<crate::object::inbuxa_mail_rule::MailRule>>),
|
||||
HoldExport(Box<GetRequest<crate::object::inbuxa_hold_export::HoldExport>>),
|
||||
ProtocolPolicy(Box<GetRequest<crate::object::inbuxa_protocol_policy::ProtocolPolicy>>),
|
||||
TenantProtocolPolicy(
|
||||
@@ -158,6 +159,7 @@ pub enum SetRequestMethod<'x> {
|
||||
AuditVerification(Box<SetRequest<'x, crate::object::inbuxa_audit::AuditVerification>>),
|
||||
AccountLock(Box<SetRequest<'x, crate::object::inbuxa_account_lock::AccountLock>>),
|
||||
LegalHold(Box<SetRequest<'x, crate::object::inbuxa_legal_hold::LegalHold>>),
|
||||
MailRule(Box<SetRequest<'x, crate::object::inbuxa_mail_rule::MailRule>>),
|
||||
HoldExport(Box<SetRequest<'x, crate::object::inbuxa_hold_export::HoldExport>>),
|
||||
ProtocolPolicy(Box<SetRequest<'x, crate::object::inbuxa_protocol_policy::ProtocolPolicy>>),
|
||||
TenantProtocolPolicy(
|
||||
|
||||
@@ -609,6 +609,21 @@ impl<'de> Visitor<'de> for CallVisitor {
|
||||
return Err(de::Error::invalid_length(1, &self));
|
||||
}
|
||||
},
|
||||
// inbuxa: DLP and mail flow rules
|
||||
(MethodFunction::Get, MethodObject::MailRule) => match seq.next_element() {
|
||||
Ok(Some(value)) => RequestMethod::Get(GetRequestMethod::MailRule(value)),
|
||||
Err(err) => RequestMethod::invalid(err),
|
||||
Ok(None) => {
|
||||
return Err(de::Error::invalid_length(1, &self));
|
||||
}
|
||||
},
|
||||
(MethodFunction::Set, MethodObject::MailRule) => match seq.next_element() {
|
||||
Ok(Some(value)) => RequestMethod::Set(SetRequestMethod::MailRule(value)),
|
||||
Err(err) => RequestMethod::invalid(err),
|
||||
Ok(None) => {
|
||||
return Err(de::Error::invalid_length(1, &self));
|
||||
}
|
||||
},
|
||||
// inbuxa: legal hold
|
||||
(MethodFunction::Get, MethodObject::LegalHold) => match seq.next_element() {
|
||||
Ok(Some(value)) => RequestMethod::Get(GetRequestMethod::LegalHold(value)),
|
||||
|
||||
@@ -110,6 +110,7 @@ pub enum GetResponseMethod {
|
||||
AuditSettings(GetResponse<crate::object::inbuxa_audit::AuditSettings>),
|
||||
AccountLock(GetResponse<crate::object::inbuxa_account_lock::AccountLock>),
|
||||
LegalHold(GetResponse<crate::object::inbuxa_legal_hold::LegalHold>),
|
||||
MailRule(GetResponse<crate::object::inbuxa_mail_rule::MailRule>),
|
||||
HoldExport(GetResponse<crate::object::inbuxa_hold_export::HoldExport>),
|
||||
ProtocolPolicy(GetResponse<crate::object::inbuxa_protocol_policy::ProtocolPolicy>),
|
||||
TenantProtocolPolicy(
|
||||
@@ -145,6 +146,7 @@ pub enum SetResponseMethod {
|
||||
AuditVerification(Box<SetResponse<crate::object::inbuxa_audit::AuditVerification>>),
|
||||
AccountLock(Box<SetResponse<crate::object::inbuxa_account_lock::AccountLock>>),
|
||||
LegalHold(Box<SetResponse<crate::object::inbuxa_legal_hold::LegalHold>>),
|
||||
MailRule(Box<SetResponse<crate::object::inbuxa_mail_rule::MailRule>>),
|
||||
HoldExport(Box<SetResponse<crate::object::inbuxa_hold_export::HoldExport>>),
|
||||
Explanation(Box<SetResponse<crate::object::inbuxa_explanation::Explanation>>),
|
||||
ProtocolPolicy(Box<SetResponse<crate::object::inbuxa_protocol_policy::ProtocolPolicy>>),
|
||||
@@ -799,6 +801,18 @@ impl<'x> From<SetResponse<crate::object::inbuxa_account_lock::AccountLock>> for
|
||||
}
|
||||
|
||||
// inbuxa: legal hold
|
||||
impl<'x> From<GetResponse<crate::object::inbuxa_mail_rule::MailRule>> for ResponseMethod<'x> {
|
||||
fn from(value: GetResponse<crate::object::inbuxa_mail_rule::MailRule>) -> Self {
|
||||
ResponseMethod::Get(GetResponseMethod::MailRule(value))
|
||||
}
|
||||
}
|
||||
|
||||
impl<'x> From<SetResponse<crate::object::inbuxa_mail_rule::MailRule>> for ResponseMethod<'x> {
|
||||
fn from(value: SetResponse<crate::object::inbuxa_mail_rule::MailRule>) -> Self {
|
||||
ResponseMethod::Set(SetResponseMethod::MailRule(Box::new(value)))
|
||||
}
|
||||
}
|
||||
|
||||
impl<'x> From<GetResponse<crate::object::inbuxa_legal_hold::LegalHold>> for ResponseMethod<'x> {
|
||||
fn from(value: GetResponse<crate::object::inbuxa_legal_hold::LegalHold>) -> Self {
|
||||
ResponseMethod::Get(GetResponseMethod::LegalHold(value))
|
||||
|
||||
@@ -103,6 +103,16 @@ impl JmapAuthorization for AccessToken {
|
||||
// inbuxa: account lock (AL-12)
|
||||
GetRequestMethod::AccountLock(_) => Permission::SysAccountLockGet,
|
||||
GetRequestMethod::LegalHold(_) => Permission::SysLegalHoldGet,
|
||||
// inbuxa: DLP and mail flow rules share an object; either
|
||||
// permission reaches it, and the handler shows each kind
|
||||
// only to those who may see it
|
||||
GetRequestMethod::MailRule(_) => {
|
||||
if self.has_permission(Permission::SysMailRuleGet) {
|
||||
Permission::SysMailRuleGet
|
||||
} else {
|
||||
Permission::SysDlpPolicyGet
|
||||
}
|
||||
}
|
||||
GetRequestMethod::HoldExport(_) => Permission::SysLegalHoldExport,
|
||||
// inbuxa: legacy protocols off. It takes listeners away and
|
||||
// puts them back, so it takes the listener's permissions
|
||||
@@ -247,6 +257,19 @@ impl JmapAuthorization for AccessToken {
|
||||
Permission::SysLegalHoldUpdate,
|
||||
Permission::SysLegalHoldUpdate,
|
||||
),
|
||||
// inbuxa: DLP and mail flow rules: either change
|
||||
// permission gets in; the handler checks each rule's kind
|
||||
SetRequestMethod::MailRule(_) => {
|
||||
if self.has_permission(Permission::SysMailRuleUpdate)
|
||||
|| self.has_permission(Permission::SysDlpPolicyUpdate)
|
||||
{
|
||||
Ok(())
|
||||
} else {
|
||||
Err(trc::JmapEvent::Forbidden
|
||||
.into_err()
|
||||
.details("You are not authorized to change mail rules"))
|
||||
}
|
||||
}
|
||||
// inbuxa: LH-12, exporting held data
|
||||
SetRequestMethod::HoldExport(s) => validate_set(
|
||||
s,
|
||||
@@ -407,6 +430,7 @@ impl JmapAuthorization for AccessToken {
|
||||
| MethodObject::AccountLock
|
||||
| MethodObject::LegalHold
|
||||
| MethodObject::HoldExport
|
||||
| MethodObject::MailRule
|
||||
| MethodObject::ProtocolPolicy
|
||||
| MethodObject::TenantProtocolPolicy => Permission::JmapEmailChanges,
|
||||
// inbuxa: x:MaskedEmail/changes reads what /get reads
|
||||
|
||||
@@ -279,6 +279,9 @@ impl RequestHandler for Server {
|
||||
SetResponseMethod::LegalHold(set_response) => {
|
||||
set_response.update_created_ids(&mut response);
|
||||
}
|
||||
SetResponseMethod::MailRule(set_response) => {
|
||||
set_response.update_created_ids(&mut response);
|
||||
}
|
||||
SetResponseMethod::HoldExport(set_response) => {
|
||||
set_response.update_created_ids(&mut response);
|
||||
}
|
||||
@@ -486,6 +489,11 @@ impl RequestHandler for Server {
|
||||
resolve_account_id(&mut req.account_id, method_name.obj, access_token)?;
|
||||
crate::inbuxa::legal_hold::get(self, *req).await?.into()
|
||||
}
|
||||
// inbuxa: DLP and mail flow rules
|
||||
GetRequestMethod::MailRule(mut req) => {
|
||||
resolve_account_id(&mut req.account_id, method_name.obj, access_token)?;
|
||||
crate::inbuxa::mail_rule::get(self, access_token, *req).await?.into()
|
||||
}
|
||||
// inbuxa: the audit log (AU-9)
|
||||
GetRequestMethod::AuditEvent(mut req) => {
|
||||
resolve_account_id(&mut req.account_id, method_name.obj, access_token)?;
|
||||
@@ -910,6 +918,22 @@ impl RequestHandler for Server {
|
||||
.await?
|
||||
.into()
|
||||
}
|
||||
SetRequestMethod::MailRule(mut req) => {
|
||||
resolve_account_id(&mut req.account_id, method_name.obj, access_token)?;
|
||||
let reason = req.arguments.reason.clone();
|
||||
crate::inbuxa::audit::recorded(
|
||||
self,
|
||||
access_token,
|
||||
session,
|
||||
&method_name.obj.to_string(),
|
||||
None,
|
||||
reason,
|
||||
*req,
|
||||
|req| Box::pin(crate::inbuxa::mail_rule::set(self, access_token, req)),
|
||||
)
|
||||
.await?
|
||||
.into()
|
||||
}
|
||||
SetRequestMethod::AuditExport(mut req) => {
|
||||
resolve_account_id(&mut req.account_id, method_name.obj, access_token)?;
|
||||
crate::inbuxa::audit_log::export_set(self, access_token, session, *req)
|
||||
|
||||
@@ -429,6 +429,7 @@ impl IntermediateChangesResponse {
|
||||
| MethodObject::AccountLock
|
||||
| MethodObject::LegalHold
|
||||
| MethodObject::HoldExport
|
||||
| MethodObject::MailRule
|
||||
| MethodObject::ProtocolPolicy
|
||||
| MethodObject::TenantProtocolPolicy
|
||||
| MethodObject::Registry(_) => unreachable!(),
|
||||
|
||||
@@ -0,0 +1,327 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! `inbuxa:MailRule` (dlp-and-mail-flow-rules spec, §2.2, §2.8): mail flow
|
||||
//! rules and DLP rules. One object, two kinds, each with its own
|
||||
//! permissions: `sysMailRuleGet`/`Update` for transport rules,
|
||||
//! `sysDlpPolicyGet`/`Update` for DLP rules. Rules are the server's: nobody
|
||||
//! in a tenant reaches them (settled answer 3). The request layer records
|
||||
//! every change in the audit log.
|
||||
|
||||
use common::{Server, auth::AccessToken};
|
||||
use inbuxa_features::mailflow::rules::{self, Kind, Rule};
|
||||
use jmap_proto::{
|
||||
error::set::SetError,
|
||||
method::{
|
||||
get::{GetRequest, GetResponse},
|
||||
set::{SetRequest, SetResponse},
|
||||
},
|
||||
object::inbuxa_mail_rule::{MailRule, MailRuleProperty as P, MailRuleValue},
|
||||
request::IntoValid,
|
||||
types::date::UTCDate,
|
||||
};
|
||||
use jmap_tools::{Key, Map, Property, Value};
|
||||
use registry::schema::enums::Permission;
|
||||
use std::borrow::Cow;
|
||||
use store::write::now;
|
||||
use types::id::Id;
|
||||
|
||||
type RValue = Value<'static, P, MailRuleValue>;
|
||||
|
||||
const ALL: &[P] = &[
|
||||
P::Id,
|
||||
P::Name,
|
||||
P::Description,
|
||||
P::Kind,
|
||||
P::Enabled,
|
||||
P::Priority,
|
||||
P::Direction,
|
||||
P::Conditions,
|
||||
P::Exceptions,
|
||||
P::Actions,
|
||||
P::StopProcessing,
|
||||
P::CreatedBy,
|
||||
P::CreatedAt,
|
||||
P::UpdatedAt,
|
||||
];
|
||||
|
||||
/// Properties the server sets; a client that sends them is refused.
|
||||
const SERVER_SET: &[P] = &[P::Id, P::CreatedBy, P::CreatedAt, P::UpdatedAt];
|
||||
|
||||
fn can_see(access_token: &AccessToken, kind: Kind) -> bool {
|
||||
access_token.has_permission(match kind {
|
||||
Kind::Dlp => Permission::SysDlpPolicyGet,
|
||||
Kind::Transport => Permission::SysMailRuleGet,
|
||||
})
|
||||
}
|
||||
|
||||
fn can_change(access_token: &AccessToken, kind: Kind) -> bool {
|
||||
access_token.has_permission(match kind {
|
||||
Kind::Dlp => Permission::SysDlpPolicyUpdate,
|
||||
Kind::Transport => Permission::SysMailRuleUpdate,
|
||||
})
|
||||
}
|
||||
|
||||
fn server_level(access_token: &AccessToken) -> trc::Result<()> {
|
||||
if access_token.tenant_id().is_some() {
|
||||
Err(trc::JmapEvent::Forbidden
|
||||
.into_err()
|
||||
.details("Mail rules are the server's."))
|
||||
} else {
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
fn json_to_value(json: serde_json::Value) -> RValue {
|
||||
match json {
|
||||
serde_json::Value::Null => Value::Null,
|
||||
serde_json::Value::Bool(b) => Value::Bool(b),
|
||||
serde_json::Value::Number(n) => {
|
||||
if let Some(n) = n.as_u64() {
|
||||
Value::Number(n.into())
|
||||
} else if let Some(n) = n.as_i64() {
|
||||
Value::Number(n.into())
|
||||
} else {
|
||||
Value::Number(n.as_f64().unwrap_or_default().into())
|
||||
}
|
||||
}
|
||||
serde_json::Value::String(s) => Value::Str(Cow::Owned(s)),
|
||||
serde_json::Value::Array(items) => {
|
||||
Value::Array(items.into_iter().map(json_to_value).collect())
|
||||
}
|
||||
serde_json::Value::Object(map) => {
|
||||
let mut out = Map::with_capacity(map.len());
|
||||
for (key, value) in map {
|
||||
out.insert_unchecked(Key::Owned(key), json_to_value(value));
|
||||
}
|
||||
Value::Object(out)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn date(seconds: u64) -> RValue {
|
||||
Value::Str(UTCDate::from_timestamp(seconds as i64).to_string().into())
|
||||
}
|
||||
|
||||
fn to_value(rule: &Rule, properties: &[P]) -> RValue {
|
||||
let json = serde_json::to_value(rule).unwrap_or_default();
|
||||
let mut out = Map::with_capacity(properties.len());
|
||||
for property in properties {
|
||||
let value = match property {
|
||||
P::Id => Value::Element(MailRuleValue::Id(Id::from(rule.id))),
|
||||
P::CreatedAt => date(rule.created_at),
|
||||
P::UpdatedAt => date(rule.updated_at),
|
||||
other => json
|
||||
.get(other.to_cow().as_ref())
|
||||
.cloned()
|
||||
.map_or(Value::Null, json_to_value),
|
||||
};
|
||||
out.insert_unchecked(Key::Property(property.clone()), value);
|
||||
}
|
||||
Value::Object(out)
|
||||
}
|
||||
|
||||
/// A rule as sent: its JSON object, top-level keys only those a client may
|
||||
/// set.
|
||||
fn client_json(value: Value<'_, P, MailRuleValue>) -> Result<serde_json::Map<String, serde_json::Value>, SetError<P>> {
|
||||
let mut map = serde_json::Map::new();
|
||||
for (key, value) in value.into_expanded_object() {
|
||||
match &key {
|
||||
Key::Property(p) if SERVER_SET.contains(p) => {
|
||||
return Err(SetError::invalid_properties()
|
||||
.with_property(p.clone())
|
||||
.with_description("The server sets this."));
|
||||
}
|
||||
Key::Property(p) => {
|
||||
map.insert(p.to_cow().into_owned(), value.into());
|
||||
}
|
||||
_ => {
|
||||
return Err(SetError::invalid_properties().with_property(key.clone().into_owned()));
|
||||
}
|
||||
}
|
||||
}
|
||||
Ok(map)
|
||||
}
|
||||
|
||||
fn parse(json: serde_json::Map<String, serde_json::Value>) -> Result<Rule, SetError<P>> {
|
||||
let rule: Rule = serde_json::from_value(serde_json::Value::Object(json)).map_err(|err| {
|
||||
SetError::invalid_properties().with_description(format!("Not a valid rule: {err}"))
|
||||
})?;
|
||||
rule.validate().map_err(|invalid| {
|
||||
let property = invalid.property.parse::<P>().unwrap_or(P::Name);
|
||||
SetError::invalid_properties()
|
||||
.with_property(property)
|
||||
.with_description(invalid.reason)
|
||||
})?;
|
||||
Ok(rule)
|
||||
}
|
||||
|
||||
fn forbidden(kind: Kind) -> SetError<P> {
|
||||
SetError::forbidden().with_description(match kind {
|
||||
Kind::Dlp => "Changing DLP rules needs the permission to change DLP rules.",
|
||||
Kind::Transport => "Changing mail flow rules needs the permission to change them.",
|
||||
})
|
||||
}
|
||||
|
||||
fn rule_id(id: Id) -> Option<u32> {
|
||||
u32::try_from(id.id()).ok()
|
||||
}
|
||||
|
||||
/// `inbuxa:MailRule/get`: the rules the caller may see, in the order they
|
||||
/// run.
|
||||
pub async fn get(
|
||||
server: &Server,
|
||||
access_token: &AccessToken,
|
||||
mut request: GetRequest<MailRule>,
|
||||
) -> trc::Result<GetResponse<MailRule>> {
|
||||
server_level(access_token)?;
|
||||
let properties = request.unwrap_properties(ALL);
|
||||
let (ids, not_found) = request.unwrap_ids(server.core.jmap.get_max_objects)?;
|
||||
let mut response = GetResponse {
|
||||
account_id: request.account_id.into(),
|
||||
state: None,
|
||||
list: Vec::new(),
|
||||
not_found,
|
||||
};
|
||||
let visible: Vec<Rule> = rules::all(server.store())
|
||||
.await?
|
||||
.into_iter()
|
||||
.filter(|rule| can_see(access_token, rule.kind))
|
||||
.collect();
|
||||
match ids {
|
||||
None => {
|
||||
response.list = visible
|
||||
.iter()
|
||||
.map(|rule| to_value(rule, &properties))
|
||||
.collect()
|
||||
}
|
||||
Some(ids) => {
|
||||
for id in ids {
|
||||
match rule_id(id).and_then(|id| visible.iter().find(|r| r.id == id)) {
|
||||
Some(rule) => response.list.push(to_value(rule, &properties)),
|
||||
None => response.push_not_found(id),
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
Ok(response)
|
||||
}
|
||||
|
||||
/// `inbuxa:MailRule/set`: create, change or delete rules, each checked
|
||||
/// against the permissions for its kind (and, on a change of kind, both).
|
||||
pub async fn set(
|
||||
server: &Server,
|
||||
access_token: &AccessToken,
|
||||
mut request: SetRequest<'_, MailRule>,
|
||||
) -> trc::Result<SetResponse<MailRule>> {
|
||||
server_level(access_token)?;
|
||||
let mut response = SetResponse::from_request(&request, server.core.jmap.set_max_objects)?;
|
||||
let data = server.store();
|
||||
let actor = server.audit_actor(access_token).await;
|
||||
|
||||
for (client_id, value) in request.unwrap_create() {
|
||||
let rule = match client_json(value).and_then(parse) {
|
||||
Ok(rule) => rule,
|
||||
Err(error) => {
|
||||
response.not_created.append(client_id, error);
|
||||
continue;
|
||||
}
|
||||
};
|
||||
if !can_change(access_token, rule.kind) {
|
||||
response.not_created.append(client_id, forbidden(rule.kind));
|
||||
continue;
|
||||
}
|
||||
let at = now();
|
||||
let rule = Rule {
|
||||
created_by: actor.name.clone(),
|
||||
created_at: at,
|
||||
updated_at: at,
|
||||
..rule
|
||||
};
|
||||
let id = rules::create(data, &rule).await?;
|
||||
let mut out = Map::with_capacity(1);
|
||||
out.insert_unchecked(
|
||||
Key::Property(P::Id),
|
||||
Value::Element(MailRuleValue::Id(Id::from(id))),
|
||||
);
|
||||
response.created.insert(client_id, Value::Object(out));
|
||||
}
|
||||
|
||||
for (id, value) in request.unwrap_update().into_valid() {
|
||||
let Some(current) = (match rule_id(id) {
|
||||
Some(rule_id) => rules::get(data, rule_id).await?,
|
||||
None => None,
|
||||
}) else {
|
||||
response.not_updated.append(id, SetError::not_found());
|
||||
continue;
|
||||
};
|
||||
if !can_see(access_token, current.kind) {
|
||||
response.not_updated.append(id, SetError::not_found());
|
||||
continue;
|
||||
}
|
||||
if !can_change(access_token, current.kind) {
|
||||
response.not_updated.append(id, forbidden(current.kind));
|
||||
continue;
|
||||
}
|
||||
// The stored rule, with each property sent replacing its own
|
||||
let mut json = match serde_json::to_value(¤t) {
|
||||
Ok(serde_json::Value::Object(map)) => map,
|
||||
_ => serde_json::Map::new(),
|
||||
};
|
||||
let changes = match client_json(value) {
|
||||
Ok(changes) => changes,
|
||||
Err(error) => {
|
||||
response.not_updated.append(id, error);
|
||||
continue;
|
||||
}
|
||||
};
|
||||
json.extend(changes);
|
||||
let next = match parse(json) {
|
||||
Ok(next) => next,
|
||||
Err(error) => {
|
||||
response.not_updated.append(id, error);
|
||||
continue;
|
||||
}
|
||||
};
|
||||
if next.kind != current.kind && !can_change(access_token, next.kind) {
|
||||
response.not_updated.append(id, forbidden(next.kind));
|
||||
continue;
|
||||
}
|
||||
let next = Rule {
|
||||
id: current.id,
|
||||
created_by: current.created_by.clone(),
|
||||
created_at: current.created_at,
|
||||
updated_at: now(),
|
||||
..next
|
||||
};
|
||||
if next != current {
|
||||
rules::update(data, &next).await?;
|
||||
}
|
||||
response.updated.append(id, None);
|
||||
}
|
||||
|
||||
for id in request.unwrap_destroy().into_valid() {
|
||||
let Some(current) = (match rule_id(id) {
|
||||
Some(rule_id) => rules::get(data, rule_id).await?,
|
||||
None => None,
|
||||
}) else {
|
||||
response.not_destroyed.append(id, SetError::not_found());
|
||||
continue;
|
||||
};
|
||||
if !can_see(access_token, current.kind) {
|
||||
response.not_destroyed.append(id, SetError::not_found());
|
||||
continue;
|
||||
}
|
||||
if !can_change(access_token, current.kind) {
|
||||
response.not_destroyed.append(id, forbidden(current.kind));
|
||||
continue;
|
||||
}
|
||||
rules::delete(data, current.id).await?;
|
||||
response.destroyed.push(id);
|
||||
}
|
||||
|
||||
Ok(response)
|
||||
}
|
||||
@@ -10,6 +10,7 @@
|
||||
pub mod access;
|
||||
pub mod account_lock;
|
||||
pub mod legal_hold;
|
||||
pub mod mail_rule;
|
||||
pub mod hold_export;
|
||||
pub mod hold_export_api;
|
||||
pub mod audit;
|
||||
|
||||
@@ -1748,6 +1748,13 @@ pub enum Permission {
|
||||
SysLegalHoldExport = 672,
|
||||
// inbuxa: personal-data catalog, the data inventory and compliance overview
|
||||
SysComplianceGet = 673,
|
||||
// inbuxa: DLP and mail flow rules
|
||||
SysMailRuleGet = 674,
|
||||
SysMailRuleUpdate = 675,
|
||||
SysDlpPolicyGet = 676,
|
||||
SysDlpPolicyUpdate = 677,
|
||||
SysDlpReviewGet = 678,
|
||||
SysDlpReviewUpdate = 679,
|
||||
SysAccountGet = 219,
|
||||
SysAccountCreate = 220,
|
||||
SysAccountUpdate = 221,
|
||||
|
||||
@@ -7091,6 +7091,12 @@ impl EnumImpl for Permission {
|
||||
b"sysLegalHoldUpdate" => Permission::SysLegalHoldUpdate,
|
||||
b"sysLegalHoldExport" => Permission::SysLegalHoldExport,
|
||||
b"sysComplianceGet" => Permission::SysComplianceGet,
|
||||
b"sysMailRuleGet" => Permission::SysMailRuleGet,
|
||||
b"sysMailRuleUpdate" => Permission::SysMailRuleUpdate,
|
||||
b"sysDlpPolicyGet" => Permission::SysDlpPolicyGet,
|
||||
b"sysDlpPolicyUpdate" => Permission::SysDlpPolicyUpdate,
|
||||
b"sysDlpReviewGet" => Permission::SysDlpReviewGet,
|
||||
b"sysDlpReviewUpdate" => Permission::SysDlpReviewUpdate,
|
||||
b"sysAccountGet" => Permission::SysAccountGet,
|
||||
b"sysAccountCreate" => Permission::SysAccountCreate,
|
||||
b"sysAccountUpdate" => Permission::SysAccountUpdate,
|
||||
@@ -7781,6 +7787,12 @@ impl EnumImpl for Permission {
|
||||
Permission::SysLegalHoldUpdate => "sysLegalHoldUpdate",
|
||||
Permission::SysLegalHoldExport => "sysLegalHoldExport",
|
||||
Permission::SysComplianceGet => "sysComplianceGet",
|
||||
Permission::SysMailRuleGet => "sysMailRuleGet",
|
||||
Permission::SysMailRuleUpdate => "sysMailRuleUpdate",
|
||||
Permission::SysDlpPolicyGet => "sysDlpPolicyGet",
|
||||
Permission::SysDlpPolicyUpdate => "sysDlpPolicyUpdate",
|
||||
Permission::SysDlpReviewGet => "sysDlpReviewGet",
|
||||
Permission::SysDlpReviewUpdate => "sysDlpReviewUpdate",
|
||||
Permission::SysAccountGet => "sysAccountGet",
|
||||
Permission::SysAccountCreate => "sysAccountCreate",
|
||||
Permission::SysAccountUpdate => "sysAccountUpdate",
|
||||
@@ -8464,6 +8476,12 @@ impl EnumImpl for Permission {
|
||||
671 => Some(Permission::SysLegalHoldUpdate),
|
||||
672 => Some(Permission::SysLegalHoldExport),
|
||||
673 => Some(Permission::SysComplianceGet),
|
||||
674 => Some(Permission::SysMailRuleGet),
|
||||
675 => Some(Permission::SysMailRuleUpdate),
|
||||
676 => Some(Permission::SysDlpPolicyGet),
|
||||
677 => Some(Permission::SysDlpPolicyUpdate),
|
||||
678 => Some(Permission::SysDlpReviewGet),
|
||||
679 => Some(Permission::SysDlpReviewUpdate),
|
||||
219 => Some(Permission::SysAccountGet),
|
||||
220 => Some(Permission::SysAccountCreate),
|
||||
221 => Some(Permission::SysAccountUpdate),
|
||||
@@ -8908,7 +8926,7 @@ impl EnumImpl for Permission {
|
||||
}
|
||||
}
|
||||
|
||||
const COUNT: usize = 674;
|
||||
const COUNT: usize = 680;
|
||||
}
|
||||
|
||||
impl serde::Serialize for Permission {
|
||||
|
||||
@@ -81,7 +81,7 @@ fn legacy_setting(name: &str, is_set: impl Fn(&str) -> bool) -> Option<String> {
|
||||
#[macro_export]
|
||||
macro_rules! brand_version {
|
||||
() => {
|
||||
"2026.9.28.6"
|
||||
"2026.9.28.5"
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
@@ -139,6 +139,14 @@ DLP adds **detectors**. Each counts what it finds, and a rule sets a minimum
|
||||
either side, in the languages where the identifier is used ("passport",
|
||||
"Reisepass", "pasaporte"...).
|
||||
|
||||
A check that about one random number in ten passes (Luhn, mod 10, mod 11) is
|
||||
too weak for a bare run of digits: invoice and phone numbers would match. So
|
||||
a checked identifier that is only digits (SSN, SIN, NHS, TFN, Medicare…)
|
||||
counts alone in the written form it's issued in (`536-22-1234`,
|
||||
`130 692 544`, `943 476 5919`), and as bare digits only beside a word. ABA
|
||||
routing numbers and NPIs are never written with separators, so they always
|
||||
need a word. (Refinement made while building phase 2, 2026-09-28.)
|
||||
|
||||
The catalog (settled answer 6: the recognized, protected identifiers, not a
|
||||
chosen few). Each row is one table entry and one check function in
|
||||
`crates/features/src/mailflow/detectors/`:
|
||||
@@ -157,10 +165,10 @@ chosen few). Each row is one table entry and one check function in
|
||||
| US | Social Security number | Checked | `AAA-GG-SSSS`, or nine digits with a word; never area 000, 666 or 9xx, group 00, serial 0000 |
|
||||
| US | ITIN | Checked | 9XX-GG-SSSS with the IRS's group ranges |
|
||||
| US | EIN | Needs a word | a valid IRS prefix and seven digits |
|
||||
| US | Bank routing number (ABA) | Checked | nine digits, a valid Federal Reserve prefix, the 3-7-1 checksum |
|
||||
| US | Bank routing number (ABA) | Needs a word | nine digits, a valid Federal Reserve prefix, the 3-7-1 checksum |
|
||||
| US | Driver's license | Needs a word | each state's published format |
|
||||
| US | Medicare Beneficiary Identifier | Checked | CMS's 11-character pattern and excluded letters |
|
||||
| US | National Provider Identifier | Checked | ten digits, Luhn over the `80840` prefix |
|
||||
| US | National Provider Identifier | Needs a word | ten digits, Luhn over the `80840` prefix |
|
||||
| US | DEA registration number | Checked | two letters, seven digits, DEA's check digit |
|
||||
| UK | National Insurance number | Checked | two letters (HMRC's excluded prefixes), six digits, A–D |
|
||||
| UK | NHS number | Checked | ten digits, mod 11 |
|
||||
|
||||
Binary file not shown.
@@ -79,6 +79,21 @@ lockedAt = ["metadata"]
|
||||
lockedBy = ["identifier"]
|
||||
delegates = ["identifier"]
|
||||
|
||||
[object."inbuxa:MailRule"]
|
||||
file = "inbuxa_mail_rule.rs"
|
||||
default = "none"
|
||||
whose = ["administrator", "holder", "correspondent"]
|
||||
where = ["data-store"]
|
||||
scope = "server"
|
||||
retention = "unbounded"
|
||||
[object."inbuxa:MailRule".properties]
|
||||
name = ["content"]
|
||||
description = ["content"]
|
||||
conditions = ["contact", "content"]
|
||||
exceptions = ["contact", "content"]
|
||||
actions = ["contact", "content"]
|
||||
createdBy = ["identifier"]
|
||||
|
||||
[object."inbuxa:LegalHold"]
|
||||
file = "inbuxa_legal_hold.rs"
|
||||
default = "none"
|
||||
|
||||
Binary file not shown.
@@ -1 +1 @@
|
||||
k496pjVWlQ2p4bkZCh8agzaCKMLD4c3Z9WtoxpckYDU
|
||||
yF7PlBR3UBqxlabhW5zZ5WacQWsynG1wNYEiJVAl6w4
|
||||
@@ -0,0 +1,201 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! `inbuxa:MailRule` (dlp-and-mail-flow-rules spec, §2.2, §2.8): rules are
|
||||
//! stored, listed in the order they run, checked when written, kept apart
|
||||
//! by kind for permissions, and every change is audited.
|
||||
|
||||
use crate::utils::{
|
||||
account::Account,
|
||||
server::{TestServer, TestServerBuilder},
|
||||
};
|
||||
use registry::schema::{
|
||||
prelude::{ObjectType, Property},
|
||||
structs::{CustomRoles, Role, UserRoles},
|
||||
};
|
||||
use registry::types::map::Map;
|
||||
use serde_json::{Value, json};
|
||||
|
||||
const USING: &[&str] = &[
|
||||
"urn:ietf:params:jmap:core",
|
||||
"urn:inbuxa:jmap",
|
||||
"urn:inbuxa:jmap:registry",
|
||||
];
|
||||
|
||||
async fn call(account: &Account, method: &str, mut arguments: Value) -> (String, Value) {
|
||||
if arguments.get("accountId").is_none() {
|
||||
arguments["accountId"] = account.id_string().into();
|
||||
}
|
||||
let response = account.jmap_request(USING, json!([[method, arguments, "0"]])).await;
|
||||
let call = response
|
||||
.0
|
||||
.pointer("/methodResponses/0")
|
||||
.cloned()
|
||||
.unwrap_or_else(|| panic!("{method}: {}", response.0));
|
||||
(call[0].as_str().unwrap_or_default().to_string(), call[1].clone())
|
||||
}
|
||||
|
||||
fn dlp_rule() -> Value {
|
||||
json!({
|
||||
"name": "Cards leaving",
|
||||
"kind": "dlp",
|
||||
"direction": "outgoing",
|
||||
"priority": 10,
|
||||
"conditions": [
|
||||
{"type": "recipientOutside"},
|
||||
{"type": "detected", "detectors": [{"id": "payment-card", "atLeast": 5}]}
|
||||
],
|
||||
"actions": [{"type": "hold", "notice": "Held for review", "notifySender": true}]
|
||||
})
|
||||
}
|
||||
|
||||
fn transport_rule() -> Value {
|
||||
json!({
|
||||
"name": "Disclaimer",
|
||||
"kind": "transport",
|
||||
"direction": "outgoing",
|
||||
"priority": 1,
|
||||
"conditions": [{"type": "recipientOutside"}],
|
||||
"actions": [{"type": "addDisclaimer", "text": "Sent by Example Co.", "position": "bottom"}]
|
||||
})
|
||||
}
|
||||
|
||||
async fn names(account: &Account) -> Vec<String> {
|
||||
let (name, response) = call(account, "inbuxa:MailRule/get", json!({"ids": null})).await;
|
||||
assert_eq!(name, "inbuxa:MailRule/get", "{response}");
|
||||
response["list"]
|
||||
.as_array()
|
||||
.unwrap()
|
||||
.iter()
|
||||
.map(|r| r["name"].as_str().unwrap().to_string())
|
||||
.collect()
|
||||
}
|
||||
|
||||
pub async fn test(test: &mut TestServer) {
|
||||
println!("Running mail rule tests...");
|
||||
let admin = test.account("[email protected]");
|
||||
|
||||
// Created, then listed in the order they run
|
||||
let (_, response) = call(
|
||||
&admin,
|
||||
"inbuxa:MailRule/set",
|
||||
json!({"create": {"d": dlp_rule(), "t": transport_rule()}}),
|
||||
)
|
||||
.await;
|
||||
let dlp_id = response["created"]["d"]["id"]
|
||||
.as_str()
|
||||
.unwrap_or_else(|| panic!("DLP rule created: {response}"))
|
||||
.to_string();
|
||||
let transport_id = response["created"]["t"]["id"].as_str().unwrap().to_string();
|
||||
assert_eq!(names(&admin).await, vec!["Disclaimer", "Cards leaving"]);
|
||||
|
||||
let (_, response) = call(&admin, "inbuxa:MailRule/get", json!({"ids": [dlp_id]})).await;
|
||||
let rule = &response["list"][0];
|
||||
assert_eq!(rule["conditions"][1]["detectors"][0]["atLeast"], 5, "{rule}");
|
||||
assert_eq!(rule["actions"][0]["notifySender"], true);
|
||||
assert_eq!(rule["createdBy"], "[email protected]");
|
||||
assert!(rule["createdAt"].as_str().is_some_and(|d| d.ends_with('Z')), "{rule}");
|
||||
|
||||
// Checked when written
|
||||
let mut inbound = dlp_rule();
|
||||
inbound["direction"] = "incoming".into();
|
||||
let mut unknown = dlp_rule();
|
||||
unknown["conditions"][1]["detectors"][0]["id"] = "no-such-detector".into();
|
||||
let (_, response) = call(
|
||||
&admin,
|
||||
"inbuxa:MailRule/set",
|
||||
json!({"create": {"a": inbound, "b": unknown, "c": {"name": "x", "kind": "dlp"}}}),
|
||||
)
|
||||
.await;
|
||||
assert_eq!(response["notCreated"]["a"]["properties"][0], "direction", "{response}");
|
||||
assert!(
|
||||
response["notCreated"]["b"]["description"].as_str().unwrap().contains("no-such-detector"),
|
||||
"{response}"
|
||||
);
|
||||
assert!(response["notCreated"].get("c").is_some(), "{response}");
|
||||
|
||||
// Changed in place; what the server sets can't be sent
|
||||
let (_, response) = call(
|
||||
&admin,
|
||||
"inbuxa:MailRule/set",
|
||||
json!({"update": {
|
||||
transport_id.as_str(): {"name": "Footer", "priority": 50},
|
||||
dlp_id.as_str(): {"createdBy": "someone else"}
|
||||
}}),
|
||||
)
|
||||
.await;
|
||||
assert!(response["updated"].get(transport_id.as_str()).is_some(), "{response}");
|
||||
assert_eq!(response["notUpdated"][dlp_id.as_str()]["properties"][0], "createdBy", "{response}");
|
||||
assert_eq!(names(&admin).await, vec!["Cards leaving", "Footer"]);
|
||||
|
||||
// A compliance officer sees DLP rules, not mail flow rules, and changes
|
||||
// neither (settled answer 4)
|
||||
let mut officer_role = None;
|
||||
for id in admin
|
||||
.registry_query_ids(ObjectType::Role, Vec::<(&str, &str)>::new(), Vec::<&str>::new())
|
||||
.await
|
||||
{
|
||||
let role = admin.registry_get::<Role>(id).await;
|
||||
if role.description == "Compliance Officer" && role.member_tenant_id.is_none() {
|
||||
officer_role = Some(id);
|
||||
}
|
||||
}
|
||||
let officer = admin
|
||||
.create_user_account("[email protected]", "officer-secret-7731", "Officer", &[], vec![])
|
||||
.await;
|
||||
admin
|
||||
.registry_update_object(
|
||||
ObjectType::Account,
|
||||
officer.id(),
|
||||
json!({Property::Roles: UserRoles::Custom(CustomRoles {
|
||||
role_ids: Map::new(vec![officer_role.expect("the officer role")]),
|
||||
})}),
|
||||
)
|
||||
.await;
|
||||
assert_eq!(names(&officer).await, vec!["Cards leaving"]);
|
||||
let (name, response) =
|
||||
call(&officer, "inbuxa:MailRule/set", json!({"create": {"d": dlp_rule()}})).await;
|
||||
assert_eq!(name, "error", "the officer created a DLP rule: {response}");
|
||||
|
||||
// Deleted
|
||||
let (_, response) = call(
|
||||
&admin,
|
||||
"inbuxa:MailRule/set",
|
||||
json!({"destroy": [transport_id]}),
|
||||
)
|
||||
.await;
|
||||
assert_eq!(response["destroyed"][0], transport_id.as_str(), "{response}");
|
||||
assert_eq!(names(&admin).await, vec!["Cards leaving"]);
|
||||
|
||||
// Every change is in the audit log
|
||||
let (_, response) = call(
|
||||
&admin,
|
||||
"inbuxa:AuditEvent/query",
|
||||
json!({"filter": {"targetKind": "inbuxa:MailRule"}, "calculateTotal": true}),
|
||||
)
|
||||
.await;
|
||||
assert!(
|
||||
response["total"].as_u64().unwrap_or(0) >= 4,
|
||||
"creates, update and destroy audited: {response}"
|
||||
);
|
||||
}
|
||||
|
||||
#[ignore]
|
||||
#[tokio::test(flavor = "multi_thread")]
|
||||
pub async fn mail_rules_tests() {
|
||||
let mut test = TestServerBuilder::new("mail_rules_tests")
|
||||
.await
|
||||
.with_default_listeners()
|
||||
.await
|
||||
.build()
|
||||
.await;
|
||||
let admin = test.create_admin_account("[email protected]").await;
|
||||
test.insert_account(admin);
|
||||
self::test(&mut test).await;
|
||||
if test.is_reset() {
|
||||
test.temp_dir.delete();
|
||||
}
|
||||
}
|
||||
@@ -14,6 +14,7 @@ pub mod ai_explain;
|
||||
pub mod account_lock; // inbuxa: account lock with delegation
|
||||
pub mod legal_hold; // inbuxa: legal hold
|
||||
pub mod compliance; // inbuxa: the compliance roles
|
||||
pub mod mail_rules; // inbuxa: DLP and mail flow rules
|
||||
pub mod audit; // inbuxa: the audit log
|
||||
pub mod authorization;
|
||||
pub mod auto_reload; // inbuxa: registry writes apply at once
|
||||
|
||||
Reference in New Issue
Block a user