DLP: the detector framework, the region-free detectors, word lists and attachment text #99

Merged
jcoffey-dev merged 2 commits from feature/dlp-detectors into main 2026-09-29 00:13:21 +00:00
9 changed files with 1782 additions and 0 deletions
Generated
+4
View File
@@ -3947,9 +3947,12 @@ name = "inbuxa-features"
version = "0.16.22"
dependencies = [
"ahash",
"aho-corasick",
"base64 0.23.1",
"flate2",
"jmap_proto",
"quick-xml 0.41.0",
"regex",
"registry",
"serde",
"serde_json",
@@ -3961,6 +3964,7 @@ dependencies = [
"types",
"utils",
"xxhash-rust",
"zip",
]
[[package]]
+5
View File
@@ -21,6 +21,11 @@ base64 = "0.23"
sha2 = "0.11"
flate2 = "1.1"
tokio = { version = "1.53", features = ["sync", "rt"] }
# inbuxa: DLP detectors and attachment text (dlp-and-mail-flow-rules spec)
regex = "1.13.1"
aho-corasick = "1.1"
zip = "8.6"
quick-xml = "0.41"
[dev-dependencies]
tokio = { version = "1.53", features = ["macros", "rt"] }
+1
View File
@@ -23,6 +23,7 @@ pub mod audit;
pub mod branding;
pub mod hold;
pub mod lock;
pub mod mailflow;
pub mod masked_email;
pub mod privacy;
pub mod security;
@@ -0,0 +1,522 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Detectors that aren't tied to one country (§2.3, region "Any").
use super::{Detector, Findings, Region, Strength, checks, digits, stands_alone, word_near};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"payment-card",
"Payment card number",
Region::Any,
Strength::Checked,
payment_card,
),
Detector::new("iban", "IBAN", Region::Any, Strength::Checked, iban),
Detector::new(
"swift-bic",
"SWIFT/BIC code",
Region::Any,
Strength::NeedsWord,
swift_bic,
),
Detector::new(
"email-addresses",
"Email addresses",
Region::Any,
Strength::Checked,
email_addresses,
),
Detector::new(
"phone-numbers",
"Phone numbers",
Region::Any,
Strength::NeedsWord,
phone_numbers,
),
Detector::new(
"date-of-birth",
"Date of birth",
Region::Any,
Strength::NeedsWord,
date_of_birth,
),
Detector::new(
"passport",
"Passport number",
Region::Any,
Strength::NeedsWord,
passport,
),
Detector::new(
"private-key",
"Private key",
Region::Any,
Strength::Checked,
private_key,
),
Detector::new(
"credentials",
"Cloud and service credentials",
Region::Any,
Strength::Checked,
credentials,
),
];
fn re(pattern: &str) -> Regex {
Regex::new(pattern).expect("detector pattern")
}
// --- Payment cards --------------------------------------------------------
/// Issuer prefixes (ISO/IEC 7812 IINs) and the lengths each network issues.
fn card_network(number: &str) -> bool {
let len = number.len();
let prefix = |n: usize| number[..n].parse::<u32>().unwrap_or(0);
match number.as_bytes()[0] {
// Visa
b'4' => matches!(len, 13 | 16 | 19),
b'5' => {
// Mastercard 51–55; Maestro 50, 56–58
(51..=55).contains(&prefix(2)) && len == 16
|| matches!(prefix(2), 50 | 56..=58) && (12..=19).contains(&len)
}
// Mastercard 2221–2720
b'2' => (2221..=2720).contains(&prefix(4)) && len == 16,
b'3' => {
// American Express 34, 37; JCB 3528–3589; Diners 300–305, 36, 38, 39
matches!(prefix(2), 34 | 37) && len == 15
|| (3528..=3589).contains(&prefix(4)) && (16..=19).contains(&len)
|| ((300..=305).contains(&prefix(3)) || matches!(prefix(2), 36 | 38 | 39))
&& (14..=19).contains(&len)
}
// Discover 6011, 644–649, 65; UnionPay 62; Maestro 6x
b'6' => (12..=19).contains(&len),
_ => false,
}
}
fn is_card(number: &str) -> bool {
(12..=19).contains(&number.len()) && card_network(number) && checks::luhn(number)
}
static CARD: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d(?:[ -]?\d){11,18}\b"));
fn payment_card(text: &str, findings: &mut Findings) {
for m in CARD.find_iter(text) {
if !stands_alone(text, m.start(), m.end()) {
continue;
}
let whole = digits(m.as_str());
if is_card(&whole) {
findings.insert(whole);
continue;
}
// Two numbers side by side ("4242 4242 4242 4242 2031"): try each
// run of whole groups
let groups: Vec<String> = m.as_str().split([' ', '-']).map(digits).collect();
'runs: for from in 0..groups.len() {
let mut number = String::new();
for group in &groups[from..] {
number.push_str(group);
if is_card(&number) {
findings.insert(number);
break 'runs;
}
}
}
}
}
// --- IBAN -----------------------------------------------------------------
static IBAN: LazyLock<Regex> =
LazyLock::new(|| re(r"\b[A-Za-z]{2}\d{2}(?:[ ]?[A-Za-z0-9]){11,30}"));
fn iban(text: &str, findings: &mut Findings) {
// The pattern can run on into the next words, even the next IBAN: after
// each hit, look again from where that IBAN ended
let mut from = 0;
while let Some(m) = IBAN.find_at(text, from) {
from = m.start() + 1;
let compact = m.as_str().replace(' ', "").to_ascii_uppercase();
let Some(len) = checks::iban_length(&compact[..2]) else {
continue;
};
if compact.len() < len {
continue;
}
// Where the country's length ends in the text, spaces counted
let mut seen = 0;
let Some(end) = m
.as_str()
.char_indices()
.find(|(_, c)| {
if *c != ' ' {
seen += 1;
}
seen == len
})
.map(|(i, c)| m.start() + i + c.len_utf8())
else {
continue;
};
let candidate = &compact[..len];
if stands_alone(text, m.start(), end) && checks::iban(candidate) {
findings.insert(candidate);
from = end;
}
}
}
// --- SWIFT/BIC ------------------------------------------------------------
static BIC: LazyLock<Regex> =
LazyLock::new(|| re(r"\b[A-Z]{4}[A-Z]{2}[A-Z0-9]{2}(?:[A-Z0-9]{3})?\b"));
const BIC_WORDS: &[&str] = &[
"swift",
"bic",
"swift/bic",
"bank",
"banque",
"bankverbindung",
];
fn swift_bic(text: &str, findings: &mut Findings) {
for m in BIC.find_iter(text) {
let code = m.as_str();
if checks::is_country(&code[4..6]) && word_near(text, m.start(), m.end(), BIC_WORDS) {
findings.insert(code);
}
}
}
// --- Contact lists --------------------------------------------------------
static EMAIL: LazyLock<Regex> =
LazyLock::new(|| re(r"(?i)\b[a-z0-9._%+-]+@[a-z0-9-]+(?:\.[a-z0-9-]+)*\.[a-z]{2,}\b"));
fn email_addresses(text: &str, findings: &mut Findings) {
for m in EMAIL.find_iter(text) {
findings.insert(m.as_str().to_lowercase());
}
}
/// International form: found alone. National form: only with a word.
static PHONE_INTL: LazyLock<Regex> = LazyLock::new(|| re(r"\+\d{1,3}(?:[ .-]?\(?\d{1,4}\)?){2,5}"));
static PHONE_NATIONAL: LazyLock<Regex> =
LazyLock::new(|| re(r"\(?\d{2,4}\)?[ .-]\d{3,4}[ .-]\d{3,4}"));
const PHONE_WORDS: &[&str] = &[
"phone",
"tel",
"telephone",
"mobile",
"cell",
"fax",
"telefon",
"téléphone",
"teléfono",
"telefono",
"handy",
"portable",
"móvil",
"cellulare",
"mobiel",
];
fn phone_numbers(text: &str, findings: &mut Findings) {
let mut international = Vec::new();
for m in PHONE_INTL.find_iter(text) {
let number = digits(m.as_str());
if (8..=15).contains(&number.len()) && stands_alone(text, m.start() + 1, m.end()) {
findings.insert(number);
international.push(m.range());
}
}
for m in PHONE_NATIONAL.find_iter(text) {
let number = digits(m.as_str());
// Not the tail of an international number already counted
if international.iter().any(|r| r.contains(&m.start())) {
continue;
}
if (9..=11).contains(&number.len())
&& stands_alone(text, m.start(), m.end())
&& !text[..m.start()].ends_with('+')
&& word_near(text, m.start(), m.end(), PHONE_WORDS)
{
findings.insert(number);
}
}
}
// --- Date of birth --------------------------------------------------------
static DATE_ISO: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{4})-(\d{2})-(\d{2})\b"));
static DATE_NUMERIC: LazyLock<Regex> =
LazyLock::new(|| re(r"\b(\d{1,2})[./-](\d{1,2})[./-](\d{4})\b"));
static DATE_WORDS: LazyLock<Regex> = LazyLock::new(|| {
re(
r"(?i)\b(?:(\d{1,2})\s+(jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)[a-z]*\.?,?\s+(\d{4})|(jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)[a-z]*\.?\s+(\d{1,2}),?\s+(\d{4}))\b",
)
});
const BIRTH_WORDS: &[&str] = &[
"born",
"birth",
"dob",
"d.o.b",
"birthday",
"birthdate",
"geburtsdatum",
"geboren",
"naissance",
"né le",
"née le",
"nacimiento",
"nacido",
"nacida",
"nascita",
"nato il",
"nata il",
"geboortedatum",
"födelsedatum",
"fødselsdato",
"syntymäaika",
"urodzenia",
"nascimento",
];
fn valid_date(year: u32, month: u32, day: u32) -> bool {
let days = match month {
1 | 3 | 5 | 7 | 8 | 10 | 12 => 31,
4 | 6 | 9 | 11 => 30,
2 if year.is_multiple_of(4) && (!year.is_multiple_of(100) || year.is_multiple_of(400)) => {
29
}
2 => 28,
_ => return false,
};
(1900..=2100).contains(&year) && (1..=days).contains(&day)
}
fn month_number(name: &str) -> u32 {
const MONTHS: [&str; 12] = [
"jan", "feb", "mar", "apr", "may", "jun", "jul", "aug", "sep", "oct", "nov", "dec",
];
let name = name.to_lowercase();
MONTHS
.iter()
.position(|m| *m == name)
.map_or(0, |i| i as u32 + 1)
}
fn date_of_birth(text: &str, findings: &mut Findings) {
let mut add = |start: usize, end: usize, key: String| {
if word_near(text, start, end, BIRTH_WORDS) {
findings.insert(key);
}
};
let num = |s: &str| s.parse::<u32>().unwrap_or(0);
for c in DATE_ISO.captures_iter(text) {
let (y, m, d) = (num(&c[1]), num(&c[2]), num(&c[3]));
let whole = c.get(0).unwrap();
if valid_date(y, m, d) {
add(whole.start(), whole.end(), format!("{y:04}{m:02}{d:02}"));
}
}
for c in DATE_NUMERIC.captures_iter(text) {
let (a, b, y) = (num(&c[1]), num(&c[2]), num(&c[3]));
let whole = c.get(0).unwrap();
// Day first or month first: either reading that is a real date
if valid_date(y, b, a) || valid_date(y, a, b) {
add(whole.start(), whole.end(), whole.as_str().to_string());
}
}
for c in DATE_WORDS.captures_iter(text) {
let whole = c.get(0).unwrap();
let (d, m, y) = match (c.get(1), c.get(4)) {
(Some(d), _) => (num(d.as_str()), month_number(&c[2]), num(&c[3])),
(_, Some(m)) => (num(&c[5]), month_number(m.as_str()), num(&c[6])),
_ => continue,
};
if valid_date(y, m, d) {
add(whole.start(), whole.end(), format!("{y:04}{m:02}{d:02}"));
}
}
}
// --- Passport -------------------------------------------------------------
static PASSPORT: LazyLock<Regex> = LazyLock::new(|| re(r"\b[A-Z0-9]{6,9}\b"));
const PASSPORT_WORDS: &[&str] = &[
"passport",
"passeport",
"reisepass",
"pasaporte",
"passaporto",
"paspoort",
"passnummer",
"pass-nr",
"passport no",
"pasaporte n.º",
"passaporte",
];
fn passport(text: &str, findings: &mut Findings) {
for m in PASSPORT.find_iter(text) {
let value = m.as_str();
if value.bytes().filter(u8::is_ascii_digit).count() >= 5
&& word_near(text, m.start(), m.end(), PASSPORT_WORDS)
{
findings.insert(value);
}
}
}
// --- Keys and credentials -------------------------------------------------
static PRIVATE_KEY: LazyLock<Regex> = LazyLock::new(|| {
re(
r"-----BEGIN (?:(?:RSA|EC|DSA|OPENSSH|ENCRYPTED|PGP) )?PRIVATE KEY(?: BLOCK)?-----\s*([A-Za-z0-9+/=:\s-]{0,64})",
)
});
fn private_key(text: &str, findings: &mut Findings) {
for c in PRIVATE_KEY.captures_iter(text) {
// Each key once, by the start of its body
let body: String = c[1].chars().filter(|c| !c.is_whitespace()).collect();
let whole = c.get(0).unwrap();
findings.insert(if body.is_empty() {
format!("@{}", whole.start())
} else {
body
});
}
}
/// Published token formats: AWS access key IDs, GitHub tokens, Slack
/// tokens, Stripe live secret and restricted keys, Google API keys.
static CREDENTIAL: LazyLock<Regex> = LazyLock::new(|| {
re(concat!(
r"\b(?:",
r"(?:AKIA|ASIA|ABIA|ACCA)[A-Z0-9]{16}",
r"|gh[pousr]_[A-Za-z0-9]{36}",
r"|github_pat_[A-Za-z0-9_]{82}",
r"|xox[abposr]-[A-Za-z0-9-]{10,72}",
r"|(?:sk|rk)_live_[A-Za-z0-9]{24,99}",
r"|AIza[0-9A-Za-z_-]{35}",
r")\b"
))
});
fn credentials(text: &str, findings: &mut Findings) {
for m in CREDENTIAL.find_iter(text) {
findings.insert(m.as_str());
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn payment_cards() {
// Networks' and processors' published test numbers
let text = "Visa 4242 4242 4242 4242, MC 5555-5555-5555-4444, Amex 378282246310005, \
Discover 6011111111111117, JCB 3566002020360505, Diners 30569309025904, \
UnionPay 6200000000000005, Mastercard 2-series 2223003122003222";
assert_eq!(count("payment-card", text), 8);
// Luhn fails, wrong network length, inside a longer number
assert_eq!(count("payment-card", "4242424242424241"), 0);
assert_eq!(count("payment-card", "378282246310005 0"), 1);
assert_eq!(count("payment-card", "order 94242424242424242 shipped"), 0);
// The same number twice counts once
assert_eq!(
count("payment-card", "4242424242424242 and 4242-4242-4242-4242"),
1
);
// A card followed by a year
assert_eq!(count("payment-card", "card 4242 4242 4242 4242 2031"), 1);
}
#[test]
fn ibans() {
let text =
"Pay GB29 NWBK 6016 1331 9268 19 or de89370400440532013000 (NL91ABNA0417164300).";
assert_eq!(count("iban", text), 3);
assert_eq!(count("iban", "GB29 NWBK 6016 1331 9268 18"), 0);
// Runs into the next word: still found at the country's length
assert_eq!(count("iban", "IBAN NL91ABNA0417164300 BIC ABNANL2A"), 1);
}
#[test]
fn swift_codes_need_a_word() {
assert_eq!(count("swift-bic", "SWIFT: DEUTDEFF500"), 1);
assert_eq!(count("swift-bic", "BIC NWBKGB2L"), 1);
assert_eq!(count("swift-bic", "HAPPYDAYS DEUTDEFF"), 0);
// Not a country in positions 5–6
assert_eq!(count("swift-bic", "BIC DEUTZZFF"), 0);
}
#[test]
fn email_and_phone_lists() {
let list = "[email protected], [email protected], [email protected], [email protected]";
assert_eq!(count("email-addresses", list), 3);
assert_eq!(
count("phone-numbers", "+44 20 7946 0958, +1 (415) 555-2671"),
2
);
assert_eq!(count("phone-numbers", "call 020 7946 0958"), 0);
assert_eq!(count("phone-numbers", "Tel: 020 7946 0958"), 1);
assert_eq!(count("phone-numbers", "invoice 020 7946 0958"), 0);
// One number, not also its national tail
assert_eq!(count("phone-numbers", "Tel: +44 20 7946 0958"), 1);
}
#[test]
fn dates_of_birth() {
assert_eq!(count("date-of-birth", "DOB: 1984-02-29"), 1);
assert_eq!(count("date-of-birth", "Geburtsdatum 31.12.1970"), 1);
assert_eq!(count("date-of-birth", "born on March 3, 1962"), 1);
assert_eq!(count("date-of-birth", "date of birth 3 Mar 1962"), 1);
// Not a real date, no word, a meeting
assert_eq!(count("date-of-birth", "DOB: 1985-02-29"), 0);
assert_eq!(count("date-of-birth", "invoice 1984-02-29"), 0);
assert_eq!(count("date-of-birth", "Meeting on 12/05/2026"), 0);
}
#[test]
fn passports_need_a_word() {
assert_eq!(count("passport", "Passport number: 533380006"), 1);
assert_eq!(count("passport", "Reisepass C01X00T47"), 1);
assert_eq!(count("passport", "Order 533380006 shipped"), 0);
// Mostly letters: a word, not a number
assert_eq!(count("passport", "passport PASSWORD"), 0);
}
#[test]
fn keys_and_credentials() {
let key = "-----BEGIN OPENSSH PRIVATE KEY-----\nb3BlbnNzaC1rZXktdjEAAAAABG5vbmUAAAAEbm9uZQ\n-----END OPENSSH PRIVATE KEY-----";
assert_eq!(count("private-key", key), 1);
assert_eq!(count("private-key", "-----BEGIN PUBLIC KEY-----\nMFkw"), 0);
// Documentation examples of each format
let tokens = "AKIAIOSFODNN7EXAMPLE ghp_0123456789abcdefghijklmnopqrstuvwxyz \
AIzaSyA-0123456789abcdefghijklmnopqrstu";
assert_eq!(count("credentials", tokens), 3);
assert_eq!(count("credentials", "AKIA123 ghp_short"), 0);
}
}
@@ -0,0 +1,227 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Check-digit algorithms, each from its public definition.
/// The Luhn check (ISO/IEC 7812-1, Annex B) over a string of ASCII digits.
pub fn luhn(digits: &str) -> bool {
if digits.len() < 2 || !digits.bytes().all(|b| b.is_ascii_digit()) {
return false;
}
let sum: u32 = digits
.bytes()
.rev()
.enumerate()
.map(|(i, b)| {
let d = u32::from(b - b'0');
if i % 2 == 1 {
let d = d * 2;
if d > 9 { d - 9 } else { d }
} else {
d
}
})
.sum();
sum % 10 == 0
}
/// ISO 13616 IBAN lengths, by country, from the IBAN registry.
const IBAN_LENGTHS: &[(&str, usize)] = &[
("AD", 24),
("AE", 23),
("AL", 28),
("AT", 20),
("AZ", 28),
("BA", 20),
("BE", 16),
("BG", 22),
("BH", 22),
("BI", 27),
("BR", 29),
("BY", 28),
("CH", 21),
("CR", 22),
("CY", 28),
("CZ", 24),
("DE", 22),
("DJ", 27),
("DK", 18),
("DO", 28),
("EE", 20),
("EG", 29),
("ES", 24),
("FI", 18),
("FK", 18),
("FO", 18),
("FR", 27),
("GB", 22),
("GE", 22),
("GI", 23),
("GL", 18),
("GR", 27),
("GT", 28),
("HN", 28),
("HR", 21),
("HU", 28),
("IE", 22),
("IL", 23),
("IQ", 23),
("IS", 26),
("IT", 27),
("JO", 30),
("KW", 30),
("KZ", 20),
("LB", 28),
("LC", 32),
("LI", 21),
("LT", 20),
("LU", 20),
("LV", 21),
("LY", 25),
("MC", 27),
("MD", 24),
("ME", 22),
("MK", 19),
("MN", 20),
("MR", 27),
("MT", 31),
("MU", 30),
("NI", 28),
("NL", 18),
("NO", 15),
("OM", 23),
("PK", 24),
("PL", 28),
("PS", 29),
("PT", 25),
("QA", 29),
("RO", 24),
("RS", 22),
("RU", 33),
("SA", 24),
("SC", 31),
("SD", 18),
("SE", 24),
("SI", 19),
("SK", 24),
("SM", 27),
("SO", 23),
("ST", 25),
("SV", 28),
("TL", 23),
("TN", 24),
("TR", 26),
("UA", 29),
("VA", 22),
("VG", 24),
("XK", 20),
("YE", 30),
];
/// The IBAN length for a country code, if the country uses IBANs.
pub fn iban_length(country: &str) -> Option<usize> {
IBAN_LENGTHS
.iter()
.find(|(code, _)| *code == country)
.map(|(_, len)| *len)
}
/// ISO 13616 / ISO 7064 MOD 97-10 over an IBAN with no spaces, upper case:
/// move the first four characters to the end, turn letters into 10–35, and
/// the number mod 97 must be 1. Also checks the country's length.
pub fn iban(iban: &str) -> bool {
if iban.len() < 5
|| !iban
.bytes()
.all(|b| b.is_ascii_uppercase() || b.is_ascii_digit())
{
return false;
}
if iban_length(&iban[..2]) != Some(iban.len())
|| !iban[2..4].bytes().all(|b| b.is_ascii_digit())
{
return false;
}
let mut remainder: u32 = 0;
for b in iban[4..].bytes().chain(iban[..4].bytes()) {
let value = if b.is_ascii_digit() {
u32::from(b - b'0')
} else {
u32::from(b - b'A') + 10
};
remainder = if value >= 10 {
(remainder * 100 + value) % 97
} else {
(remainder * 10 + value) % 97
};
}
remainder == 1
}
/// ISO 3166-1 alpha-2 country codes, for SWIFT/BIC positions 5–6.
const COUNTRIES: &str = "AD AE AF AG AI AL AM AO AQ AR AS AT AU AW AX AZ BA BB BD BE BF BG BH BI BJ \
BL BM BN BO BQ BR BS BT BV BW BY BZ CA CC CD CF CG CH CI CK CL CM CN CO CR CU CV CW CX CY CZ DE DJ \
DK DM DO DZ EC EE EG EH ER ES ET FI FJ FK FM FO FR GA GB GD GE GF GG GH GI GL GM GN GP GQ GR GS GT \
GU GW GY HK HM HN HR HT HU ID IE IL IM IN IO IQ IR IS IT JE JM JO JP KE KG KH KI KM KN KP KR KW KY \
KZ LA LB LC LI LK LR LS LT LU LV LY MA MC MD ME MF MG MH MK ML MM MN MO MP MQ MR MS MT MU MV MW MX \
MY MZ NA NC NE NF NG NI NL NO NP NR NU NZ OM PA PE PF PG PH PK PL PM PN PR PS PT PW PY QA RE RO RS \
RU RW SA SB SC SD SE SG SH SI SJ SK SL SM SN SO SR SS ST SV SX SY SZ TC TD TF TG TH TJ TK TL TM TN \
TO TR TT TV TW TZ UA UG UM US UY UZ VA VC VE VG VI VN VU WF WS XK YE YT ZA ZM ZW";
pub fn is_country(code: &str) -> bool {
code.len() == 2 && COUNTRIES.split(' ').any(|c| c == code)
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn luhn_known_numbers() {
// Published test card numbers
for good in [
"4242424242424242",
"5555555555554444",
"378282246310005",
"79927398713",
] {
assert!(luhn(good), "{good}");
}
for bad in ["4242424242424241", "79927398710", "1", "12a4"] {
assert!(!luhn(bad), "{bad}");
}
}
#[test]
fn iban_registry_examples() {
// The IBAN registry's own examples
for good in [
"GB29NWBK60161331926819",
"DE89370400440532013000",
"FR1420041010050500013M02606",
"NL91ABNA0417164300",
"BE68539007547034",
"NO9386011117947",
"CH9300762011623852957",
] {
assert!(iban(good), "{good}");
}
for bad in [
"GB29NWBK60161331926818", // check fails
"GB29NWBK6016133192681", // too short for GB
"ZZ29NWBK60161331926819", // no such country
"DE8937040044053201300A", // letters where DE has none still fail mod 97
] {
assert!(!iban(bad), "{bad}");
}
}
#[test]
fn countries() {
assert!(is_country("DE") && is_country("US") && is_country("XK"));
assert!(!is_country("ZZ") && !is_country("D"));
}
}
@@ -0,0 +1,199 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Detectors (dlp-and-mail-flow-rules spec, §2.3): each finds one kind of
//! identifier in text and reports the distinct ones it found.
//!
//! A detector is one of two strengths:
//!
//! - **Checked**: the identifier carries a published check digit or
//! checksum, so a random number rarely passes; found on its own.
//! - **Needs a word**: the format alone is too common, so a candidate counts
//! only with a corroborating word within [`WINDOW`] characters either
//! side.
//!
//! Findings are distinct normalized values (digits only, upper case), so the
//! same card number pasted twice counts once. They stay in memory: callers
//! read only [`Findings::len`].
pub mod any;
pub mod checks;
use ahash::AHashSet;
/// How far, in characters, a corroborating word may be from a candidate.
pub const WINDOW: usize = 50;
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Strength {
Checked,
NeedsWord,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Region {
Any,
Us,
Uk,
Canada,
Australia,
Eu,
Europe,
Asia,
Americas,
Africa,
}
/// The distinct values one detector found.
#[derive(Debug, Default)]
pub struct Findings(AHashSet<String>);
impl Findings {
pub fn insert(&mut self, value: impl Into<String>) {
self.0.insert(value.into());
}
pub fn len(&self) -> usize {
self.0.len()
}
pub fn is_empty(&self) -> bool {
self.0.is_empty()
}
}
pub struct Detector {
/// Stable id, stored in rules: `payment-card`, `iban`, `us-ssn`.
pub id: &'static str,
pub name: &'static str,
pub region: Region,
pub strength: Strength,
find: fn(&str, &mut Findings),
}
impl Detector {
pub const fn new(
id: &'static str,
name: &'static str,
region: Region,
strength: Strength,
find: fn(&str, &mut Findings),
) -> Self {
Self {
id,
name,
region,
strength,
find,
}
}
/// Adds what this detector finds in `text` to `findings`. Call once per
/// piece of text (subject, each part, each attachment) with the same
/// `findings`, then read its length.
pub fn find(&self, text: &str, findings: &mut Findings) {
(self.find)(text, findings)
}
/// The distinct values found in one text.
pub fn count(&self, text: &str) -> usize {
let mut findings = Findings::default();
self.find(text, &mut findings);
findings.len()
}
}
/// Every detector, in the order the console lists them.
pub fn all() -> impl Iterator<Item = &'static Detector> {
any::DETECTORS.iter()
}
pub fn by_id(id: &str) -> Option<&'static Detector> {
all().find(|detector| detector.id == id)
}
/// Whether one of `words` appears, as a whole word and ignoring case, within
/// [`WINDOW`] characters before `start` or after `end` (byte offsets of the
/// candidate in `text`). The window is widened by the longest word, so a
/// word that reaches into it still counts whole.
pub fn word_near(text: &str, start: usize, end: usize, words: &[&str]) -> bool {
let reach = WINDOW + words.iter().map(|w| w.chars().count()).max().unwrap_or(0);
let before = text[..start]
.char_indices()
.rev()
.nth(reach - 1)
.map_or(0, |(i, _)| i);
let after = text[end..]
.char_indices()
.nth(reach)
.map_or(text.len(), |(i, _)| end + i);
let window = text[before..after].to_lowercase();
words.iter().any(|word| contains_word(&window, word))
}
/// Whether `word` (lower case) appears in `haystack` (lower case) with no
/// letter or digit on either side.
pub fn contains_word(haystack: &str, word: &str) -> bool {
haystack.match_indices(word).any(|(i, _)| {
let before_ok = haystack[..i]
.chars()
.next_back()
.is_none_or(|c| !c.is_alphanumeric());
let after_ok = haystack[i + word.len()..]
.chars()
.next()
.is_none_or(|c| !c.is_alphanumeric());
before_ok && after_ok
})
}
/// Whether the match at `start..end` stands alone: no digit or letter
/// directly before or after it, so `123-45-6789` isn't found inside a
/// longer run of digits.
pub fn stands_alone(text: &str, start: usize, end: usize) -> bool {
let before = text[..start].chars().next_back();
let after = text[end..].chars().next();
before.is_none_or(|c| !c.is_alphanumeric()) && after.is_none_or(|c| !c.is_alphanumeric())
}
/// The ASCII digits of `s`.
pub fn digits(s: &str) -> String {
s.chars().filter(char::is_ascii_digit).collect()
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn words_are_whole_and_near() {
let text = "Your passport number is X1234567, thanks";
let start = text.find("X123").unwrap();
assert!(word_near(text, start, start + 8, &["passport"]));
assert!(!word_near(text, start, start + 8, &["pass"]));
let far = format!("passport{}X1234567", " ".repeat(60));
let start = far.find("X123").unwrap();
assert!(!word_near(&far, start, start + 8, &["passport"]));
}
#[test]
fn near_counts_characters_not_bytes() {
// 45 two-byte characters between the word and the candidate: within
// 50 characters, though over 50 bytes
let text = format!("passport {} X1234567", "é".repeat(45));
let start = text.find("X123").unwrap();
assert!(word_near(&text, start, start + 8, &["passport"]));
}
#[test]
fn ids_are_unique() {
let mut seen = AHashSet::new();
for detector in all() {
assert!(seen.insert(detector.id), "duplicate id {}", detector.id);
assert!(by_id(detector.id).is_some());
}
}
}
+692
View File
@@ -0,0 +1,692 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! The text of an attachment, for the detectors (§2.3), or why there isn't
//! one.
//!
//! Read: text files (plain, CSV, JSON, XML, HTML), Office Open XML (DOCX,
//! XLSX, PPTX) and OpenDocument (ODT, ODS, ODP) documents, and ZIP archives
//! one level deep. **Can't be inspected**: encrypted or password-protected
//! files, PDF (settled answer 2), the older binary Office formats, archives
//! inside archives, and anything past the limits. Everything else (images,
//! audio, programs) has no text to read and is neither.
//!
//! Office files are ZIP archives of XML, read here with the `zip` and
//! `quick-xml` crates the server already uses: no outside converter runs.
use quick_xml::{Reader, XmlVersion, events::Event};
use std::io::{Cursor, Read};
/// How much may be unpacked from one attachment, and from how many entries.
#[derive(Debug, Clone, Copy)]
pub struct Limits {
pub max_unpacked: u64,
pub max_entries: usize,
}
impl Default for Limits {
fn default() -> Self {
Self {
max_unpacked: 50 * 1024 * 1024,
max_entries: 10_000,
}
}
}
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum Extracted {
/// The text to check.
Text(String),
/// A kind of file with no text in it: nothing to check, nothing missed.
NoText,
/// A file that may hold text the detectors couldn't read.
NotInspectable(Why),
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Why {
Encrypted,
Pdf,
LegacyOffice,
NestedArchive,
TooLarge,
Damaged,
}
impl Why {
pub fn as_str(&self) -> &'static str {
match self {
Why::Encrypted => "encrypted",
Why::Pdf => "pdf",
Why::LegacyOffice => "legacy-office",
Why::NestedArchive => "nested-archive",
Why::TooLarge => "too-large",
Why::Damaged => "damaged",
}
}
}
const OLE_MAGIC: &[u8] = &[0xD0, 0xCF, 0x11, 0xE0, 0xA1, 0xB1, 0x1A, 0xE1];
const ZIP_MAGIC: &[u8] = b"PK\x03\x04";
/// What an attachment says, from its declared type, its file name and, above
/// all, its first bytes.
pub fn extract(
content_type: &str,
file_name: Option<&str>,
data: &[u8],
limits: &Limits,
) -> Extracted {
extract_at(content_type, file_name, data, limits, 0)
}
fn extract_at(
content_type: &str,
file_name: Option<&str>,
data: &[u8],
limits: &Limits,
depth: u8,
) -> Extracted {
let content_type = content_type.to_ascii_lowercase();
let extension = file_name
.and_then(|name| name.rsplit_once('.'))
.map(|(_, ext)| ext.to_ascii_lowercase())
.unwrap_or_default();
if data.len() as u64 > limits.max_unpacked {
return Extracted::NotInspectable(Why::TooLarge);
}
if data.starts_with(b"%PDF-") || content_type == "application/pdf" || extension == "pdf" {
return Extracted::NotInspectable(Why::Pdf);
}
if data.starts_with(OLE_MAGIC) {
// An encrypted OOXML file is an OLE container holding the encrypted
// package; any other OLE file is a legacy .doc, .xls or .ppt
return Extracted::NotInspectable(if has_utf16(data, "EncryptedPackage") {
Why::Encrypted
} else {
Why::LegacyOffice
});
}
if data.starts_with(ZIP_MAGIC) {
if depth > 0 {
return Extracted::NotInspectable(Why::NestedArchive);
}
return zip(data, limits);
}
if is_text(&content_type, &extension) {
let text = decode_text(data);
return Extracted::Text(
if content_type == "text/html" || matches!(extension.as_str(), "html" | "htm") {
strip_html(&text)
} else {
text
},
);
}
Extracted::NoText
}
fn is_text(content_type: &str, extension: &str) -> bool {
content_type.starts_with("text/")
|| matches!(
content_type,
"application/json"
| "application/xml"
| "application/csv"
| "application/x-csv"
| "message/rfc822"
)
|| matches!(
extension,
"txt"
| "csv"
| "tsv"
| "json"
| "xml"
| "md"
| "log"
| "html"
| "htm"
| "eml"
| "ics"
| "vcf"
)
}
/// UTF-16 with a byte order mark, else UTF-8 (lossy).
fn decode_text(data: &[u8]) -> String {
let utf16 = |bytes: &[u8], big: bool| {
let units: Vec<u16> = bytes
.chunks_exact(2)
.map(|c| {
if big {
u16::from_be_bytes([c[0], c[1]])
} else {
u16::from_le_bytes([c[0], c[1]])
}
})
.collect();
String::from_utf16_lossy(&units)
};
match data {
[0xFF, 0xFE, rest @ ..] => utf16(rest, false),
[0xFE, 0xFF, rest @ ..] => utf16(rest, true),
[0xEF, 0xBB, 0xBF, rest @ ..] => String::from_utf8_lossy(rest).into_owned(),
_ => String::from_utf8_lossy(data).into_owned(),
}
}
fn has_utf16(data: &[u8], needle: &str) -> bool {
let needle: Vec<u8> = needle.encode_utf16().flat_map(u16::to_le_bytes).collect();
data.windows(needle.len()).any(|w| w == needle.as_slice())
}
/// Tags out, the common entities decoded, block ends as new lines.
fn strip_html(html: &str) -> String {
let mut out = String::with_capacity(html.len());
let mut in_tag = false;
let mut skip_until: Option<&str> = None;
let lower = html.to_ascii_lowercase();
let mut i = 0;
let bytes = html.as_bytes();
while i < bytes.len() {
if let Some(end) = skip_until {
match lower[i..].find(end) {
Some(at) => {
i += at + end.len();
skip_until = None;
}
None => break,
}
continue;
}
let c = bytes[i];
if in_tag {
if c == b'>' {
in_tag = false;
}
i += 1;
continue;
}
if c == b'<' {
if lower[i..].starts_with("<script") {
skip_until = Some("</script>");
} else if lower[i..].starts_with("<style") {
skip_until = Some("</style>");
} else {
if [
"<br", "<p", "</p", "<div", "</div", "<tr", "<li", "<td", "<th",
]
.iter()
.any(|t| lower[i..].starts_with(t))
{
out.push(
if lower[i..].starts_with("<td") || lower[i..].starts_with("<th") {
'\t'
} else {
'\n'
},
);
}
in_tag = true;
}
i += 1;
continue;
}
// Copy up to the next tag
let next = html[i..].find('<').map_or(html.len(), |at| i + at);
out.push_str(&html[i..next]);
i = next;
}
for (entity, text) in [
("&nbsp;", " "),
("&lt;", "<"),
("&gt;", ">"),
("&quot;", "\""),
("&#39;", "'"),
("&amp;", "&"),
] {
out = out.replace(entity, text);
}
out
}
/// A ZIP file: an Office document, an OpenDocument, or an archive.
fn zip(data: &[u8], limits: &Limits) -> Extracted {
let Ok(mut archive) = zip::ZipArchive::new(Cursor::new(data)) else {
return Extracted::NotInspectable(Why::Damaged);
};
if archive.len() > limits.max_entries {
return Extracted::NotInspectable(Why::TooLarge);
}
let mut names = Vec::with_capacity(archive.len());
let mut declared: u64 = 0;
for i in 0..archive.len() {
let Ok(entry) = archive.by_index_raw(i) else {
return Extracted::NotInspectable(Why::Damaged);
};
if entry.encrypted() {
return Extracted::NotInspectable(Why::Encrypted);
}
declared = declared.saturating_add(entry.size());
names.push(entry.name().to_string());
}
if declared > limits.max_unpacked {
return Extracted::NotInspectable(Why::TooLarge);
}
let mut budget = limits.max_unpacked;
let mut read =
|archive: &mut zip::ZipArchive<Cursor<&[u8]>>, name: &str| -> Result<Vec<u8>, Why> {
let entry = archive.by_name(name).map_err(|_| Why::Damaged)?;
let mut bytes = Vec::new();
// Declared sizes can lie: stop at the budget whatever they say
entry
.take(budget + 1)
.read_to_end(&mut bytes)
.map_err(|_| Why::Damaged)?;
if bytes.len() as u64 > budget {
return Err(Why::TooLarge);
}
budget -= bytes.len() as u64;
Ok(bytes)
};
let has = |name: &str| names.iter().any(|n| n == name);
let mut text = String::new();
let result: Result<(), Why> = (|| {
if has("[Content_Types].xml") {
// Office Open XML: the parts that hold what a person wrote
let mut shared = Vec::new();
if has("xl/sharedStrings.xml") {
shared = xml_strings(&read(&mut archive, "xl/sharedStrings.xml")?, "si");
}
for name in names.iter().filter(|n| ooxml_text_part(n)) {
let xml = read(&mut archive, name)?;
if name.starts_with("xl/worksheets/") {
xlsx_sheet(&xml, &mut text);
} else {
xml_text(&xml, &mut text);
}
text.push('\n');
}
text.extend(shared.iter().map(|s| format!("{s}\n")));
} else if names.first().is_some_and(|n| n == "mimetype")
&& read(&mut archive, "mimetype")?.starts_with(b"application/vnd.oasis.opendocument")
{
// OpenDocument: an encrypted one says so in its manifest
if has("META-INF/manifest.xml")
&& contains(
&read(&mut archive, "META-INF/manifest.xml")?,
b"encryption-data",
)
{
return Err(Why::Encrypted);
}
for name in ["content.xml", "styles.xml"] {
if has(name) {
xml_text(&read(&mut archive, name)?, &mut text);
text.push('\n');
}
}
} else {
// An archive: each file inside, one level deep
for name in names.iter().filter(|n| !n.ends_with('/')) {
let bytes = read(&mut archive, name)?;
match extract_at("", Some(name), &bytes, limits, 1) {
Extracted::Text(inner) => {
text.push_str(&inner);
text.push('\n');
}
Extracted::NoText => {}
Extracted::NotInspectable(why) => return Err(why),
}
}
}
Ok(())
})();
match result {
Ok(()) => Extracted::Text(text),
Err(why) => Extracted::NotInspectable(why),
}
}
fn ooxml_text_part(name: &str) -> bool {
let xml = name.ends_with(".xml");
xml && (name == "word/document.xml"
|| [
"word/header",
"word/footer",
"word/footnotes",
"word/endnotes",
"word/comments",
]
.iter()
.any(|p| name.starts_with(p))
|| name.starts_with("xl/worksheets/sheet")
|| name.starts_with("ppt/slides/slide")
|| name.starts_with("ppt/notesSlides/"))
}
fn contains(haystack: &[u8], needle: &[u8]) -> bool {
haystack.windows(needle.len()).any(|w| w == needle)
}
/// The local name of a tag, without its namespace prefix.
fn local(name: &[u8]) -> &[u8] {
name.rsplit(|b| *b == b':').next().unwrap_or(name)
}
fn push_entity(entity: &[u8], out: &mut String) {
match entity {
b"lt" => out.push('<'),
b"gt" => out.push('>'),
b"amp" => out.push('&'),
b"apos" => out.push('\''),
b"quot" => out.push('"'),
_ => {
let code = match entity {
[b'#', b'x' | b'X', hex @ ..] => std::str::from_utf8(hex)
.ok()
.and_then(|h| u32::from_str_radix(h, 16).ok()),
[b'#', dec @ ..] => std::str::from_utf8(dec).ok().and_then(|d| d.parse().ok()),
_ => None,
};
if let Some(c) = code.and_then(char::from_u32) {
out.push(c);
}
}
}
}
/// Every text node, runs joined as written, a new line after each paragraph
/// or row and a tab after each cell, so a number split across runs is whole
/// again.
fn xml_text(xml: &[u8], out: &mut String) {
let mut reader = Reader::from_reader(xml);
let mut buf = Vec::new();
loop {
match reader.read_event_into(&mut buf) {
Ok(Event::Text(t)) => {
if let Ok(text) = t.xml_content(XmlVersion::Implicit1_0) {
out.push_str(&text);
}
}
Ok(Event::CData(t)) => out.push_str(&String::from_utf8_lossy(&t)),
Ok(Event::GeneralRef(entity)) => push_entity(&entity, out),
Ok(Event::End(e)) => match local(e.name().as_ref()) {
b"p" | b"h" | b"tr" | b"row" | b"table-row" | b"br" => out.push('\n'),
b"tc" | b"c" | b"table-cell" | b"tab" => out.push('\t'),
_ => {}
},
Ok(Event::Empty(e)) => match local(e.name().as_ref()) {
b"br" | b"line-break" => out.push('\n'),
b"tab" | b"s" => out.push(' '),
_ => {}
},
Ok(Event::Eof) | Err(_) => break,
_ => {}
}
buf.clear();
}
}
/// The text of each `item` element (a shared string in XLSX).
fn xml_strings(xml: &[u8], item: &str) -> Vec<String> {
let mut reader = Reader::from_reader(xml);
let mut buf = Vec::new();
let mut items = Vec::new();
let mut current: Option<String> = None;
loop {
match reader.read_event_into(&mut buf) {
Ok(Event::Start(e)) if local(e.name().as_ref()) == item.as_bytes() => {
current = Some(String::new())
}
Ok(Event::End(e)) if local(e.name().as_ref()) == item.as_bytes() => {
items.extend(current.take());
}
Ok(Event::Text(t)) => {
if let (Some(s), Ok(text)) =
(current.as_mut(), t.xml_content(XmlVersion::Implicit1_0))
{
s.push_str(&text);
}
}
Ok(Event::GeneralRef(entity)) => {
if let Some(s) = current.as_mut() {
push_entity(&entity, s);
}
}
Ok(Event::Eof) | Err(_) => break,
_ => {}
}
buf.clear();
}
items
}
/// A worksheet's cell values: numbers and inline strings. Cells holding a
/// shared string are skipped here; the shared strings are read whole.
fn xlsx_sheet(xml: &[u8], out: &mut String) {
let mut reader = Reader::from_reader(xml);
let mut buf = Vec::new();
let mut shared_cell = false;
let mut in_value = false;
loop {
match reader.read_event_into(&mut buf) {
Ok(Event::Start(e)) => match local(e.name().as_ref()) {
b"c" => {
shared_cell = e
.attributes()
.flatten()
.any(|a| a.key.as_ref() == b"t" && a.value.as_ref() == b"s");
}
b"v" | b"t" => in_value = true,
_ => {}
},
Ok(Event::End(e)) => match local(e.name().as_ref()) {
b"v" | b"t" => in_value = false,
b"c" => out.push('\t'),
b"row" => out.push('\n'),
_ => {}
},
// A shared string's cell holds only its index: the string itself
// is added with the shared strings
Ok(Event::Text(t)) if in_value && !shared_cell => {
if let Ok(text) = t.xml_content(XmlVersion::Implicit1_0) {
out.push_str(&text);
}
}
Ok(Event::Eof) | Err(_) => break,
_ => {}
}
buf.clear();
}
}
#[cfg(test)]
mod tests {
use super::*;
use std::io::Write;
use zip::{ZipWriter, write::SimpleFileOptions};
fn zip_of(files: &[(&str, &str)]) -> Vec<u8> {
let mut zip = ZipWriter::new(Cursor::new(Vec::new()));
for (name, body) in files {
zip.start_file(*name, SimpleFileOptions::default()).unwrap();
zip.write_all(body.as_bytes()).unwrap();
}
zip.finish().unwrap().into_inner()
}
fn text_of(extracted: Extracted) -> String {
match extracted {
Extracted::Text(text) => text,
other => panic!("expected text, got {other:?}"),
}
}
#[test]
fn plain_text_and_html() {
let limits = Limits::default();
assert_eq!(
text_of(extract("text/plain", None, b"card 4242", &limits)),
"card 4242"
);
let utf16: Vec<u8> = [0xFF, 0xFE]
.into_iter()
.chain("héllo".encode_utf16().flat_map(u16::to_le_bytes))
.collect();
assert_eq!(
text_of(extract(
"application/octet-stream",
Some("a.csv"),
&utf16,
&limits
)),
"héllo"
);
let html = "<html><style>p{}</style><p>Card&nbsp;4242</p><script>x()</script><td>a</td><td>b</td></html>";
let text = text_of(extract("text/html", None, html.as_bytes(), &limits));
assert!(
text.contains("Card 4242") && !text.contains("x()") && !text.contains("p{}"),
"{text:?}"
);
assert_eq!(
extract("image/png", Some("a.png"), b"\x89PNG....", &limits),
Extracted::NoText
);
}
#[test]
fn docx_joins_split_runs() {
let doc = r#"<w:document xmlns:w="w"><w:body><w:p><w:r><w:t>Card 4242 42</w:t></w:r><w:r><w:t>42 4242 4242</w:t></w:r></w:p><w:p><w:r><w:t>A &amp; B</w:t></w:r></w:p></w:body></w:document>"#;
let docx = zip_of(&[
("[Content_Types].xml", "<Types/>"),
("word/document.xml", doc),
]);
let text = text_of(extract(
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
Some("a.docx"),
&docx,
&Limits::default(),
));
assert!(text.contains("Card 4242 4242 4242 4242\nA & B"), "{text:?}");
}
#[test]
fn xlsx_numbers_and_shared_strings() {
let sheet = r#"<worksheet><sheetData><row><c r="A1" t="s"><v>0</v></c><c r="B1"><v>4242424242424242</v></c></row></sheetData></worksheet>"#;
let shared = r#"<sst><si><t>IBAN GB29 NWBK 6016 1331 9268 19</t></si></sst>"#;
let xlsx = zip_of(&[
("[Content_Types].xml", "<Types/>"),
("xl/sharedStrings.xml", shared),
("xl/worksheets/sheet1.xml", sheet),
]);
let text = text_of(extract("", Some("book.xlsx"), &xlsx, &Limits::default()));
assert!(
text.contains("4242424242424242") && text.contains("GB29 NWBK 6016 1331 9268 19"),
"{text:?}"
);
// The shared string's index isn't read as a value
assert!(
!text.contains("\t0\t") && !text.starts_with('0'),
"{text:?}"
);
}
#[test]
fn opendocument_and_encrypted_opendocument() {
let content = r#"<office:document-content xmlns:text="t"><text:p>SSN 078-05-1120</text:p></office:document-content>"#;
let odt = zip_of(&[
("mimetype", "application/vnd.oasis.opendocument.text"),
("content.xml", content),
]);
assert!(
text_of(extract("", Some("a.odt"), &odt, &Limits::default()))
.contains("SSN 078-05-1120")
);
let manifest = r#"<manifest:manifest><manifest:file-entry><manifest:encryption-data/></manifest:file-entry></manifest:manifest>"#;
let locked = zip_of(&[
("mimetype", "application/vnd.oasis.opendocument.text"),
("META-INF/manifest.xml", manifest),
("content.xml", "x"),
]);
assert_eq!(
extract("", Some("a.odt"), &locked, &Limits::default()),
Extracted::NotInspectable(Why::Encrypted)
);
}
#[test]
fn archives() {
let limits = Limits::default();
let archive = zip_of(&[
("notes/a.txt", "card 4242424242424242"),
("b.png", "\u{89}PNG"),
]);
assert!(
text_of(extract("application/zip", Some("x.zip"), &archive, &limits))
.contains("4242424242424242")
);
let nested = zip_of(&[(
"inner.zip",
std::str::from_utf8(&[b'P', b'K', 3, 4]).unwrap(),
)]);
assert_eq!(
extract("application/zip", Some("x.zip"), &nested, &limits),
Extracted::NotInspectable(Why::NestedArchive)
);
// Password-protected
let mut zip = ZipWriter::new(Cursor::new(Vec::new()));
zip.start_file(
"secret.txt",
SimpleFileOptions::default().with_aes_encryption(zip::AesMode::Aes256, "pw"),
)
.unwrap();
zip.write_all(b"4242424242424242").unwrap();
let locked = zip.finish().unwrap().into_inner();
assert_eq!(
extract("application/zip", Some("x.zip"), &locked, &limits),
Extracted::NotInspectable(Why::Encrypted)
);
// Past the limits
let small = Limits {
max_unpacked: 10,
max_entries: 1,
};
assert_eq!(
extract("application/zip", Some("x.zip"), &archive, &small),
Extracted::NotInspectable(Why::TooLarge)
);
assert_eq!(
extract("application/zip", None, b"PK\x03\x04garbage", &limits),
Extracted::NotInspectable(Why::Damaged)
);
}
#[test]
fn not_inspectable_kinds() {
let limits = Limits::default();
assert_eq!(
extract("application/octet-stream", None, b"%PDF-1.7 ...", &limits),
Extracted::NotInspectable(Why::Pdf)
);
let mut ole = OLE_MAGIC.to_vec();
ole.extend(std::iter::repeat_n(0, 64));
assert_eq!(
extract("", Some("old.doc"), &ole, &limits),
Extracted::NotInspectable(Why::LegacyOffice)
);
ole.extend("EncryptedPackage".encode_utf16().flat_map(u16::to_le_bytes));
assert_eq!(
extract("", Some("new.docx"), &ole, &limits),
Extracted::NotInspectable(Why::Encrypted)
);
}
}
+23
View File
@@ -0,0 +1,23 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Data loss prevention and mail flow rules (dlp-and-mail-flow-rules spec).
//!
//! Pure functions over text and attachment bytes, so everything here is
//! unit-tested without a server:
//!
//! - [`detectors`]: find identifiers in text (payment cards, IBANs,
//! national ID numbers, keys), each by its published format and check
//! (§2.3);
//! - [`words`]: an organization's own word lists and patterns;
//! - [`extract`]: the text of an attachment, or why it can't be read.
//!
//! Nothing here writes what it finds anywhere: callers get counts, and the
//! matched text never leaves the evaluation (§2.7).
pub mod detectors;
pub mod extract;
pub mod words;
+109
View File
@@ -0,0 +1,109 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! An organization's own word lists and patterns (§2.3). Both count
//! occurrences, not distinct values: "confidential" three times is three.
use aho_corasick::{AhoCorasick, AhoCorasickBuilder, MatchKind};
use regex::{Regex, RegexBuilder};
/// How large a compiled pattern may grow. Keeps a rule someone writes from
/// making every message slow to send.
const PATTERN_SIZE_LIMIT: usize = 1 << 20;
/// Words and phrases, matched whole and ignoring case.
#[derive(Debug, Clone)]
pub struct WordList {
matcher: AhoCorasick,
}
impl WordList {
/// Builds a list from words or phrases; empty entries are skipped.
pub fn new<I, S>(words: I) -> Result<Self, String>
where
I: IntoIterator<Item = S>,
S: AsRef<str>,
{
let words: Vec<String> = words
.into_iter()
.map(|w| w.as_ref().trim().to_lowercase())
.filter(|w| !w.is_empty())
.collect();
if words.is_empty() {
return Err("The list has no words".into());
}
AhoCorasickBuilder::new()
.match_kind(MatchKind::LeftmostLongest)
.build(&words)
.map(|matcher| Self { matcher })
.map_err(|err| err.to_string())
}
/// How many times any word of the list appears in `text`.
pub fn count(&self, text: &str) -> usize {
let text = text.to_lowercase();
self.matcher
.find_iter(&text)
.filter(|m| super::detectors::stands_alone(&text, m.start(), m.end()))
.count()
}
}
/// An organization's regular expression.
#[derive(Debug, Clone)]
pub struct Pattern {
regex: Regex,
}
impl Pattern {
/// Compiles `pattern`, or says why it can't be used. Matching ignores
/// case unless the pattern turns that off with `(?-i)`.
pub fn new(pattern: &str) -> Result<Self, String> {
RegexBuilder::new(pattern)
.case_insensitive(true)
.size_limit(PATTERN_SIZE_LIMIT)
.build()
.map(|regex| Self { regex })
.map_err(|err| err.to_string())
}
/// How many times the pattern matches in `text`.
pub fn count(&self, text: &str) -> usize {
self.regex.find_iter(text).filter(|m| !m.is_empty()).count()
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn words_whole_and_any_case() {
let list = WordList::new(["Project Falcon", "confidential", " "]).unwrap();
assert_eq!(
list.count(
"CONFIDENTIAL: project falcon notes. Not confidentiality, not projectfalcon."
),
2
);
assert_eq!(list.count("Confidential, confidential and confidential"), 3);
// Non-ASCII case folding
let list = WordList::new(["GEHEIM", "Straße"]).unwrap();
assert_eq!(list.count("streng geheim, STRASSE ist nicht Straße"), 2);
assert!(WordList::new(["", " "]).is_err());
}
#[test]
fn patterns() {
let pattern = Pattern::new(r"\bPRJ-\d{4}\b").unwrap();
assert_eq!(pattern.count("prj-1234 and PRJ-5678, not PRJ-12"), 2);
assert!(Pattern::new("(unclosed").is_err());
// Too large to compile within the limit
assert!(Pattern::new(r"\w{1000}\w{1000}\w{1000}").is_err());
// Empty matches don't count
assert_eq!(Pattern::new("x*").unwrap().count("abc"), 0);
}
}