diff --git a/Cargo.lock b/Cargo.lock index f624cd5..d14ec2d 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -3947,9 +3947,12 @@ name = "inbuxa-features" version = "0.16.22" dependencies = [ "ahash", + "aho-corasick", "base64 0.23.1", "flate2", "jmap_proto", + "quick-xml 0.41.0", + "regex", "registry", "serde", "serde_json", @@ -3961,6 +3964,7 @@ dependencies = [ "types", "utils", "xxhash-rust", + "zip", ] [[package]] diff --git a/crates/features/Cargo.toml b/crates/features/Cargo.toml index cbee6d6..3054682 100644 --- a/crates/features/Cargo.toml +++ b/crates/features/Cargo.toml @@ -21,6 +21,11 @@ base64 = "0.23" sha2 = "0.11" flate2 = "1.1" tokio = { version = "1.53", features = ["sync", "rt"] } +# inbuxa: DLP detectors and attachment text (dlp-and-mail-flow-rules spec) +regex = "1.13.1" +aho-corasick = "1.1" +zip = "8.6" +quick-xml = "0.41" [dev-dependencies] tokio = { version = "1.53", features = ["macros", "rt"] } diff --git a/crates/features/src/lib.rs b/crates/features/src/lib.rs index 9145a72..9df288c 100644 --- a/crates/features/src/lib.rs +++ b/crates/features/src/lib.rs @@ -23,6 +23,7 @@ pub mod audit; pub mod branding; pub mod hold; pub mod lock; +pub mod mailflow; pub mod masked_email; pub mod privacy; pub mod security; diff --git a/crates/features/src/mailflow/detectors/any.rs b/crates/features/src/mailflow/detectors/any.rs new file mode 100644 index 0000000..1120cb7 --- /dev/null +++ b/crates/features/src/mailflow/detectors/any.rs @@ -0,0 +1,522 @@ +/* + * SPDX-FileCopyrightText: 2026 Coffey Labs + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +//! Detectors that aren't tied to one country (§2.3, region "Any"). + +use super::{Detector, Findings, Region, Strength, checks, digits, stands_alone, word_near}; +use regex::Regex; +use std::sync::LazyLock; + +pub static DETECTORS: &[Detector] = &[ + Detector::new( + "payment-card", + "Payment card number", + Region::Any, + Strength::Checked, + payment_card, + ), + Detector::new("iban", "IBAN", Region::Any, Strength::Checked, iban), + Detector::new( + "swift-bic", + "SWIFT/BIC code", + Region::Any, + Strength::NeedsWord, + swift_bic, + ), + Detector::new( + "email-addresses", + "Email addresses", + Region::Any, + Strength::Checked, + email_addresses, + ), + Detector::new( + "phone-numbers", + "Phone numbers", + Region::Any, + Strength::NeedsWord, + phone_numbers, + ), + Detector::new( + "date-of-birth", + "Date of birth", + Region::Any, + Strength::NeedsWord, + date_of_birth, + ), + Detector::new( + "passport", + "Passport number", + Region::Any, + Strength::NeedsWord, + passport, + ), + Detector::new( + "private-key", + "Private key", + Region::Any, + Strength::Checked, + private_key, + ), + Detector::new( + "credentials", + "Cloud and service credentials", + Region::Any, + Strength::Checked, + credentials, + ), +]; + +fn re(pattern: &str) -> Regex { + Regex::new(pattern).expect("detector pattern") +} + +// --- Payment cards -------------------------------------------------------- + +/// Issuer prefixes (ISO/IEC 7812 IINs) and the lengths each network issues. +fn card_network(number: &str) -> bool { + let len = number.len(); + let prefix = |n: usize| number[..n].parse::().unwrap_or(0); + match number.as_bytes()[0] { + // Visa + b'4' => matches!(len, 13 | 16 | 19), + b'5' => { + // Mastercard 51–55; Maestro 50, 56–58 + (51..=55).contains(&prefix(2)) && len == 16 + || matches!(prefix(2), 50 | 56..=58) && (12..=19).contains(&len) + } + // Mastercard 2221–2720 + b'2' => (2221..=2720).contains(&prefix(4)) && len == 16, + b'3' => { + // American Express 34, 37; JCB 3528–3589; Diners 300–305, 36, 38, 39 + matches!(prefix(2), 34 | 37) && len == 15 + || (3528..=3589).contains(&prefix(4)) && (16..=19).contains(&len) + || ((300..=305).contains(&prefix(3)) || matches!(prefix(2), 36 | 38 | 39)) + && (14..=19).contains(&len) + } + // Discover 6011, 644–649, 65; UnionPay 62; Maestro 6x + b'6' => (12..=19).contains(&len), + _ => false, + } +} + +fn is_card(number: &str) -> bool { + (12..=19).contains(&number.len()) && card_network(number) && checks::luhn(number) +} + +static CARD: LazyLock = LazyLock::new(|| re(r"\b\d(?:[ -]?\d){11,18}\b")); + +fn payment_card(text: &str, findings: &mut Findings) { + for m in CARD.find_iter(text) { + if !stands_alone(text, m.start(), m.end()) { + continue; + } + let whole = digits(m.as_str()); + if is_card(&whole) { + findings.insert(whole); + continue; + } + // Two numbers side by side ("4242 4242 4242 4242 2031"): try each + // run of whole groups + let groups: Vec = m.as_str().split([' ', '-']).map(digits).collect(); + 'runs: for from in 0..groups.len() { + let mut number = String::new(); + for group in &groups[from..] { + number.push_str(group); + if is_card(&number) { + findings.insert(number); + break 'runs; + } + } + } + } +} + +// --- IBAN ----------------------------------------------------------------- + +static IBAN: LazyLock = + LazyLock::new(|| re(r"\b[A-Za-z]{2}\d{2}(?:[ ]?[A-Za-z0-9]){11,30}")); + +fn iban(text: &str, findings: &mut Findings) { + // The pattern can run on into the next words, even the next IBAN: after + // each hit, look again from where that IBAN ended + let mut from = 0; + while let Some(m) = IBAN.find_at(text, from) { + from = m.start() + 1; + let compact = m.as_str().replace(' ', "").to_ascii_uppercase(); + let Some(len) = checks::iban_length(&compact[..2]) else { + continue; + }; + if compact.len() < len { + continue; + } + // Where the country's length ends in the text, spaces counted + let mut seen = 0; + let Some(end) = m + .as_str() + .char_indices() + .find(|(_, c)| { + if *c != ' ' { + seen += 1; + } + seen == len + }) + .map(|(i, c)| m.start() + i + c.len_utf8()) + else { + continue; + }; + let candidate = &compact[..len]; + if stands_alone(text, m.start(), end) && checks::iban(candidate) { + findings.insert(candidate); + from = end; + } + } +} + +// --- SWIFT/BIC ------------------------------------------------------------ + +static BIC: LazyLock = + LazyLock::new(|| re(r"\b[A-Z]{4}[A-Z]{2}[A-Z0-9]{2}(?:[A-Z0-9]{3})?\b")); + +const BIC_WORDS: &[&str] = &[ + "swift", + "bic", + "swift/bic", + "bank", + "banque", + "bankverbindung", +]; + +fn swift_bic(text: &str, findings: &mut Findings) { + for m in BIC.find_iter(text) { + let code = m.as_str(); + if checks::is_country(&code[4..6]) && word_near(text, m.start(), m.end(), BIC_WORDS) { + findings.insert(code); + } + } +} + +// --- Contact lists -------------------------------------------------------- + +static EMAIL: LazyLock = + LazyLock::new(|| re(r"(?i)\b[a-z0-9._%+-]+@[a-z0-9-]+(?:\.[a-z0-9-]+)*\.[a-z]{2,}\b")); + +fn email_addresses(text: &str, findings: &mut Findings) { + for m in EMAIL.find_iter(text) { + findings.insert(m.as_str().to_lowercase()); + } +} + +/// International form: found alone. National form: only with a word. +static PHONE_INTL: LazyLock = LazyLock::new(|| re(r"\+\d{1,3}(?:[ .-]?\(?\d{1,4}\)?){2,5}")); +static PHONE_NATIONAL: LazyLock = + LazyLock::new(|| re(r"\(?\d{2,4}\)?[ .-]\d{3,4}[ .-]\d{3,4}")); + +const PHONE_WORDS: &[&str] = &[ + "phone", + "tel", + "telephone", + "mobile", + "cell", + "fax", + "telefon", + "téléphone", + "teléfono", + "telefono", + "handy", + "portable", + "móvil", + "cellulare", + "mobiel", +]; + +fn phone_numbers(text: &str, findings: &mut Findings) { + let mut international = Vec::new(); + for m in PHONE_INTL.find_iter(text) { + let number = digits(m.as_str()); + if (8..=15).contains(&number.len()) && stands_alone(text, m.start() + 1, m.end()) { + findings.insert(number); + international.push(m.range()); + } + } + for m in PHONE_NATIONAL.find_iter(text) { + let number = digits(m.as_str()); + // Not the tail of an international number already counted + if international.iter().any(|r| r.contains(&m.start())) { + continue; + } + if (9..=11).contains(&number.len()) + && stands_alone(text, m.start(), m.end()) + && !text[..m.start()].ends_with('+') + && word_near(text, m.start(), m.end(), PHONE_WORDS) + { + findings.insert(number); + } + } +} + +// --- Date of birth -------------------------------------------------------- + +static DATE_ISO: LazyLock = LazyLock::new(|| re(r"\b(\d{4})-(\d{2})-(\d{2})\b")); +static DATE_NUMERIC: LazyLock = + LazyLock::new(|| re(r"\b(\d{1,2})[./-](\d{1,2})[./-](\d{4})\b")); +static DATE_WORDS: LazyLock = LazyLock::new(|| { + re( + r"(?i)\b(?:(\d{1,2})\s+(jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)[a-z]*\.?,?\s+(\d{4})|(jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)[a-z]*\.?\s+(\d{1,2}),?\s+(\d{4}))\b", + ) +}); + +const BIRTH_WORDS: &[&str] = &[ + "born", + "birth", + "dob", + "d.o.b", + "birthday", + "birthdate", + "geburtsdatum", + "geboren", + "naissance", + "né le", + "née le", + "nacimiento", + "nacido", + "nacida", + "nascita", + "nato il", + "nata il", + "geboortedatum", + "födelsedatum", + "fødselsdato", + "syntymäaika", + "urodzenia", + "nascimento", +]; + +fn valid_date(year: u32, month: u32, day: u32) -> bool { + let days = match month { + 1 | 3 | 5 | 7 | 8 | 10 | 12 => 31, + 4 | 6 | 9 | 11 => 30, + 2 if year.is_multiple_of(4) && (!year.is_multiple_of(100) || year.is_multiple_of(400)) => { + 29 + } + 2 => 28, + _ => return false, + }; + (1900..=2100).contains(&year) && (1..=days).contains(&day) +} + +fn month_number(name: &str) -> u32 { + const MONTHS: [&str; 12] = [ + "jan", "feb", "mar", "apr", "may", "jun", "jul", "aug", "sep", "oct", "nov", "dec", + ]; + let name = name.to_lowercase(); + MONTHS + .iter() + .position(|m| *m == name) + .map_or(0, |i| i as u32 + 1) +} + +fn date_of_birth(text: &str, findings: &mut Findings) { + let mut add = |start: usize, end: usize, key: String| { + if word_near(text, start, end, BIRTH_WORDS) { + findings.insert(key); + } + }; + let num = |s: &str| s.parse::().unwrap_or(0); + for c in DATE_ISO.captures_iter(text) { + let (y, m, d) = (num(&c[1]), num(&c[2]), num(&c[3])); + let whole = c.get(0).unwrap(); + if valid_date(y, m, d) { + add(whole.start(), whole.end(), format!("{y:04}{m:02}{d:02}")); + } + } + for c in DATE_NUMERIC.captures_iter(text) { + let (a, b, y) = (num(&c[1]), num(&c[2]), num(&c[3])); + let whole = c.get(0).unwrap(); + // Day first or month first: either reading that is a real date + if valid_date(y, b, a) || valid_date(y, a, b) { + add(whole.start(), whole.end(), whole.as_str().to_string()); + } + } + for c in DATE_WORDS.captures_iter(text) { + let whole = c.get(0).unwrap(); + let (d, m, y) = match (c.get(1), c.get(4)) { + (Some(d), _) => (num(d.as_str()), month_number(&c[2]), num(&c[3])), + (_, Some(m)) => (num(&c[5]), month_number(m.as_str()), num(&c[6])), + _ => continue, + }; + if valid_date(y, m, d) { + add(whole.start(), whole.end(), format!("{y:04}{m:02}{d:02}")); + } + } +} + +// --- Passport ------------------------------------------------------------- + +static PASSPORT: LazyLock = LazyLock::new(|| re(r"\b[A-Z0-9]{6,9}\b")); + +const PASSPORT_WORDS: &[&str] = &[ + "passport", + "passeport", + "reisepass", + "pasaporte", + "passaporto", + "paspoort", + "passnummer", + "pass-nr", + "passport no", + "pasaporte n.º", + "passaporte", +]; + +fn passport(text: &str, findings: &mut Findings) { + for m in PASSPORT.find_iter(text) { + let value = m.as_str(); + if value.bytes().filter(u8::is_ascii_digit).count() >= 5 + && word_near(text, m.start(), m.end(), PASSPORT_WORDS) + { + findings.insert(value); + } + } +} + +// --- Keys and credentials ------------------------------------------------- + +static PRIVATE_KEY: LazyLock = LazyLock::new(|| { + re( + r"-----BEGIN (?:(?:RSA|EC|DSA|OPENSSH|ENCRYPTED|PGP) )?PRIVATE KEY(?: BLOCK)?-----\s*([A-Za-z0-9+/=:\s-]{0,64})", + ) +}); + +fn private_key(text: &str, findings: &mut Findings) { + for c in PRIVATE_KEY.captures_iter(text) { + // Each key once, by the start of its body + let body: String = c[1].chars().filter(|c| !c.is_whitespace()).collect(); + let whole = c.get(0).unwrap(); + findings.insert(if body.is_empty() { + format!("@{}", whole.start()) + } else { + body + }); + } +} + +/// Published token formats: AWS access key IDs, GitHub tokens, Slack +/// tokens, Stripe live secret and restricted keys, Google API keys. +static CREDENTIAL: LazyLock = LazyLock::new(|| { + re(concat!( + r"\b(?:", + r"(?:AKIA|ASIA|ABIA|ACCA)[A-Z0-9]{16}", + r"|gh[pousr]_[A-Za-z0-9]{36}", + r"|github_pat_[A-Za-z0-9_]{82}", + r"|xox[abposr]-[A-Za-z0-9-]{10,72}", + r"|(?:sk|rk)_live_[A-Za-z0-9]{24,99}", + r"|AIza[0-9A-Za-z_-]{35}", + r")\b" + )) +}); + +fn credentials(text: &str, findings: &mut Findings) { + for m in CREDENTIAL.find_iter(text) { + findings.insert(m.as_str()); + } +} + +#[cfg(test)] +mod tests { + use crate::mailflow::detectors::by_id; + + fn count(id: &str, text: &str) -> usize { + by_id(id).unwrap().count(text) + } + + #[test] + fn payment_cards() { + // Networks' and processors' published test numbers + let text = "Visa 4242 4242 4242 4242, MC 5555-5555-5555-4444, Amex 378282246310005, \ + Discover 6011111111111117, JCB 3566002020360505, Diners 30569309025904, \ + UnionPay 6200000000000005, Mastercard 2-series 2223003122003222"; + assert_eq!(count("payment-card", text), 8); + // Luhn fails, wrong network length, inside a longer number + assert_eq!(count("payment-card", "4242424242424241"), 0); + assert_eq!(count("payment-card", "378282246310005 0"), 1); + assert_eq!(count("payment-card", "order 94242424242424242 shipped"), 0); + // The same number twice counts once + assert_eq!( + count("payment-card", "4242424242424242 and 4242-4242-4242-4242"), + 1 + ); + // A card followed by a year + assert_eq!(count("payment-card", "card 4242 4242 4242 4242 2031"), 1); + } + + #[test] + fn ibans() { + let text = + "Pay GB29 NWBK 6016 1331 9268 19 or de89370400440532013000 (NL91ABNA0417164300)."; + assert_eq!(count("iban", text), 3); + assert_eq!(count("iban", "GB29 NWBK 6016 1331 9268 18"), 0); + // Runs into the next word: still found at the country's length + assert_eq!(count("iban", "IBAN NL91ABNA0417164300 BIC ABNANL2A"), 1); + } + + #[test] + fn swift_codes_need_a_word() { + assert_eq!(count("swift-bic", "SWIFT: DEUTDEFF500"), 1); + assert_eq!(count("swift-bic", "BIC NWBKGB2L"), 1); + assert_eq!(count("swift-bic", "HAPPYDAYS DEUTDEFF"), 0); + // Not a country in positions 5–6 + assert_eq!(count("swift-bic", "BIC DEUTZZFF"), 0); + } + + #[test] + fn email_and_phone_lists() { + let list = "a@example.com, B@Example.com, c.d+x@mail.example.org, a@example.com"; + assert_eq!(count("email-addresses", list), 3); + assert_eq!( + count("phone-numbers", "+44 20 7946 0958, +1 (415) 555-2671"), + 2 + ); + assert_eq!(count("phone-numbers", "call 020 7946 0958"), 0); + assert_eq!(count("phone-numbers", "Tel: 020 7946 0958"), 1); + assert_eq!(count("phone-numbers", "invoice 020 7946 0958"), 0); + // One number, not also its national tail + assert_eq!(count("phone-numbers", "Tel: +44 20 7946 0958"), 1); + } + + #[test] + fn dates_of_birth() { + assert_eq!(count("date-of-birth", "DOB: 1984-02-29"), 1); + assert_eq!(count("date-of-birth", "Geburtsdatum 31.12.1970"), 1); + assert_eq!(count("date-of-birth", "born on March 3, 1962"), 1); + assert_eq!(count("date-of-birth", "date of birth 3 Mar 1962"), 1); + // Not a real date, no word, a meeting + assert_eq!(count("date-of-birth", "DOB: 1985-02-29"), 0); + assert_eq!(count("date-of-birth", "invoice 1984-02-29"), 0); + assert_eq!(count("date-of-birth", "Meeting on 12/05/2026"), 0); + } + + #[test] + fn passports_need_a_word() { + assert_eq!(count("passport", "Passport number: 533380006"), 1); + assert_eq!(count("passport", "Reisepass C01X00T47"), 1); + assert_eq!(count("passport", "Order 533380006 shipped"), 0); + // Mostly letters: a word, not a number + assert_eq!(count("passport", "passport PASSWORD"), 0); + } + + #[test] + fn keys_and_credentials() { + let key = "-----BEGIN OPENSSH PRIVATE KEY-----\nb3BlbnNzaC1rZXktdjEAAAAABG5vbmUAAAAEbm9uZQ\n-----END OPENSSH PRIVATE KEY-----"; + assert_eq!(count("private-key", key), 1); + assert_eq!(count("private-key", "-----BEGIN PUBLIC KEY-----\nMFkw"), 0); + // Documentation examples of each format + let tokens = "AKIAIOSFODNN7EXAMPLE ghp_0123456789abcdefghijklmnopqrstuvwxyz \ + AIzaSyA-0123456789abcdefghijklmnopqrstu"; + assert_eq!(count("credentials", tokens), 3); + assert_eq!(count("credentials", "AKIA123 ghp_short"), 0); + } +} diff --git a/crates/features/src/mailflow/detectors/checks.rs b/crates/features/src/mailflow/detectors/checks.rs new file mode 100644 index 0000000..d0320be --- /dev/null +++ b/crates/features/src/mailflow/detectors/checks.rs @@ -0,0 +1,227 @@ +/* + * SPDX-FileCopyrightText: 2026 Coffey Labs + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +//! Check-digit algorithms, each from its public definition. + +/// The Luhn check (ISO/IEC 7812-1, Annex B) over a string of ASCII digits. +pub fn luhn(digits: &str) -> bool { + if digits.len() < 2 || !digits.bytes().all(|b| b.is_ascii_digit()) { + return false; + } + let sum: u32 = digits + .bytes() + .rev() + .enumerate() + .map(|(i, b)| { + let d = u32::from(b - b'0'); + if i % 2 == 1 { + let d = d * 2; + if d > 9 { d - 9 } else { d } + } else { + d + } + }) + .sum(); + sum % 10 == 0 +} + +/// ISO 13616 IBAN lengths, by country, from the IBAN registry. +const IBAN_LENGTHS: &[(&str, usize)] = &[ + ("AD", 24), + ("AE", 23), + ("AL", 28), + ("AT", 20), + ("AZ", 28), + ("BA", 20), + ("BE", 16), + ("BG", 22), + ("BH", 22), + ("BI", 27), + ("BR", 29), + ("BY", 28), + ("CH", 21), + ("CR", 22), + ("CY", 28), + ("CZ", 24), + ("DE", 22), + ("DJ", 27), + ("DK", 18), + ("DO", 28), + ("EE", 20), + ("EG", 29), + ("ES", 24), + ("FI", 18), + ("FK", 18), + ("FO", 18), + ("FR", 27), + ("GB", 22), + ("GE", 22), + ("GI", 23), + ("GL", 18), + ("GR", 27), + ("GT", 28), + ("HN", 28), + ("HR", 21), + ("HU", 28), + ("IE", 22), + ("IL", 23), + ("IQ", 23), + ("IS", 26), + ("IT", 27), + ("JO", 30), + ("KW", 30), + ("KZ", 20), + ("LB", 28), + ("LC", 32), + ("LI", 21), + ("LT", 20), + ("LU", 20), + ("LV", 21), + ("LY", 25), + ("MC", 27), + ("MD", 24), + ("ME", 22), + ("MK", 19), + ("MN", 20), + ("MR", 27), + ("MT", 31), + ("MU", 30), + ("NI", 28), + ("NL", 18), + ("NO", 15), + ("OM", 23), + ("PK", 24), + ("PL", 28), + ("PS", 29), + ("PT", 25), + ("QA", 29), + ("RO", 24), + ("RS", 22), + ("RU", 33), + ("SA", 24), + ("SC", 31), + ("SD", 18), + ("SE", 24), + ("SI", 19), + ("SK", 24), + ("SM", 27), + ("SO", 23), + ("ST", 25), + ("SV", 28), + ("TL", 23), + ("TN", 24), + ("TR", 26), + ("UA", 29), + ("VA", 22), + ("VG", 24), + ("XK", 20), + ("YE", 30), +]; + +/// The IBAN length for a country code, if the country uses IBANs. +pub fn iban_length(country: &str) -> Option { + IBAN_LENGTHS + .iter() + .find(|(code, _)| *code == country) + .map(|(_, len)| *len) +} + +/// ISO 13616 / ISO 7064 MOD 97-10 over an IBAN with no spaces, upper case: +/// move the first four characters to the end, turn letters into 10–35, and +/// the number mod 97 must be 1. Also checks the country's length. +pub fn iban(iban: &str) -> bool { + if iban.len() < 5 + || !iban + .bytes() + .all(|b| b.is_ascii_uppercase() || b.is_ascii_digit()) + { + return false; + } + if iban_length(&iban[..2]) != Some(iban.len()) + || !iban[2..4].bytes().all(|b| b.is_ascii_digit()) + { + return false; + } + let mut remainder: u32 = 0; + for b in iban[4..].bytes().chain(iban[..4].bytes()) { + let value = if b.is_ascii_digit() { + u32::from(b - b'0') + } else { + u32::from(b - b'A') + 10 + }; + remainder = if value >= 10 { + (remainder * 100 + value) % 97 + } else { + (remainder * 10 + value) % 97 + }; + } + remainder == 1 +} + +/// ISO 3166-1 alpha-2 country codes, for SWIFT/BIC positions 5–6. +const COUNTRIES: &str = "AD AE AF AG AI AL AM AO AQ AR AS AT AU AW AX AZ BA BB BD BE BF BG BH BI BJ \ +BL BM BN BO BQ BR BS BT BV BW BY BZ CA CC CD CF CG CH CI CK CL CM CN CO CR CU CV CW CX CY CZ DE DJ \ +DK DM DO DZ EC EE EG EH ER ES ET FI FJ FK FM FO FR GA GB GD GE GF GG GH GI GL GM GN GP GQ GR GS GT \ +GU GW GY HK HM HN HR HT HU ID IE IL IM IN IO IQ IR IS IT JE JM JO JP KE KG KH KI KM KN KP KR KW KY \ +KZ LA LB LC LI LK LR LS LT LU LV LY MA MC MD ME MF MG MH MK ML MM MN MO MP MQ MR MS MT MU MV MW MX \ +MY MZ NA NC NE NF NG NI NL NO NP NR NU NZ OM PA PE PF PG PH PK PL PM PN PR PS PT PW PY QA RE RO RS \ +RU RW SA SB SC SD SE SG SH SI SJ SK SL SM SN SO SR SS ST SV SX SY SZ TC TD TF TG TH TJ TK TL TM TN \ +TO TR TT TV TW TZ UA UG UM US UY UZ VA VC VE VG VI VN VU WF WS XK YE YT ZA ZM ZW"; + +pub fn is_country(code: &str) -> bool { + code.len() == 2 && COUNTRIES.split(' ').any(|c| c == code) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn luhn_known_numbers() { + // Published test card numbers + for good in [ + "4242424242424242", + "5555555555554444", + "378282246310005", + "79927398713", + ] { + assert!(luhn(good), "{good}"); + } + for bad in ["4242424242424241", "79927398710", "1", "12a4"] { + assert!(!luhn(bad), "{bad}"); + } + } + + #[test] + fn iban_registry_examples() { + // The IBAN registry's own examples + for good in [ + "GB29NWBK60161331926819", + "DE89370400440532013000", + "FR1420041010050500013M02606", + "NL91ABNA0417164300", + "BE68539007547034", + "NO9386011117947", + "CH9300762011623852957", + ] { + assert!(iban(good), "{good}"); + } + for bad in [ + "GB29NWBK60161331926818", // check fails + "GB29NWBK6016133192681", // too short for GB + "ZZ29NWBK60161331926819", // no such country + "DE8937040044053201300A", // letters where DE has none still fail mod 97 + ] { + assert!(!iban(bad), "{bad}"); + } + } + + #[test] + fn countries() { + assert!(is_country("DE") && is_country("US") && is_country("XK")); + assert!(!is_country("ZZ") && !is_country("D")); + } +} diff --git a/crates/features/src/mailflow/detectors/mod.rs b/crates/features/src/mailflow/detectors/mod.rs new file mode 100644 index 0000000..66b1eab --- /dev/null +++ b/crates/features/src/mailflow/detectors/mod.rs @@ -0,0 +1,199 @@ +/* + * SPDX-FileCopyrightText: 2026 Coffey Labs + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +//! Detectors (dlp-and-mail-flow-rules spec, §2.3): each finds one kind of +//! identifier in text and reports the distinct ones it found. +//! +//! A detector is one of two strengths: +//! +//! - **Checked**: the identifier carries a published check digit or +//! checksum, so a random number rarely passes; found on its own. +//! - **Needs a word**: the format alone is too common, so a candidate counts +//! only with a corroborating word within [`WINDOW`] characters either +//! side. +//! +//! Findings are distinct normalized values (digits only, upper case), so the +//! same card number pasted twice counts once. They stay in memory: callers +//! read only [`Findings::len`]. + +pub mod any; +pub mod checks; + +use ahash::AHashSet; + +/// How far, in characters, a corroborating word may be from a candidate. +pub const WINDOW: usize = 50; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Strength { + Checked, + NeedsWord, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Region { + Any, + Us, + Uk, + Canada, + Australia, + Eu, + Europe, + Asia, + Americas, + Africa, +} + +/// The distinct values one detector found. +#[derive(Debug, Default)] +pub struct Findings(AHashSet); + +impl Findings { + pub fn insert(&mut self, value: impl Into) { + self.0.insert(value.into()); + } + + pub fn len(&self) -> usize { + self.0.len() + } + + pub fn is_empty(&self) -> bool { + self.0.is_empty() + } +} + +pub struct Detector { + /// Stable id, stored in rules: `payment-card`, `iban`, `us-ssn`. + pub id: &'static str, + pub name: &'static str, + pub region: Region, + pub strength: Strength, + find: fn(&str, &mut Findings), +} + +impl Detector { + pub const fn new( + id: &'static str, + name: &'static str, + region: Region, + strength: Strength, + find: fn(&str, &mut Findings), + ) -> Self { + Self { + id, + name, + region, + strength, + find, + } + } + + /// Adds what this detector finds in `text` to `findings`. Call once per + /// piece of text (subject, each part, each attachment) with the same + /// `findings`, then read its length. + pub fn find(&self, text: &str, findings: &mut Findings) { + (self.find)(text, findings) + } + + /// The distinct values found in one text. + pub fn count(&self, text: &str) -> usize { + let mut findings = Findings::default(); + self.find(text, &mut findings); + findings.len() + } +} + +/// Every detector, in the order the console lists them. +pub fn all() -> impl Iterator { + any::DETECTORS.iter() +} + +pub fn by_id(id: &str) -> Option<&'static Detector> { + all().find(|detector| detector.id == id) +} + +/// Whether one of `words` appears, as a whole word and ignoring case, within +/// [`WINDOW`] characters before `start` or after `end` (byte offsets of the +/// candidate in `text`). The window is widened by the longest word, so a +/// word that reaches into it still counts whole. +pub fn word_near(text: &str, start: usize, end: usize, words: &[&str]) -> bool { + let reach = WINDOW + words.iter().map(|w| w.chars().count()).max().unwrap_or(0); + let before = text[..start] + .char_indices() + .rev() + .nth(reach - 1) + .map_or(0, |(i, _)| i); + let after = text[end..] + .char_indices() + .nth(reach) + .map_or(text.len(), |(i, _)| end + i); + let window = text[before..after].to_lowercase(); + words.iter().any(|word| contains_word(&window, word)) +} + +/// Whether `word` (lower case) appears in `haystack` (lower case) with no +/// letter or digit on either side. +pub fn contains_word(haystack: &str, word: &str) -> bool { + haystack.match_indices(word).any(|(i, _)| { + let before_ok = haystack[..i] + .chars() + .next_back() + .is_none_or(|c| !c.is_alphanumeric()); + let after_ok = haystack[i + word.len()..] + .chars() + .next() + .is_none_or(|c| !c.is_alphanumeric()); + before_ok && after_ok + }) +} + +/// Whether the match at `start..end` stands alone: no digit or letter +/// directly before or after it, so `123-45-6789` isn't found inside a +/// longer run of digits. +pub fn stands_alone(text: &str, start: usize, end: usize) -> bool { + let before = text[..start].chars().next_back(); + let after = text[end..].chars().next(); + before.is_none_or(|c| !c.is_alphanumeric()) && after.is_none_or(|c| !c.is_alphanumeric()) +} + +/// The ASCII digits of `s`. +pub fn digits(s: &str) -> String { + s.chars().filter(char::is_ascii_digit).collect() +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn words_are_whole_and_near() { + let text = "Your passport number is X1234567, thanks"; + let start = text.find("X123").unwrap(); + assert!(word_near(text, start, start + 8, &["passport"])); + assert!(!word_near(text, start, start + 8, &["pass"])); + let far = format!("passport{}X1234567", " ".repeat(60)); + let start = far.find("X123").unwrap(); + assert!(!word_near(&far, start, start + 8, &["passport"])); + } + + #[test] + fn near_counts_characters_not_bytes() { + // 45 two-byte characters between the word and the candidate: within + // 50 characters, though over 50 bytes + let text = format!("passport {} X1234567", "é".repeat(45)); + let start = text.find("X123").unwrap(); + assert!(word_near(&text, start, start + 8, &["passport"])); + } + + #[test] + fn ids_are_unique() { + let mut seen = AHashSet::new(); + for detector in all() { + assert!(seen.insert(detector.id), "duplicate id {}", detector.id); + assert!(by_id(detector.id).is_some()); + } + } +} diff --git a/crates/features/src/mailflow/extract.rs b/crates/features/src/mailflow/extract.rs new file mode 100644 index 0000000..ff233ff --- /dev/null +++ b/crates/features/src/mailflow/extract.rs @@ -0,0 +1,692 @@ +/* + * SPDX-FileCopyrightText: 2026 Coffey Labs + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +//! The text of an attachment, for the detectors (§2.3), or why there isn't +//! one. +//! +//! Read: text files (plain, CSV, JSON, XML, HTML), Office Open XML (DOCX, +//! XLSX, PPTX) and OpenDocument (ODT, ODS, ODP) documents, and ZIP archives +//! one level deep. **Can't be inspected**: encrypted or password-protected +//! files, PDF (settled answer 2), the older binary Office formats, archives +//! inside archives, and anything past the limits. Everything else (images, +//! audio, programs) has no text to read and is neither. +//! +//! Office files are ZIP archives of XML, read here with the `zip` and +//! `quick-xml` crates the server already uses: no outside converter runs. + +use quick_xml::{Reader, XmlVersion, events::Event}; +use std::io::{Cursor, Read}; + +/// How much may be unpacked from one attachment, and from how many entries. +#[derive(Debug, Clone, Copy)] +pub struct Limits { + pub max_unpacked: u64, + pub max_entries: usize, +} + +impl Default for Limits { + fn default() -> Self { + Self { + max_unpacked: 50 * 1024 * 1024, + max_entries: 10_000, + } + } +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum Extracted { + /// The text to check. + Text(String), + /// A kind of file with no text in it: nothing to check, nothing missed. + NoText, + /// A file that may hold text the detectors couldn't read. + NotInspectable(Why), +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Why { + Encrypted, + Pdf, + LegacyOffice, + NestedArchive, + TooLarge, + Damaged, +} + +impl Why { + pub fn as_str(&self) -> &'static str { + match self { + Why::Encrypted => "encrypted", + Why::Pdf => "pdf", + Why::LegacyOffice => "legacy-office", + Why::NestedArchive => "nested-archive", + Why::TooLarge => "too-large", + Why::Damaged => "damaged", + } + } +} + +const OLE_MAGIC: &[u8] = &[0xD0, 0xCF, 0x11, 0xE0, 0xA1, 0xB1, 0x1A, 0xE1]; +const ZIP_MAGIC: &[u8] = b"PK\x03\x04"; + +/// What an attachment says, from its declared type, its file name and, above +/// all, its first bytes. +pub fn extract( + content_type: &str, + file_name: Option<&str>, + data: &[u8], + limits: &Limits, +) -> Extracted { + extract_at(content_type, file_name, data, limits, 0) +} + +fn extract_at( + content_type: &str, + file_name: Option<&str>, + data: &[u8], + limits: &Limits, + depth: u8, +) -> Extracted { + let content_type = content_type.to_ascii_lowercase(); + let extension = file_name + .and_then(|name| name.rsplit_once('.')) + .map(|(_, ext)| ext.to_ascii_lowercase()) + .unwrap_or_default(); + + if data.len() as u64 > limits.max_unpacked { + return Extracted::NotInspectable(Why::TooLarge); + } + if data.starts_with(b"%PDF-") || content_type == "application/pdf" || extension == "pdf" { + return Extracted::NotInspectable(Why::Pdf); + } + if data.starts_with(OLE_MAGIC) { + // An encrypted OOXML file is an OLE container holding the encrypted + // package; any other OLE file is a legacy .doc, .xls or .ppt + return Extracted::NotInspectable(if has_utf16(data, "EncryptedPackage") { + Why::Encrypted + } else { + Why::LegacyOffice + }); + } + if data.starts_with(ZIP_MAGIC) { + if depth > 0 { + return Extracted::NotInspectable(Why::NestedArchive); + } + return zip(data, limits); + } + if is_text(&content_type, &extension) { + let text = decode_text(data); + return Extracted::Text( + if content_type == "text/html" || matches!(extension.as_str(), "html" | "htm") { + strip_html(&text) + } else { + text + }, + ); + } + Extracted::NoText +} + +fn is_text(content_type: &str, extension: &str) -> bool { + content_type.starts_with("text/") + || matches!( + content_type, + "application/json" + | "application/xml" + | "application/csv" + | "application/x-csv" + | "message/rfc822" + ) + || matches!( + extension, + "txt" + | "csv" + | "tsv" + | "json" + | "xml" + | "md" + | "log" + | "html" + | "htm" + | "eml" + | "ics" + | "vcf" + ) +} + +/// UTF-16 with a byte order mark, else UTF-8 (lossy). +fn decode_text(data: &[u8]) -> String { + let utf16 = |bytes: &[u8], big: bool| { + let units: Vec = bytes + .chunks_exact(2) + .map(|c| { + if big { + u16::from_be_bytes([c[0], c[1]]) + } else { + u16::from_le_bytes([c[0], c[1]]) + } + }) + .collect(); + String::from_utf16_lossy(&units) + }; + match data { + [0xFF, 0xFE, rest @ ..] => utf16(rest, false), + [0xFE, 0xFF, rest @ ..] => utf16(rest, true), + [0xEF, 0xBB, 0xBF, rest @ ..] => String::from_utf8_lossy(rest).into_owned(), + _ => String::from_utf8_lossy(data).into_owned(), + } +} + +fn has_utf16(data: &[u8], needle: &str) -> bool { + let needle: Vec = needle.encode_utf16().flat_map(u16::to_le_bytes).collect(); + data.windows(needle.len()).any(|w| w == needle.as_slice()) +} + +/// Tags out, the common entities decoded, block ends as new lines. +fn strip_html(html: &str) -> String { + let mut out = String::with_capacity(html.len()); + let mut in_tag = false; + let mut skip_until: Option<&str> = None; + let lower = html.to_ascii_lowercase(); + let mut i = 0; + let bytes = html.as_bytes(); + while i < bytes.len() { + if let Some(end) = skip_until { + match lower[i..].find(end) { + Some(at) => { + i += at + end.len(); + skip_until = None; + } + None => break, + } + continue; + } + let c = bytes[i]; + if in_tag { + if c == b'>' { + in_tag = false; + } + i += 1; + continue; + } + if c == b'<' { + if lower[i..].starts_with(""); + } else if lower[i..].starts_with(""); + } else { + if [ + ""), + (""", "\""), + ("'", "'"), + ("&", "&"), + ] { + out = out.replace(entity, text); + } + out +} + +/// A ZIP file: an Office document, an OpenDocument, or an archive. +fn zip(data: &[u8], limits: &Limits) -> Extracted { + let Ok(mut archive) = zip::ZipArchive::new(Cursor::new(data)) else { + return Extracted::NotInspectable(Why::Damaged); + }; + if archive.len() > limits.max_entries { + return Extracted::NotInspectable(Why::TooLarge); + } + let mut names = Vec::with_capacity(archive.len()); + let mut declared: u64 = 0; + for i in 0..archive.len() { + let Ok(entry) = archive.by_index_raw(i) else { + return Extracted::NotInspectable(Why::Damaged); + }; + if entry.encrypted() { + return Extracted::NotInspectable(Why::Encrypted); + } + declared = declared.saturating_add(entry.size()); + names.push(entry.name().to_string()); + } + if declared > limits.max_unpacked { + return Extracted::NotInspectable(Why::TooLarge); + } + let mut budget = limits.max_unpacked; + let mut read = + |archive: &mut zip::ZipArchive>, name: &str| -> Result, Why> { + let entry = archive.by_name(name).map_err(|_| Why::Damaged)?; + let mut bytes = Vec::new(); + // Declared sizes can lie: stop at the budget whatever they say + entry + .take(budget + 1) + .read_to_end(&mut bytes) + .map_err(|_| Why::Damaged)?; + if bytes.len() as u64 > budget { + return Err(Why::TooLarge); + } + budget -= bytes.len() as u64; + Ok(bytes) + }; + + let has = |name: &str| names.iter().any(|n| n == name); + let mut text = String::new(); + let result: Result<(), Why> = (|| { + if has("[Content_Types].xml") { + // Office Open XML: the parts that hold what a person wrote + let mut shared = Vec::new(); + if has("xl/sharedStrings.xml") { + shared = xml_strings(&read(&mut archive, "xl/sharedStrings.xml")?, "si"); + } + for name in names.iter().filter(|n| ooxml_text_part(n)) { + let xml = read(&mut archive, name)?; + if name.starts_with("xl/worksheets/") { + xlsx_sheet(&xml, &mut text); + } else { + xml_text(&xml, &mut text); + } + text.push('\n'); + } + text.extend(shared.iter().map(|s| format!("{s}\n"))); + } else if names.first().is_some_and(|n| n == "mimetype") + && read(&mut archive, "mimetype")?.starts_with(b"application/vnd.oasis.opendocument") + { + // OpenDocument: an encrypted one says so in its manifest + if has("META-INF/manifest.xml") + && contains( + &read(&mut archive, "META-INF/manifest.xml")?, + b"encryption-data", + ) + { + return Err(Why::Encrypted); + } + for name in ["content.xml", "styles.xml"] { + if has(name) { + xml_text(&read(&mut archive, name)?, &mut text); + text.push('\n'); + } + } + } else { + // An archive: each file inside, one level deep + for name in names.iter().filter(|n| !n.ends_with('/')) { + let bytes = read(&mut archive, name)?; + match extract_at("", Some(name), &bytes, limits, 1) { + Extracted::Text(inner) => { + text.push_str(&inner); + text.push('\n'); + } + Extracted::NoText => {} + Extracted::NotInspectable(why) => return Err(why), + } + } + } + Ok(()) + })(); + match result { + Ok(()) => Extracted::Text(text), + Err(why) => Extracted::NotInspectable(why), + } +} + +fn ooxml_text_part(name: &str) -> bool { + let xml = name.ends_with(".xml"); + xml && (name == "word/document.xml" + || [ + "word/header", + "word/footer", + "word/footnotes", + "word/endnotes", + "word/comments", + ] + .iter() + .any(|p| name.starts_with(p)) + || name.starts_with("xl/worksheets/sheet") + || name.starts_with("ppt/slides/slide") + || name.starts_with("ppt/notesSlides/")) +} + +fn contains(haystack: &[u8], needle: &[u8]) -> bool { + haystack.windows(needle.len()).any(|w| w == needle) +} + +/// The local name of a tag, without its namespace prefix. +fn local(name: &[u8]) -> &[u8] { + name.rsplit(|b| *b == b':').next().unwrap_or(name) +} + +fn push_entity(entity: &[u8], out: &mut String) { + match entity { + b"lt" => out.push('<'), + b"gt" => out.push('>'), + b"amp" => out.push('&'), + b"apos" => out.push('\''), + b"quot" => out.push('"'), + _ => { + let code = match entity { + [b'#', b'x' | b'X', hex @ ..] => std::str::from_utf8(hex) + .ok() + .and_then(|h| u32::from_str_radix(h, 16).ok()), + [b'#', dec @ ..] => std::str::from_utf8(dec).ok().and_then(|d| d.parse().ok()), + _ => None, + }; + if let Some(c) = code.and_then(char::from_u32) { + out.push(c); + } + } + } +} + +/// Every text node, runs joined as written, a new line after each paragraph +/// or row and a tab after each cell, so a number split across runs is whole +/// again. +fn xml_text(xml: &[u8], out: &mut String) { + let mut reader = Reader::from_reader(xml); + let mut buf = Vec::new(); + loop { + match reader.read_event_into(&mut buf) { + Ok(Event::Text(t)) => { + if let Ok(text) = t.xml_content(XmlVersion::Implicit1_0) { + out.push_str(&text); + } + } + Ok(Event::CData(t)) => out.push_str(&String::from_utf8_lossy(&t)), + Ok(Event::GeneralRef(entity)) => push_entity(&entity, out), + Ok(Event::End(e)) => match local(e.name().as_ref()) { + b"p" | b"h" | b"tr" | b"row" | b"table-row" | b"br" => out.push('\n'), + b"tc" | b"c" | b"table-cell" | b"tab" => out.push('\t'), + _ => {} + }, + Ok(Event::Empty(e)) => match local(e.name().as_ref()) { + b"br" | b"line-break" => out.push('\n'), + b"tab" | b"s" => out.push(' '), + _ => {} + }, + Ok(Event::Eof) | Err(_) => break, + _ => {} + } + buf.clear(); + } +} + +/// The text of each `item` element (a shared string in XLSX). +fn xml_strings(xml: &[u8], item: &str) -> Vec { + let mut reader = Reader::from_reader(xml); + let mut buf = Vec::new(); + let mut items = Vec::new(); + let mut current: Option = None; + loop { + match reader.read_event_into(&mut buf) { + Ok(Event::Start(e)) if local(e.name().as_ref()) == item.as_bytes() => { + current = Some(String::new()) + } + Ok(Event::End(e)) if local(e.name().as_ref()) == item.as_bytes() => { + items.extend(current.take()); + } + Ok(Event::Text(t)) => { + if let (Some(s), Ok(text)) = + (current.as_mut(), t.xml_content(XmlVersion::Implicit1_0)) + { + s.push_str(&text); + } + } + Ok(Event::GeneralRef(entity)) => { + if let Some(s) = current.as_mut() { + push_entity(&entity, s); + } + } + Ok(Event::Eof) | Err(_) => break, + _ => {} + } + buf.clear(); + } + items +} + +/// A worksheet's cell values: numbers and inline strings. Cells holding a +/// shared string are skipped here; the shared strings are read whole. +fn xlsx_sheet(xml: &[u8], out: &mut String) { + let mut reader = Reader::from_reader(xml); + let mut buf = Vec::new(); + let mut shared_cell = false; + let mut in_value = false; + loop { + match reader.read_event_into(&mut buf) { + Ok(Event::Start(e)) => match local(e.name().as_ref()) { + b"c" => { + shared_cell = e + .attributes() + .flatten() + .any(|a| a.key.as_ref() == b"t" && a.value.as_ref() == b"s"); + } + b"v" | b"t" => in_value = true, + _ => {} + }, + Ok(Event::End(e)) => match local(e.name().as_ref()) { + b"v" | b"t" => in_value = false, + b"c" => out.push('\t'), + b"row" => out.push('\n'), + _ => {} + }, + // A shared string's cell holds only its index: the string itself + // is added with the shared strings + Ok(Event::Text(t)) if in_value && !shared_cell => { + if let Ok(text) = t.xml_content(XmlVersion::Implicit1_0) { + out.push_str(&text); + } + } + Ok(Event::Eof) | Err(_) => break, + _ => {} + } + buf.clear(); + } +} + +#[cfg(test)] +mod tests { + use super::*; + use std::io::Write; + use zip::{ZipWriter, write::SimpleFileOptions}; + + fn zip_of(files: &[(&str, &str)]) -> Vec { + let mut zip = ZipWriter::new(Cursor::new(Vec::new())); + for (name, body) in files { + zip.start_file(*name, SimpleFileOptions::default()).unwrap(); + zip.write_all(body.as_bytes()).unwrap(); + } + zip.finish().unwrap().into_inner() + } + + fn text_of(extracted: Extracted) -> String { + match extracted { + Extracted::Text(text) => text, + other => panic!("expected text, got {other:?}"), + } + } + + #[test] + fn plain_text_and_html() { + let limits = Limits::default(); + assert_eq!( + text_of(extract("text/plain", None, b"card 4242", &limits)), + "card 4242" + ); + let utf16: Vec = [0xFF, 0xFE] + .into_iter() + .chain("héllo".encode_utf16().flat_map(u16::to_le_bytes)) + .collect(); + assert_eq!( + text_of(extract( + "application/octet-stream", + Some("a.csv"), + &utf16, + &limits + )), + "héllo" + ); + let html = "

Card 4242

ab"; + let text = text_of(extract("text/html", None, html.as_bytes(), &limits)); + assert!( + text.contains("Card 4242") && !text.contains("x()") && !text.contains("p{}"), + "{text:?}" + ); + assert_eq!( + extract("image/png", Some("a.png"), b"\x89PNG....", &limits), + Extracted::NoText + ); + } + + #[test] + fn docx_joins_split_runs() { + let doc = r#"Card 4242 4242 4242 4242A & B"#; + let docx = zip_of(&[ + ("[Content_Types].xml", ""), + ("word/document.xml", doc), + ]); + let text = text_of(extract( + "application/vnd.openxmlformats-officedocument.wordprocessingml.document", + Some("a.docx"), + &docx, + &Limits::default(), + )); + assert!(text.contains("Card 4242 4242 4242 4242\nA & B"), "{text:?}"); + } + + #[test] + fn xlsx_numbers_and_shared_strings() { + let sheet = r#"04242424242424242"#; + let shared = r#"IBAN GB29 NWBK 6016 1331 9268 19"#; + let xlsx = zip_of(&[ + ("[Content_Types].xml", ""), + ("xl/sharedStrings.xml", shared), + ("xl/worksheets/sheet1.xml", sheet), + ]); + let text = text_of(extract("", Some("book.xlsx"), &xlsx, &Limits::default())); + assert!( + text.contains("4242424242424242") && text.contains("GB29 NWBK 6016 1331 9268 19"), + "{text:?}" + ); + // The shared string's index isn't read as a value + assert!( + !text.contains("\t0\t") && !text.starts_with('0'), + "{text:?}" + ); + } + + #[test] + fn opendocument_and_encrypted_opendocument() { + let content = r#"SSN 078-05-1120"#; + let odt = zip_of(&[ + ("mimetype", "application/vnd.oasis.opendocument.text"), + ("content.xml", content), + ]); + assert!( + text_of(extract("", Some("a.odt"), &odt, &Limits::default())) + .contains("SSN 078-05-1120") + ); + let manifest = r#""#; + let locked = zip_of(&[ + ("mimetype", "application/vnd.oasis.opendocument.text"), + ("META-INF/manifest.xml", manifest), + ("content.xml", "x"), + ]); + assert_eq!( + extract("", Some("a.odt"), &locked, &Limits::default()), + Extracted::NotInspectable(Why::Encrypted) + ); + } + + #[test] + fn archives() { + let limits = Limits::default(); + let archive = zip_of(&[ + ("notes/a.txt", "card 4242424242424242"), + ("b.png", "\u{89}PNG"), + ]); + assert!( + text_of(extract("application/zip", Some("x.zip"), &archive, &limits)) + .contains("4242424242424242") + ); + let nested = zip_of(&[( + "inner.zip", + std::str::from_utf8(&[b'P', b'K', 3, 4]).unwrap(), + )]); + assert_eq!( + extract("application/zip", Some("x.zip"), &nested, &limits), + Extracted::NotInspectable(Why::NestedArchive) + ); + + // Password-protected + let mut zip = ZipWriter::new(Cursor::new(Vec::new())); + zip.start_file( + "secret.txt", + SimpleFileOptions::default().with_aes_encryption(zip::AesMode::Aes256, "pw"), + ) + .unwrap(); + zip.write_all(b"4242424242424242").unwrap(); + let locked = zip.finish().unwrap().into_inner(); + assert_eq!( + extract("application/zip", Some("x.zip"), &locked, &limits), + Extracted::NotInspectable(Why::Encrypted) + ); + + // Past the limits + let small = Limits { + max_unpacked: 10, + max_entries: 1, + }; + assert_eq!( + extract("application/zip", Some("x.zip"), &archive, &small), + Extracted::NotInspectable(Why::TooLarge) + ); + assert_eq!( + extract("application/zip", None, b"PK\x03\x04garbage", &limits), + Extracted::NotInspectable(Why::Damaged) + ); + } + + #[test] + fn not_inspectable_kinds() { + let limits = Limits::default(); + assert_eq!( + extract("application/octet-stream", None, b"%PDF-1.7 ...", &limits), + Extracted::NotInspectable(Why::Pdf) + ); + let mut ole = OLE_MAGIC.to_vec(); + ole.extend(std::iter::repeat_n(0, 64)); + assert_eq!( + extract("", Some("old.doc"), &ole, &limits), + Extracted::NotInspectable(Why::LegacyOffice) + ); + ole.extend("EncryptedPackage".encode_utf16().flat_map(u16::to_le_bytes)); + assert_eq!( + extract("", Some("new.docx"), &ole, &limits), + Extracted::NotInspectable(Why::Encrypted) + ); + } +} diff --git a/crates/features/src/mailflow/mod.rs b/crates/features/src/mailflow/mod.rs new file mode 100644 index 0000000..d218aac --- /dev/null +++ b/crates/features/src/mailflow/mod.rs @@ -0,0 +1,23 @@ +/* + * SPDX-FileCopyrightText: 2026 Coffey Labs + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +//! Data loss prevention and mail flow rules (dlp-and-mail-flow-rules spec). +//! +//! Pure functions over text and attachment bytes, so everything here is +//! unit-tested without a server: +//! +//! - [`detectors`]: find identifiers in text (payment cards, IBANs, +//! national ID numbers, keys), each by its published format and check +//! (§2.3); +//! - [`words`]: an organization's own word lists and patterns; +//! - [`extract`]: the text of an attachment, or why it can't be read. +//! +//! Nothing here writes what it finds anywhere: callers get counts, and the +//! matched text never leaves the evaluation (§2.7). + +pub mod detectors; +pub mod extract; +pub mod words; diff --git a/crates/features/src/mailflow/words.rs b/crates/features/src/mailflow/words.rs new file mode 100644 index 0000000..e171ff2 --- /dev/null +++ b/crates/features/src/mailflow/words.rs @@ -0,0 +1,109 @@ +/* + * SPDX-FileCopyrightText: 2026 Coffey Labs + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +//! An organization's own word lists and patterns (§2.3). Both count +//! occurrences, not distinct values: "confidential" three times is three. + +use aho_corasick::{AhoCorasick, AhoCorasickBuilder, MatchKind}; +use regex::{Regex, RegexBuilder}; + +/// How large a compiled pattern may grow. Keeps a rule someone writes from +/// making every message slow to send. +const PATTERN_SIZE_LIMIT: usize = 1 << 20; + +/// Words and phrases, matched whole and ignoring case. +#[derive(Debug, Clone)] +pub struct WordList { + matcher: AhoCorasick, +} + +impl WordList { + /// Builds a list from words or phrases; empty entries are skipped. + pub fn new(words: I) -> Result + where + I: IntoIterator, + S: AsRef, + { + let words: Vec = words + .into_iter() + .map(|w| w.as_ref().trim().to_lowercase()) + .filter(|w| !w.is_empty()) + .collect(); + if words.is_empty() { + return Err("The list has no words".into()); + } + AhoCorasickBuilder::new() + .match_kind(MatchKind::LeftmostLongest) + .build(&words) + .map(|matcher| Self { matcher }) + .map_err(|err| err.to_string()) + } + + /// How many times any word of the list appears in `text`. + pub fn count(&self, text: &str) -> usize { + let text = text.to_lowercase(); + self.matcher + .find_iter(&text) + .filter(|m| super::detectors::stands_alone(&text, m.start(), m.end())) + .count() + } +} + +/// An organization's regular expression. +#[derive(Debug, Clone)] +pub struct Pattern { + regex: Regex, +} + +impl Pattern { + /// Compiles `pattern`, or says why it can't be used. Matching ignores + /// case unless the pattern turns that off with `(?-i)`. + pub fn new(pattern: &str) -> Result { + RegexBuilder::new(pattern) + .case_insensitive(true) + .size_limit(PATTERN_SIZE_LIMIT) + .build() + .map(|regex| Self { regex }) + .map_err(|err| err.to_string()) + } + + /// How many times the pattern matches in `text`. + pub fn count(&self, text: &str) -> usize { + self.regex.find_iter(text).filter(|m| !m.is_empty()).count() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn words_whole_and_any_case() { + let list = WordList::new(["Project Falcon", "confidential", " "]).unwrap(); + assert_eq!( + list.count( + "CONFIDENTIAL: project falcon notes. Not confidentiality, not projectfalcon." + ), + 2 + ); + assert_eq!(list.count("Confidential, confidential and confidential"), 3); + // Non-ASCII case folding + let list = WordList::new(["GEHEIM", "Straße"]).unwrap(); + assert_eq!(list.count("streng geheim, STRASSE ist nicht Straße"), 2); + assert!(WordList::new(["", " "]).is_err()); + } + + #[test] + fn patterns() { + let pattern = Pattern::new(r"\bPRJ-\d{4}\b").unwrap(); + assert_eq!(pattern.count("prj-1234 and PRJ-5678, not PRJ-12"), 2); + assert!(Pattern::new("(unclosed").is_err()); + // Too large to compile within the limit + assert!(Pattern::new(r"\w{1000}\w{1000}\w{1000}").is_err()); + // Empty matches don't count + assert_eq!(Pattern::new("x*").unwrap().count("abc"), 0); + } +}