From dc49bf4d1428834a9f21bbb8998f8a95346a2c42 Mon Sep 17 00:00:00 2001 From: John Coffey Date: Mon, 28 Sep 2026 17:00:49 -0700 Subject: [PATCH 1/2] DLP: the detector framework, the region-free detectors, word lists and attachment text Phase 2a of the DLP and mail flow rules spec: pure functions in crates/features/src/mailflow, nothing wired into the mail path yet. - Detectors report distinct values found, each either checked by its published check digit or counted only beside a corroborating word within 50 characters. This PR adds the region-free ones: payment cards (issuer prefixes, Luhn), IBAN (registry lengths, mod 97), SWIFT/BIC, email addresses and phone numbers in bulk, dates of birth, passport numbers, private keys and published service-token formats. Regional identifiers follow, a region per PR. - Word lists (Aho-Corasick, whole words, any case) and patterns (regex with a compiled-size limit) count occurrences. - Attachment text: text files with or without a UTF-16 mark, HTML, DOCX/XLSX/PPTX, ODT/ODS/ODP and ZIP archives one level deep, read with the zip and quick-xml crates the workspace already has. Encrypted files, PDF, legacy binary Office files, nested archives and anything past the limits come back as not inspectable, with why. 21 unit tests, against the networks' test card numbers and the IBAN registry's own examples among others. --- crates/features/Cargo.toml | 5 + crates/features/src/lib.rs | 1 + crates/features/src/mailflow/detectors/any.rs | 522 +++++++++++++ .../features/src/mailflow/detectors/checks.rs | 227 ++++++ crates/features/src/mailflow/detectors/mod.rs | 199 +++++ crates/features/src/mailflow/extract.rs | 692 ++++++++++++++++++ crates/features/src/mailflow/mod.rs | 23 + crates/features/src/mailflow/words.rs | 109 +++ 8 files changed, 1778 insertions(+) create mode 100644 crates/features/src/mailflow/detectors/any.rs create mode 100644 crates/features/src/mailflow/detectors/checks.rs create mode 100644 crates/features/src/mailflow/detectors/mod.rs create mode 100644 crates/features/src/mailflow/extract.rs create mode 100644 crates/features/src/mailflow/mod.rs create mode 100644 crates/features/src/mailflow/words.rs diff --git a/crates/features/Cargo.toml b/crates/features/Cargo.toml index cbee6d6..3054682 100644 --- a/crates/features/Cargo.toml +++ b/crates/features/Cargo.toml @@ -21,6 +21,11 @@ base64 = "0.23" sha2 = "0.11" flate2 = "1.1" tokio = { version = "1.53", features = ["sync", "rt"] } +# inbuxa: DLP detectors and attachment text (dlp-and-mail-flow-rules spec) +regex = "1.13.1" +aho-corasick = "1.1" +zip = "8.6" +quick-xml = "0.41" [dev-dependencies] tokio = { version = "1.53", features = ["macros", "rt"] } diff --git a/crates/features/src/lib.rs b/crates/features/src/lib.rs index 9145a72..9df288c 100644 --- a/crates/features/src/lib.rs +++ b/crates/features/src/lib.rs @@ -23,6 +23,7 @@ pub mod audit; pub mod branding; pub mod hold; pub mod lock; +pub mod mailflow; pub mod masked_email; pub mod privacy; pub mod security; diff --git a/crates/features/src/mailflow/detectors/any.rs b/crates/features/src/mailflow/detectors/any.rs new file mode 100644 index 0000000..1120cb7 --- /dev/null +++ b/crates/features/src/mailflow/detectors/any.rs @@ -0,0 +1,522 @@ +/* + * SPDX-FileCopyrightText: 2026 Coffey Labs + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +//! Detectors that aren't tied to one country (§2.3, region "Any"). + +use super::{Detector, Findings, Region, Strength, checks, digits, stands_alone, word_near}; +use regex::Regex; +use std::sync::LazyLock; + +pub static DETECTORS: &[Detector] = &[ + Detector::new( + "payment-card", + "Payment card number", + Region::Any, + Strength::Checked, + payment_card, + ), + Detector::new("iban", "IBAN", Region::Any, Strength::Checked, iban), + Detector::new( + "swift-bic", + "SWIFT/BIC code", + Region::Any, + Strength::NeedsWord, + swift_bic, + ), + Detector::new( + "email-addresses", + "Email addresses", + Region::Any, + Strength::Checked, + email_addresses, + ), + Detector::new( + "phone-numbers", + "Phone numbers", + Region::Any, + Strength::NeedsWord, + phone_numbers, + ), + Detector::new( + "date-of-birth", + "Date of birth", + Region::Any, + Strength::NeedsWord, + date_of_birth, + ), + Detector::new( + "passport", + "Passport number", + Region::Any, + Strength::NeedsWord, + passport, + ), + Detector::new( + "private-key", + "Private key", + Region::Any, + Strength::Checked, + private_key, + ), + Detector::new( + "credentials", + "Cloud and service credentials", + Region::Any, + Strength::Checked, + credentials, + ), +]; + +fn re(pattern: &str) -> Regex { + Regex::new(pattern).expect("detector pattern") +} + +// --- Payment cards -------------------------------------------------------- + +/// Issuer prefixes (ISO/IEC 7812 IINs) and the lengths each network issues. +fn card_network(number: &str) -> bool { + let len = number.len(); + let prefix = |n: usize| number[..n].parse::().unwrap_or(0); + match number.as_bytes()[0] { + // Visa + b'4' => matches!(len, 13 | 16 | 19), + b'5' => { + // Mastercard 51–55; Maestro 50, 56–58 + (51..=55).contains(&prefix(2)) && len == 16 + || matches!(prefix(2), 50 | 56..=58) && (12..=19).contains(&len) + } + // Mastercard 2221–2720 + b'2' => (2221..=2720).contains(&prefix(4)) && len == 16, + b'3' => { + // American Express 34, 37; JCB 3528–3589; Diners 300–305, 36, 38, 39 + matches!(prefix(2), 34 | 37) && len == 15 + || (3528..=3589).contains(&prefix(4)) && (16..=19).contains(&len) + || ((300..=305).contains(&prefix(3)) || matches!(prefix(2), 36 | 38 | 39)) + && (14..=19).contains(&len) + } + // Discover 6011, 644–649, 65; UnionPay 62; Maestro 6x + b'6' => (12..=19).contains(&len), + _ => false, + } +} + +fn is_card(number: &str) -> bool { + (12..=19).contains(&number.len()) && card_network(number) && checks::luhn(number) +} + +static CARD: LazyLock = LazyLock::new(|| re(r"\b\d(?:[ -]?\d){11,18}\b")); + +fn payment_card(text: &str, findings: &mut Findings) { + for m in CARD.find_iter(text) { + if !stands_alone(text, m.start(), m.end()) { + continue; + } + let whole = digits(m.as_str()); + if is_card(&whole) { + findings.insert(whole); + continue; + } + // Two numbers side by side ("4242 4242 4242 4242 2031"): try each + // run of whole groups + let groups: Vec = m.as_str().split([' ', '-']).map(digits).collect(); + 'runs: for from in 0..groups.len() { + let mut number = String::new(); + for group in &groups[from..] { + number.push_str(group); + if is_card(&number) { + findings.insert(number); + break 'runs; + } + } + } + } +} + +// --- IBAN ----------------------------------------------------------------- + +static IBAN: LazyLock = + LazyLock::new(|| re(r"\b[A-Za-z]{2}\d{2}(?:[ ]?[A-Za-z0-9]){11,30}")); + +fn iban(text: &str, findings: &mut Findings) { + // The pattern can run on into the next words, even the next IBAN: after + // each hit, look again from where that IBAN ended + let mut from = 0; + while let Some(m) = IBAN.find_at(text, from) { + from = m.start() + 1; + let compact = m.as_str().replace(' ', "").to_ascii_uppercase(); + let Some(len) = checks::iban_length(&compact[..2]) else { + continue; + }; + if compact.len() < len { + continue; + } + // Where the country's length ends in the text, spaces counted + let mut seen = 0; + let Some(end) = m + .as_str() + .char_indices() + .find(|(_, c)| { + if *c != ' ' { + seen += 1; + } + seen == len + }) + .map(|(i, c)| m.start() + i + c.len_utf8()) + else { + continue; + }; + let candidate = &compact[..len]; + if stands_alone(text, m.start(), end) && checks::iban(candidate) { + findings.insert(candidate); + from = end; + } + } +} + +// --- SWIFT/BIC ------------------------------------------------------------ + +static BIC: LazyLock = + LazyLock::new(|| re(r"\b[A-Z]{4}[A-Z]{2}[A-Z0-9]{2}(?:[A-Z0-9]{3})?\b")); + +const BIC_WORDS: &[&str] = &[ + "swift", + "bic", + "swift/bic", + "bank", + "banque", + "bankverbindung", +]; + +fn swift_bic(text: &str, findings: &mut Findings) { + for m in BIC.find_iter(text) { + let code = m.as_str(); + if checks::is_country(&code[4..6]) && word_near(text, m.start(), m.end(), BIC_WORDS) { + findings.insert(code); + } + } +} + +// --- Contact lists -------------------------------------------------------- + +static EMAIL: LazyLock = + LazyLock::new(|| re(r"(?i)\b[a-z0-9._%+-]+@[a-z0-9-]+(?:\.[a-z0-9-]+)*\.[a-z]{2,}\b")); + +fn email_addresses(text: &str, findings: &mut Findings) { + for m in EMAIL.find_iter(text) { + findings.insert(m.as_str().to_lowercase()); + } +} + +/// International form: found alone. National form: only with a word. +static PHONE_INTL: LazyLock = LazyLock::new(|| re(r"\+\d{1,3}(?:[ .-]?\(?\d{1,4}\)?){2,5}")); +static PHONE_NATIONAL: LazyLock = + LazyLock::new(|| re(r"\(?\d{2,4}\)?[ .-]\d{3,4}[ .-]\d{3,4}")); + +const PHONE_WORDS: &[&str] = &[ + "phone", + "tel", + "telephone", + "mobile", + "cell", + "fax", + "telefon", + "téléphone", + "teléfono", + "telefono", + "handy", + "portable", + "móvil", + "cellulare", + "mobiel", +]; + +fn phone_numbers(text: &str, findings: &mut Findings) { + let mut international = Vec::new(); + for m in PHONE_INTL.find_iter(text) { + let number = digits(m.as_str()); + if (8..=15).contains(&number.len()) && stands_alone(text, m.start() + 1, m.end()) { + findings.insert(number); + international.push(m.range()); + } + } + for m in PHONE_NATIONAL.find_iter(text) { + let number = digits(m.as_str()); + // Not the tail of an international number already counted + if international.iter().any(|r| r.contains(&m.start())) { + continue; + } + if (9..=11).contains(&number.len()) + && stands_alone(text, m.start(), m.end()) + && !text[..m.start()].ends_with('+') + && word_near(text, m.start(), m.end(), PHONE_WORDS) + { + findings.insert(number); + } + } +} + +// --- Date of birth -------------------------------------------------------- + +static DATE_ISO: LazyLock = LazyLock::new(|| re(r"\b(\d{4})-(\d{2})-(\d{2})\b")); +static DATE_NUMERIC: LazyLock = + LazyLock::new(|| re(r"\b(\d{1,2})[./-](\d{1,2})[./-](\d{4})\b")); +static DATE_WORDS: LazyLock = LazyLock::new(|| { + re( + r"(?i)\b(?:(\d{1,2})\s+(jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)[a-z]*\.?,?\s+(\d{4})|(jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)[a-z]*\.?\s+(\d{1,2}),?\s+(\d{4}))\b", + ) +}); + +const BIRTH_WORDS: &[&str] = &[ + "born", + "birth", + "dob", + "d.o.b", + "birthday", + "birthdate", + "geburtsdatum", + "geboren", + "naissance", + "né le", + "née le", + "nacimiento", + "nacido", + "nacida", + "nascita", + "nato il", + "nata il", + "geboortedatum", + "födelsedatum", + "fødselsdato", + "syntymäaika", + "urodzenia", + "nascimento", +]; + +fn valid_date(year: u32, month: u32, day: u32) -> bool { + let days = match month { + 1 | 3 | 5 | 7 | 8 | 10 | 12 => 31, + 4 | 6 | 9 | 11 => 30, + 2 if year.is_multiple_of(4) && (!year.is_multiple_of(100) || year.is_multiple_of(400)) => { + 29 + } + 2 => 28, + _ => return false, + }; + (1900..=2100).contains(&year) && (1..=days).contains(&day) +} + +fn month_number(name: &str) -> u32 { + const MONTHS: [&str; 12] = [ + "jan", "feb", "mar", "apr", "may", "jun", "jul", "aug", "sep", "oct", "nov", "dec", + ]; + let name = name.to_lowercase(); + MONTHS + .iter() + .position(|m| *m == name) + .map_or(0, |i| i as u32 + 1) +} + +fn date_of_birth(text: &str, findings: &mut Findings) { + let mut add = |start: usize, end: usize, key: String| { + if word_near(text, start, end, BIRTH_WORDS) { + findings.insert(key); + } + }; + let num = |s: &str| s.parse::().unwrap_or(0); + for c in DATE_ISO.captures_iter(text) { + let (y, m, d) = (num(&c[1]), num(&c[2]), num(&c[3])); + let whole = c.get(0).unwrap(); + if valid_date(y, m, d) { + add(whole.start(), whole.end(), format!("{y:04}{m:02}{d:02}")); + } + } + for c in DATE_NUMERIC.captures_iter(text) { + let (a, b, y) = (num(&c[1]), num(&c[2]), num(&c[3])); + let whole = c.get(0).unwrap(); + // Day first or month first: either reading that is a real date + if valid_date(y, b, a) || valid_date(y, a, b) { + add(whole.start(), whole.end(), whole.as_str().to_string()); + } + } + for c in DATE_WORDS.captures_iter(text) { + let whole = c.get(0).unwrap(); + let (d, m, y) = match (c.get(1), c.get(4)) { + (Some(d), _) => (num(d.as_str()), month_number(&c[2]), num(&c[3])), + (_, Some(m)) => (num(&c[5]), month_number(m.as_str()), num(&c[6])), + _ => continue, + }; + if valid_date(y, m, d) { + add(whole.start(), whole.end(), format!("{y:04}{m:02}{d:02}")); + } + } +} + +// --- Passport ------------------------------------------------------------- + +static PASSPORT: LazyLock = LazyLock::new(|| re(r"\b[A-Z0-9]{6,9}\b")); + +const PASSPORT_WORDS: &[&str] = &[ + "passport", + "passeport", + "reisepass", + "pasaporte", + "passaporto", + "paspoort", + "passnummer", + "pass-nr", + "passport no", + "pasaporte n.º", + "passaporte", +]; + +fn passport(text: &str, findings: &mut Findings) { + for m in PASSPORT.find_iter(text) { + let value = m.as_str(); + if value.bytes().filter(u8::is_ascii_digit).count() >= 5 + && word_near(text, m.start(), m.end(), PASSPORT_WORDS) + { + findings.insert(value); + } + } +} + +// --- Keys and credentials ------------------------------------------------- + +static PRIVATE_KEY: LazyLock = LazyLock::new(|| { + re( + r"-----BEGIN (?:(?:RSA|EC|DSA|OPENSSH|ENCRYPTED|PGP) )?PRIVATE KEY(?: BLOCK)?-----\s*([A-Za-z0-9+/=:\s-]{0,64})", + ) +}); + +fn private_key(text: &str, findings: &mut Findings) { + for c in PRIVATE_KEY.captures_iter(text) { + // Each key once, by the start of its body + let body: String = c[1].chars().filter(|c| !c.is_whitespace()).collect(); + let whole = c.get(0).unwrap(); + findings.insert(if body.is_empty() { + format!("@{}", whole.start()) + } else { + body + }); + } +} + +/// Published token formats: AWS access key IDs, GitHub tokens, Slack +/// tokens, Stripe live secret and restricted keys, Google API keys. +static CREDENTIAL: LazyLock = LazyLock::new(|| { + re(concat!( + r"\b(?:", + r"(?:AKIA|ASIA|ABIA|ACCA)[A-Z0-9]{16}", + r"|gh[pousr]_[A-Za-z0-9]{36}", + r"|github_pat_[A-Za-z0-9_]{82}", + r"|xox[abposr]-[A-Za-z0-9-]{10,72}", + r"|(?:sk|rk)_live_[A-Za-z0-9]{24,99}", + r"|AIza[0-9A-Za-z_-]{35}", + r")\b" + )) +}); + +fn credentials(text: &str, findings: &mut Findings) { + for m in CREDENTIAL.find_iter(text) { + findings.insert(m.as_str()); + } +} + +#[cfg(test)] +mod tests { + use crate::mailflow::detectors::by_id; + + fn count(id: &str, text: &str) -> usize { + by_id(id).unwrap().count(text) + } + + #[test] + fn payment_cards() { + // Networks' and processors' published test numbers + let text = "Visa 4242 4242 4242 4242, MC 5555-5555-5555-4444, Amex 378282246310005, \ + Discover 6011111111111117, JCB 3566002020360505, Diners 30569309025904, \ + UnionPay 6200000000000005, Mastercard 2-series 2223003122003222"; + assert_eq!(count("payment-card", text), 8); + // Luhn fails, wrong network length, inside a longer number + assert_eq!(count("payment-card", "4242424242424241"), 0); + assert_eq!(count("payment-card", "378282246310005 0"), 1); + assert_eq!(count("payment-card", "order 94242424242424242 shipped"), 0); + // The same number twice counts once + assert_eq!( + count("payment-card", "4242424242424242 and 4242-4242-4242-4242"), + 1 + ); + // A card followed by a year + assert_eq!(count("payment-card", "card 4242 4242 4242 4242 2031"), 1); + } + + #[test] + fn ibans() { + let text = + "Pay GB29 NWBK 6016 1331 9268 19 or de89370400440532013000 (NL91ABNA0417164300)."; + assert_eq!(count("iban", text), 3); + assert_eq!(count("iban", "GB29 NWBK 6016 1331 9268 18"), 0); + // Runs into the next word: still found at the country's length + assert_eq!(count("iban", "IBAN NL91ABNA0417164300 BIC ABNANL2A"), 1); + } + + #[test] + fn swift_codes_need_a_word() { + assert_eq!(count("swift-bic", "SWIFT: DEUTDEFF500"), 1); + assert_eq!(count("swift-bic", "BIC NWBKGB2L"), 1); + assert_eq!(count("swift-bic", "HAPPYDAYS DEUTDEFF"), 0); + // Not a country in positions 5–6 + assert_eq!(count("swift-bic", "BIC DEUTZZFF"), 0); + } + + #[test] + fn email_and_phone_lists() { + let list = "a@example.com, B@Example.com, c.d+x@mail.example.org, a@example.com"; + assert_eq!(count("email-addresses", list), 3); + assert_eq!( + count("phone-numbers", "+44 20 7946 0958, +1 (415) 555-2671"), + 2 + ); + assert_eq!(count("phone-numbers", "call 020 7946 0958"), 0); + assert_eq!(count("phone-numbers", "Tel: 020 7946 0958"), 1); + assert_eq!(count("phone-numbers", "invoice 020 7946 0958"), 0); + // One number, not also its national tail + assert_eq!(count("phone-numbers", "Tel: +44 20 7946 0958"), 1); + } + + #[test] + fn dates_of_birth() { + assert_eq!(count("date-of-birth", "DOB: 1984-02-29"), 1); + assert_eq!(count("date-of-birth", "Geburtsdatum 31.12.1970"), 1); + assert_eq!(count("date-of-birth", "born on March 3, 1962"), 1); + assert_eq!(count("date-of-birth", "date of birth 3 Mar 1962"), 1); + // Not a real date, no word, a meeting + assert_eq!(count("date-of-birth", "DOB: 1985-02-29"), 0); + assert_eq!(count("date-of-birth", "invoice 1984-02-29"), 0); + assert_eq!(count("date-of-birth", "Meeting on 12/05/2026"), 0); + } + + #[test] + fn passports_need_a_word() { + assert_eq!(count("passport", "Passport number: 533380006"), 1); + assert_eq!(count("passport", "Reisepass C01X00T47"), 1); + assert_eq!(count("passport", "Order 533380006 shipped"), 0); + // Mostly letters: a word, not a number + assert_eq!(count("passport", "passport PASSWORD"), 0); + } + + #[test] + fn keys_and_credentials() { + let key = "-----BEGIN OPENSSH PRIVATE KEY-----\nb3BlbnNzaC1rZXktdjEAAAAABG5vbmUAAAAEbm9uZQ\n-----END OPENSSH PRIVATE KEY-----"; + assert_eq!(count("private-key", key), 1); + assert_eq!(count("private-key", "-----BEGIN PUBLIC KEY-----\nMFkw"), 0); + // Documentation examples of each format + let tokens = "AKIAIOSFODNN7EXAMPLE ghp_0123456789abcdefghijklmnopqrstuvwxyz \ + AIzaSyA-0123456789abcdefghijklmnopqrstu"; + assert_eq!(count("credentials", tokens), 3); + assert_eq!(count("credentials", "AKIA123 ghp_short"), 0); + } +} diff --git a/crates/features/src/mailflow/detectors/checks.rs b/crates/features/src/mailflow/detectors/checks.rs new file mode 100644 index 0000000..d0320be --- /dev/null +++ b/crates/features/src/mailflow/detectors/checks.rs @@ -0,0 +1,227 @@ +/* + * SPDX-FileCopyrightText: 2026 Coffey Labs + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +//! Check-digit algorithms, each from its public definition. + +/// The Luhn check (ISO/IEC 7812-1, Annex B) over a string of ASCII digits. +pub fn luhn(digits: &str) -> bool { + if digits.len() < 2 || !digits.bytes().all(|b| b.is_ascii_digit()) { + return false; + } + let sum: u32 = digits + .bytes() + .rev() + .enumerate() + .map(|(i, b)| { + let d = u32::from(b - b'0'); + if i % 2 == 1 { + let d = d * 2; + if d > 9 { d - 9 } else { d } + } else { + d + } + }) + .sum(); + sum % 10 == 0 +} + +/// ISO 13616 IBAN lengths, by country, from the IBAN registry. +const IBAN_LENGTHS: &[(&str, usize)] = &[ + ("AD", 24), + ("AE", 23), + ("AL", 28), + ("AT", 20), + ("AZ", 28), + ("BA", 20), + ("BE", 16), + ("BG", 22), + ("BH", 22), + ("BI", 27), + ("BR", 29), + ("BY", 28), + ("CH", 21), + ("CR", 22), + ("CY", 28), + ("CZ", 24), + ("DE", 22), + ("DJ", 27), + ("DK", 18), + ("DO", 28), + ("EE", 20), + ("EG", 29), + ("ES", 24), + ("FI", 18), + ("FK", 18), + ("FO", 18), + ("FR", 27), + ("GB", 22), + ("GE", 22), + ("GI", 23), + ("GL", 18), + ("GR", 27), + ("GT", 28), + ("HN", 28), + ("HR", 21), + ("HU", 28), + ("IE", 22), + ("IL", 23), + ("IQ", 23), + ("IS", 26), + ("IT", 27), + ("JO", 30), + ("KW", 30), + ("KZ", 20), + ("LB", 28), + ("LC", 32), + ("LI", 21), + ("LT", 20), + ("LU", 20), + ("LV", 21), + ("LY", 25), + ("MC", 27), + ("MD", 24), + ("ME", 22), + ("MK", 19), + ("MN", 20), + ("MR", 27), + ("MT", 31), + ("MU", 30), + ("NI", 28), + ("NL", 18), + ("NO", 15), + ("OM", 23), + ("PK", 24), + ("PL", 28), + ("PS", 29), + ("PT", 25), + ("QA", 29), + ("RO", 24), + ("RS", 22), + ("RU", 33), + ("SA", 24), + ("SC", 31), + ("SD", 18), + ("SE", 24), + ("SI", 19), + ("SK", 24), + ("SM", 27), + ("SO", 23), + ("ST", 25), + ("SV", 28), + ("TL", 23), + ("TN", 24), + ("TR", 26), + ("UA", 29), + ("VA", 22), + ("VG", 24), + ("XK", 20), + ("YE", 30), +]; + +/// The IBAN length for a country code, if the country uses IBANs. +pub fn iban_length(country: &str) -> Option { + IBAN_LENGTHS + .iter() + .find(|(code, _)| *code == country) + .map(|(_, len)| *len) +} + +/// ISO 13616 / ISO 7064 MOD 97-10 over an IBAN with no spaces, upper case: +/// move the first four characters to the end, turn letters into 10–35, and +/// the number mod 97 must be 1. Also checks the country's length. +pub fn iban(iban: &str) -> bool { + if iban.len() < 5 + || !iban + .bytes() + .all(|b| b.is_ascii_uppercase() || b.is_ascii_digit()) + { + return false; + } + if iban_length(&iban[..2]) != Some(iban.len()) + || !iban[2..4].bytes().all(|b| b.is_ascii_digit()) + { + return false; + } + let mut remainder: u32 = 0; + for b in iban[4..].bytes().chain(iban[..4].bytes()) { + let value = if b.is_ascii_digit() { + u32::from(b - b'0') + } else { + u32::from(b - b'A') + 10 + }; + remainder = if value >= 10 { + (remainder * 100 + value) % 97 + } else { + (remainder * 10 + value) % 97 + }; + } + remainder == 1 +} + +/// ISO 3166-1 alpha-2 country codes, for SWIFT/BIC positions 5–6. +const COUNTRIES: &str = "AD AE AF AG AI AL AM AO AQ AR AS AT AU AW AX AZ BA BB BD BE BF BG BH BI BJ \ +BL BM BN BO BQ BR BS BT BV BW BY BZ CA CC CD CF CG CH CI CK CL CM CN CO CR CU CV CW CX CY CZ DE DJ \ +DK DM DO DZ EC EE EG EH ER ES ET FI FJ FK FM FO FR GA GB GD GE GF GG GH GI GL GM GN GP GQ GR GS GT \ +GU GW GY HK HM HN HR HT HU ID IE IL IM IN IO IQ IR IS IT JE JM JO JP KE KG KH KI KM KN KP KR KW KY \ +KZ LA LB LC LI LK LR LS LT LU LV LY MA MC MD ME MF MG MH MK ML MM MN MO MP MQ MR MS MT MU MV MW MX \ +MY MZ NA NC NE NF NG NI NL NO NP NR NU NZ OM PA PE PF PG PH PK PL PM PN PR PS PT PW PY QA RE RO RS \ +RU RW SA SB SC SD SE SG SH SI SJ SK SL SM SN SO SR SS ST SV SX SY SZ TC TD TF TG TH TJ TK TL TM TN \ +TO TR TT TV TW TZ UA UG UM US UY UZ VA VC VE VG VI VN VU WF WS XK YE YT ZA ZM ZW"; + +pub fn is_country(code: &str) -> bool { + code.len() == 2 && COUNTRIES.split(' ').any(|c| c == code) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn luhn_known_numbers() { + // Published test card numbers + for good in [ + "4242424242424242", + "5555555555554444", + "378282246310005", + "79927398713", + ] { + assert!(luhn(good), "{good}"); + } + for bad in ["4242424242424241", "79927398710", "1", "12a4"] { + assert!(!luhn(bad), "{bad}"); + } + } + + #[test] + fn iban_registry_examples() { + // The IBAN registry's own examples + for good in [ + "GB29NWBK60161331926819", + "DE89370400440532013000", + "FR1420041010050500013M02606", + "NL91ABNA0417164300", + "BE68539007547034", + "NO9386011117947", + "CH9300762011623852957", + ] { + assert!(iban(good), "{good}"); + } + for bad in [ + "GB29NWBK60161331926818", // check fails + "GB29NWBK6016133192681", // too short for GB + "ZZ29NWBK60161331926819", // no such country + "DE8937040044053201300A", // letters where DE has none still fail mod 97 + ] { + assert!(!iban(bad), "{bad}"); + } + } + + #[test] + fn countries() { + assert!(is_country("DE") && is_country("US") && is_country("XK")); + assert!(!is_country("ZZ") && !is_country("D")); + } +} diff --git a/crates/features/src/mailflow/detectors/mod.rs b/crates/features/src/mailflow/detectors/mod.rs new file mode 100644 index 0000000..66b1eab --- /dev/null +++ b/crates/features/src/mailflow/detectors/mod.rs @@ -0,0 +1,199 @@ +/* + * SPDX-FileCopyrightText: 2026 Coffey Labs + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +//! Detectors (dlp-and-mail-flow-rules spec, §2.3): each finds one kind of +//! identifier in text and reports the distinct ones it found. +//! +//! A detector is one of two strengths: +//! +//! - **Checked**: the identifier carries a published check digit or +//! checksum, so a random number rarely passes; found on its own. +//! - **Needs a word**: the format alone is too common, so a candidate counts +//! only with a corroborating word within [`WINDOW`] characters either +//! side. +//! +//! Findings are distinct normalized values (digits only, upper case), so the +//! same card number pasted twice counts once. They stay in memory: callers +//! read only [`Findings::len`]. + +pub mod any; +pub mod checks; + +use ahash::AHashSet; + +/// How far, in characters, a corroborating word may be from a candidate. +pub const WINDOW: usize = 50; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Strength { + Checked, + NeedsWord, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Region { + Any, + Us, + Uk, + Canada, + Australia, + Eu, + Europe, + Asia, + Americas, + Africa, +} + +/// The distinct values one detector found. +#[derive(Debug, Default)] +pub struct Findings(AHashSet); + +impl Findings { + pub fn insert(&mut self, value: impl Into) { + self.0.insert(value.into()); + } + + pub fn len(&self) -> usize { + self.0.len() + } + + pub fn is_empty(&self) -> bool { + self.0.is_empty() + } +} + +pub struct Detector { + /// Stable id, stored in rules: `payment-card`, `iban`, `us-ssn`. + pub id: &'static str, + pub name: &'static str, + pub region: Region, + pub strength: Strength, + find: fn(&str, &mut Findings), +} + +impl Detector { + pub const fn new( + id: &'static str, + name: &'static str, + region: Region, + strength: Strength, + find: fn(&str, &mut Findings), + ) -> Self { + Self { + id, + name, + region, + strength, + find, + } + } + + /// Adds what this detector finds in `text` to `findings`. Call once per + /// piece of text (subject, each part, each attachment) with the same + /// `findings`, then read its length. + pub fn find(&self, text: &str, findings: &mut Findings) { + (self.find)(text, findings) + } + + /// The distinct values found in one text. + pub fn count(&self, text: &str) -> usize { + let mut findings = Findings::default(); + self.find(text, &mut findings); + findings.len() + } +} + +/// Every detector, in the order the console lists them. +pub fn all() -> impl Iterator { + any::DETECTORS.iter() +} + +pub fn by_id(id: &str) -> Option<&'static Detector> { + all().find(|detector| detector.id == id) +} + +/// Whether one of `words` appears, as a whole word and ignoring case, within +/// [`WINDOW`] characters before `start` or after `end` (byte offsets of the +/// candidate in `text`). The window is widened by the longest word, so a +/// word that reaches into it still counts whole. +pub fn word_near(text: &str, start: usize, end: usize, words: &[&str]) -> bool { + let reach = WINDOW + words.iter().map(|w| w.chars().count()).max().unwrap_or(0); + let before = text[..start] + .char_indices() + .rev() + .nth(reach - 1) + .map_or(0, |(i, _)| i); + let after = text[end..] + .char_indices() + .nth(reach) + .map_or(text.len(), |(i, _)| end + i); + let window = text[before..after].to_lowercase(); + words.iter().any(|word| contains_word(&window, word)) +} + +/// Whether `word` (lower case) appears in `haystack` (lower case) with no +/// letter or digit on either side. +pub fn contains_word(haystack: &str, word: &str) -> bool { + haystack.match_indices(word).any(|(i, _)| { + let before_ok = haystack[..i] + .chars() + .next_back() + .is_none_or(|c| !c.is_alphanumeric()); + let after_ok = haystack[i + word.len()..] + .chars() + .next() + .is_none_or(|c| !c.is_alphanumeric()); + before_ok && after_ok + }) +} + +/// Whether the match at `start..end` stands alone: no digit or letter +/// directly before or after it, so `123-45-6789` isn't found inside a +/// longer run of digits. +pub fn stands_alone(text: &str, start: usize, end: usize) -> bool { + let before = text[..start].chars().next_back(); + let after = text[end..].chars().next(); + before.is_none_or(|c| !c.is_alphanumeric()) && after.is_none_or(|c| !c.is_alphanumeric()) +} + +/// The ASCII digits of `s`. +pub fn digits(s: &str) -> String { + s.chars().filter(char::is_ascii_digit).collect() +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn words_are_whole_and_near() { + let text = "Your passport number is X1234567, thanks"; + let start = text.find("X123").unwrap(); + assert!(word_near(text, start, start + 8, &["passport"])); + assert!(!word_near(text, start, start + 8, &["pass"])); + let far = format!("passport{}X1234567", " ".repeat(60)); + let start = far.find("X123").unwrap(); + assert!(!word_near(&far, start, start + 8, &["passport"])); + } + + #[test] + fn near_counts_characters_not_bytes() { + // 45 two-byte characters between the word and the candidate: within + // 50 characters, though over 50 bytes + let text = format!("passport {} X1234567", "é".repeat(45)); + let start = text.find("X123").unwrap(); + assert!(word_near(&text, start, start + 8, &["passport"])); + } + + #[test] + fn ids_are_unique() { + let mut seen = AHashSet::new(); + for detector in all() { + assert!(seen.insert(detector.id), "duplicate id {}", detector.id); + assert!(by_id(detector.id).is_some()); + } + } +} diff --git a/crates/features/src/mailflow/extract.rs b/crates/features/src/mailflow/extract.rs new file mode 100644 index 0000000..ff233ff --- /dev/null +++ b/crates/features/src/mailflow/extract.rs @@ -0,0 +1,692 @@ +/* + * SPDX-FileCopyrightText: 2026 Coffey Labs + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +//! The text of an attachment, for the detectors (§2.3), or why there isn't +//! one. +//! +//! Read: text files (plain, CSV, JSON, XML, HTML), Office Open XML (DOCX, +//! XLSX, PPTX) and OpenDocument (ODT, ODS, ODP) documents, and ZIP archives +//! one level deep. **Can't be inspected**: encrypted or password-protected +//! files, PDF (settled answer 2), the older binary Office formats, archives +//! inside archives, and anything past the limits. Everything else (images, +//! audio, programs) has no text to read and is neither. +//! +//! Office files are ZIP archives of XML, read here with the `zip` and +//! `quick-xml` crates the server already uses: no outside converter runs. + +use quick_xml::{Reader, XmlVersion, events::Event}; +use std::io::{Cursor, Read}; + +/// How much may be unpacked from one attachment, and from how many entries. +#[derive(Debug, Clone, Copy)] +pub struct Limits { + pub max_unpacked: u64, + pub max_entries: usize, +} + +impl Default for Limits { + fn default() -> Self { + Self { + max_unpacked: 50 * 1024 * 1024, + max_entries: 10_000, + } + } +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum Extracted { + /// The text to check. + Text(String), + /// A kind of file with no text in it: nothing to check, nothing missed. + NoText, + /// A file that may hold text the detectors couldn't read. + NotInspectable(Why), +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Why { + Encrypted, + Pdf, + LegacyOffice, + NestedArchive, + TooLarge, + Damaged, +} + +impl Why { + pub fn as_str(&self) -> &'static str { + match self { + Why::Encrypted => "encrypted", + Why::Pdf => "pdf", + Why::LegacyOffice => "legacy-office", + Why::NestedArchive => "nested-archive", + Why::TooLarge => "too-large", + Why::Damaged => "damaged", + } + } +} + +const OLE_MAGIC: &[u8] = &[0xD0, 0xCF, 0x11, 0xE0, 0xA1, 0xB1, 0x1A, 0xE1]; +const ZIP_MAGIC: &[u8] = b"PK\x03\x04"; + +/// What an attachment says, from its declared type, its file name and, above +/// all, its first bytes. +pub fn extract( + content_type: &str, + file_name: Option<&str>, + data: &[u8], + limits: &Limits, +) -> Extracted { + extract_at(content_type, file_name, data, limits, 0) +} + +fn extract_at( + content_type: &str, + file_name: Option<&str>, + data: &[u8], + limits: &Limits, + depth: u8, +) -> Extracted { + let content_type = content_type.to_ascii_lowercase(); + let extension = file_name + .and_then(|name| name.rsplit_once('.')) + .map(|(_, ext)| ext.to_ascii_lowercase()) + .unwrap_or_default(); + + if data.len() as u64 > limits.max_unpacked { + return Extracted::NotInspectable(Why::TooLarge); + } + if data.starts_with(b"%PDF-") || content_type == "application/pdf" || extension == "pdf" { + return Extracted::NotInspectable(Why::Pdf); + } + if data.starts_with(OLE_MAGIC) { + // An encrypted OOXML file is an OLE container holding the encrypted + // package; any other OLE file is a legacy .doc, .xls or .ppt + return Extracted::NotInspectable(if has_utf16(data, "EncryptedPackage") { + Why::Encrypted + } else { + Why::LegacyOffice + }); + } + if data.starts_with(ZIP_MAGIC) { + if depth > 0 { + return Extracted::NotInspectable(Why::NestedArchive); + } + return zip(data, limits); + } + if is_text(&content_type, &extension) { + let text = decode_text(data); + return Extracted::Text( + if content_type == "text/html" || matches!(extension.as_str(), "html" | "htm") { + strip_html(&text) + } else { + text + }, + ); + } + Extracted::NoText +} + +fn is_text(content_type: &str, extension: &str) -> bool { + content_type.starts_with("text/") + || matches!( + content_type, + "application/json" + | "application/xml" + | "application/csv" + | "application/x-csv" + | "message/rfc822" + ) + || matches!( + extension, + "txt" + | "csv" + | "tsv" + | "json" + | "xml" + | "md" + | "log" + | "html" + | "htm" + | "eml" + | "ics" + | "vcf" + ) +} + +/// UTF-16 with a byte order mark, else UTF-8 (lossy). +fn decode_text(data: &[u8]) -> String { + let utf16 = |bytes: &[u8], big: bool| { + let units: Vec = bytes + .chunks_exact(2) + .map(|c| { + if big { + u16::from_be_bytes([c[0], c[1]]) + } else { + u16::from_le_bytes([c[0], c[1]]) + } + }) + .collect(); + String::from_utf16_lossy(&units) + }; + match data { + [0xFF, 0xFE, rest @ ..] => utf16(rest, false), + [0xFE, 0xFF, rest @ ..] => utf16(rest, true), + [0xEF, 0xBB, 0xBF, rest @ ..] => String::from_utf8_lossy(rest).into_owned(), + _ => String::from_utf8_lossy(data).into_owned(), + } +} + +fn has_utf16(data: &[u8], needle: &str) -> bool { + let needle: Vec = needle.encode_utf16().flat_map(u16::to_le_bytes).collect(); + data.windows(needle.len()).any(|w| w == needle.as_slice()) +} + +/// Tags out, the common entities decoded, block ends as new lines. +fn strip_html(html: &str) -> String { + let mut out = String::with_capacity(html.len()); + let mut in_tag = false; + let mut skip_until: Option<&str> = None; + let lower = html.to_ascii_lowercase(); + let mut i = 0; + let bytes = html.as_bytes(); + while i < bytes.len() { + if let Some(end) = skip_until { + match lower[i..].find(end) { + Some(at) => { + i += at + end.len(); + skip_until = None; + } + None => break, + } + continue; + } + let c = bytes[i]; + if in_tag { + if c == b'>' { + in_tag = false; + } + i += 1; + continue; + } + if c == b'<' { + if lower[i..].starts_with(""); + } else if lower[i..].starts_with(""); + } else { + if [ + ""), + (""", "\""), + ("'", "'"), + ("&", "&"), + ] { + out = out.replace(entity, text); + } + out +} + +/// A ZIP file: an Office document, an OpenDocument, or an archive. +fn zip(data: &[u8], limits: &Limits) -> Extracted { + let Ok(mut archive) = zip::ZipArchive::new(Cursor::new(data)) else { + return Extracted::NotInspectable(Why::Damaged); + }; + if archive.len() > limits.max_entries { + return Extracted::NotInspectable(Why::TooLarge); + } + let mut names = Vec::with_capacity(archive.len()); + let mut declared: u64 = 0; + for i in 0..archive.len() { + let Ok(entry) = archive.by_index_raw(i) else { + return Extracted::NotInspectable(Why::Damaged); + }; + if entry.encrypted() { + return Extracted::NotInspectable(Why::Encrypted); + } + declared = declared.saturating_add(entry.size()); + names.push(entry.name().to_string()); + } + if declared > limits.max_unpacked { + return Extracted::NotInspectable(Why::TooLarge); + } + let mut budget = limits.max_unpacked; + let mut read = + |archive: &mut zip::ZipArchive>, name: &str| -> Result, Why> { + let entry = archive.by_name(name).map_err(|_| Why::Damaged)?; + let mut bytes = Vec::new(); + // Declared sizes can lie: stop at the budget whatever they say + entry + .take(budget + 1) + .read_to_end(&mut bytes) + .map_err(|_| Why::Damaged)?; + if bytes.len() as u64 > budget { + return Err(Why::TooLarge); + } + budget -= bytes.len() as u64; + Ok(bytes) + }; + + let has = |name: &str| names.iter().any(|n| n == name); + let mut text = String::new(); + let result: Result<(), Why> = (|| { + if has("[Content_Types].xml") { + // Office Open XML: the parts that hold what a person wrote + let mut shared = Vec::new(); + if has("xl/sharedStrings.xml") { + shared = xml_strings(&read(&mut archive, "xl/sharedStrings.xml")?, "si"); + } + for name in names.iter().filter(|n| ooxml_text_part(n)) { + let xml = read(&mut archive, name)?; + if name.starts_with("xl/worksheets/") { + xlsx_sheet(&xml, &mut text); + } else { + xml_text(&xml, &mut text); + } + text.push('\n'); + } + text.extend(shared.iter().map(|s| format!("{s}\n"))); + } else if names.first().is_some_and(|n| n == "mimetype") + && read(&mut archive, "mimetype")?.starts_with(b"application/vnd.oasis.opendocument") + { + // OpenDocument: an encrypted one says so in its manifest + if has("META-INF/manifest.xml") + && contains( + &read(&mut archive, "META-INF/manifest.xml")?, + b"encryption-data", + ) + { + return Err(Why::Encrypted); + } + for name in ["content.xml", "styles.xml"] { + if has(name) { + xml_text(&read(&mut archive, name)?, &mut text); + text.push('\n'); + } + } + } else { + // An archive: each file inside, one level deep + for name in names.iter().filter(|n| !n.ends_with('/')) { + let bytes = read(&mut archive, name)?; + match extract_at("", Some(name), &bytes, limits, 1) { + Extracted::Text(inner) => { + text.push_str(&inner); + text.push('\n'); + } + Extracted::NoText => {} + Extracted::NotInspectable(why) => return Err(why), + } + } + } + Ok(()) + })(); + match result { + Ok(()) => Extracted::Text(text), + Err(why) => Extracted::NotInspectable(why), + } +} + +fn ooxml_text_part(name: &str) -> bool { + let xml = name.ends_with(".xml"); + xml && (name == "word/document.xml" + || [ + "word/header", + "word/footer", + "word/footnotes", + "word/endnotes", + "word/comments", + ] + .iter() + .any(|p| name.starts_with(p)) + || name.starts_with("xl/worksheets/sheet") + || name.starts_with("ppt/slides/slide") + || name.starts_with("ppt/notesSlides/")) +} + +fn contains(haystack: &[u8], needle: &[u8]) -> bool { + haystack.windows(needle.len()).any(|w| w == needle) +} + +/// The local name of a tag, without its namespace prefix. +fn local(name: &[u8]) -> &[u8] { + name.rsplit(|b| *b == b':').next().unwrap_or(name) +} + +fn push_entity(entity: &[u8], out: &mut String) { + match entity { + b"lt" => out.push('<'), + b"gt" => out.push('>'), + b"amp" => out.push('&'), + b"apos" => out.push('\''), + b"quot" => out.push('"'), + _ => { + let code = match entity { + [b'#', b'x' | b'X', hex @ ..] => std::str::from_utf8(hex) + .ok() + .and_then(|h| u32::from_str_radix(h, 16).ok()), + [b'#', dec @ ..] => std::str::from_utf8(dec).ok().and_then(|d| d.parse().ok()), + _ => None, + }; + if let Some(c) = code.and_then(char::from_u32) { + out.push(c); + } + } + } +} + +/// Every text node, runs joined as written, a new line after each paragraph +/// or row and a tab after each cell, so a number split across runs is whole +/// again. +fn xml_text(xml: &[u8], out: &mut String) { + let mut reader = Reader::from_reader(xml); + let mut buf = Vec::new(); + loop { + match reader.read_event_into(&mut buf) { + Ok(Event::Text(t)) => { + if let Ok(text) = t.xml_content(XmlVersion::Implicit1_0) { + out.push_str(&text); + } + } + Ok(Event::CData(t)) => out.push_str(&String::from_utf8_lossy(&t)), + Ok(Event::GeneralRef(entity)) => push_entity(&entity, out), + Ok(Event::End(e)) => match local(e.name().as_ref()) { + b"p" | b"h" | b"tr" | b"row" | b"table-row" | b"br" => out.push('\n'), + b"tc" | b"c" | b"table-cell" | b"tab" => out.push('\t'), + _ => {} + }, + Ok(Event::Empty(e)) => match local(e.name().as_ref()) { + b"br" | b"line-break" => out.push('\n'), + b"tab" | b"s" => out.push(' '), + _ => {} + }, + Ok(Event::Eof) | Err(_) => break, + _ => {} + } + buf.clear(); + } +} + +/// The text of each `item` element (a shared string in XLSX). +fn xml_strings(xml: &[u8], item: &str) -> Vec { + let mut reader = Reader::from_reader(xml); + let mut buf = Vec::new(); + let mut items = Vec::new(); + let mut current: Option = None; + loop { + match reader.read_event_into(&mut buf) { + Ok(Event::Start(e)) if local(e.name().as_ref()) == item.as_bytes() => { + current = Some(String::new()) + } + Ok(Event::End(e)) if local(e.name().as_ref()) == item.as_bytes() => { + items.extend(current.take()); + } + Ok(Event::Text(t)) => { + if let (Some(s), Ok(text)) = + (current.as_mut(), t.xml_content(XmlVersion::Implicit1_0)) + { + s.push_str(&text); + } + } + Ok(Event::GeneralRef(entity)) => { + if let Some(s) = current.as_mut() { + push_entity(&entity, s); + } + } + Ok(Event::Eof) | Err(_) => break, + _ => {} + } + buf.clear(); + } + items +} + +/// A worksheet's cell values: numbers and inline strings. Cells holding a +/// shared string are skipped here; the shared strings are read whole. +fn xlsx_sheet(xml: &[u8], out: &mut String) { + let mut reader = Reader::from_reader(xml); + let mut buf = Vec::new(); + let mut shared_cell = false; + let mut in_value = false; + loop { + match reader.read_event_into(&mut buf) { + Ok(Event::Start(e)) => match local(e.name().as_ref()) { + b"c" => { + shared_cell = e + .attributes() + .flatten() + .any(|a| a.key.as_ref() == b"t" && a.value.as_ref() == b"s"); + } + b"v" | b"t" => in_value = true, + _ => {} + }, + Ok(Event::End(e)) => match local(e.name().as_ref()) { + b"v" | b"t" => in_value = false, + b"c" => out.push('\t'), + b"row" => out.push('\n'), + _ => {} + }, + // A shared string's cell holds only its index: the string itself + // is added with the shared strings + Ok(Event::Text(t)) if in_value && !shared_cell => { + if let Ok(text) = t.xml_content(XmlVersion::Implicit1_0) { + out.push_str(&text); + } + } + Ok(Event::Eof) | Err(_) => break, + _ => {} + } + buf.clear(); + } +} + +#[cfg(test)] +mod tests { + use super::*; + use std::io::Write; + use zip::{ZipWriter, write::SimpleFileOptions}; + + fn zip_of(files: &[(&str, &str)]) -> Vec { + let mut zip = ZipWriter::new(Cursor::new(Vec::new())); + for (name, body) in files { + zip.start_file(*name, SimpleFileOptions::default()).unwrap(); + zip.write_all(body.as_bytes()).unwrap(); + } + zip.finish().unwrap().into_inner() + } + + fn text_of(extracted: Extracted) -> String { + match extracted { + Extracted::Text(text) => text, + other => panic!("expected text, got {other:?}"), + } + } + + #[test] + fn plain_text_and_html() { + let limits = Limits::default(); + assert_eq!( + text_of(extract("text/plain", None, b"card 4242", &limits)), + "card 4242" + ); + let utf16: Vec = [0xFF, 0xFE] + .into_iter() + .chain("héllo".encode_utf16().flat_map(u16::to_le_bytes)) + .collect(); + assert_eq!( + text_of(extract( + "application/octet-stream", + Some("a.csv"), + &utf16, + &limits + )), + "héllo" + ); + let html = "

Card 4242

ab"; + let text = text_of(extract("text/html", None, html.as_bytes(), &limits)); + assert!( + text.contains("Card 4242") && !text.contains("x()") && !text.contains("p{}"), + "{text:?}" + ); + assert_eq!( + extract("image/png", Some("a.png"), b"\x89PNG....", &limits), + Extracted::NoText + ); + } + + #[test] + fn docx_joins_split_runs() { + let doc = r#"Card 4242 4242 4242 4242A & B"#; + let docx = zip_of(&[ + ("[Content_Types].xml", ""), + ("word/document.xml", doc), + ]); + let text = text_of(extract( + "application/vnd.openxmlformats-officedocument.wordprocessingml.document", + Some("a.docx"), + &docx, + &Limits::default(), + )); + assert!(text.contains("Card 4242 4242 4242 4242\nA & B"), "{text:?}"); + } + + #[test] + fn xlsx_numbers_and_shared_strings() { + let sheet = r#"04242424242424242"#; + let shared = r#"IBAN GB29 NWBK 6016 1331 9268 19"#; + let xlsx = zip_of(&[ + ("[Content_Types].xml", ""), + ("xl/sharedStrings.xml", shared), + ("xl/worksheets/sheet1.xml", sheet), + ]); + let text = text_of(extract("", Some("book.xlsx"), &xlsx, &Limits::default())); + assert!( + text.contains("4242424242424242") && text.contains("GB29 NWBK 6016 1331 9268 19"), + "{text:?}" + ); + // The shared string's index isn't read as a value + assert!( + !text.contains("\t0\t") && !text.starts_with('0'), + "{text:?}" + ); + } + + #[test] + fn opendocument_and_encrypted_opendocument() { + let content = r#"SSN 078-05-1120"#; + let odt = zip_of(&[ + ("mimetype", "application/vnd.oasis.opendocument.text"), + ("content.xml", content), + ]); + assert!( + text_of(extract("", Some("a.odt"), &odt, &Limits::default())) + .contains("SSN 078-05-1120") + ); + let manifest = r#""#; + let locked = zip_of(&[ + ("mimetype", "application/vnd.oasis.opendocument.text"), + ("META-INF/manifest.xml", manifest), + ("content.xml", "x"), + ]); + assert_eq!( + extract("", Some("a.odt"), &locked, &Limits::default()), + Extracted::NotInspectable(Why::Encrypted) + ); + } + + #[test] + fn archives() { + let limits = Limits::default(); + let archive = zip_of(&[ + ("notes/a.txt", "card 4242424242424242"), + ("b.png", "\u{89}PNG"), + ]); + assert!( + text_of(extract("application/zip", Some("x.zip"), &archive, &limits)) + .contains("4242424242424242") + ); + let nested = zip_of(&[( + "inner.zip", + std::str::from_utf8(&[b'P', b'K', 3, 4]).unwrap(), + )]); + assert_eq!( + extract("application/zip", Some("x.zip"), &nested, &limits), + Extracted::NotInspectable(Why::NestedArchive) + ); + + // Password-protected + let mut zip = ZipWriter::new(Cursor::new(Vec::new())); + zip.start_file( + "secret.txt", + SimpleFileOptions::default().with_aes_encryption(zip::AesMode::Aes256, "pw"), + ) + .unwrap(); + zip.write_all(b"4242424242424242").unwrap(); + let locked = zip.finish().unwrap().into_inner(); + assert_eq!( + extract("application/zip", Some("x.zip"), &locked, &limits), + Extracted::NotInspectable(Why::Encrypted) + ); + + // Past the limits + let small = Limits { + max_unpacked: 10, + max_entries: 1, + }; + assert_eq!( + extract("application/zip", Some("x.zip"), &archive, &small), + Extracted::NotInspectable(Why::TooLarge) + ); + assert_eq!( + extract("application/zip", None, b"PK\x03\x04garbage", &limits), + Extracted::NotInspectable(Why::Damaged) + ); + } + + #[test] + fn not_inspectable_kinds() { + let limits = Limits::default(); + assert_eq!( + extract("application/octet-stream", None, b"%PDF-1.7 ...", &limits), + Extracted::NotInspectable(Why::Pdf) + ); + let mut ole = OLE_MAGIC.to_vec(); + ole.extend(std::iter::repeat_n(0, 64)); + assert_eq!( + extract("", Some("old.doc"), &ole, &limits), + Extracted::NotInspectable(Why::LegacyOffice) + ); + ole.extend("EncryptedPackage".encode_utf16().flat_map(u16::to_le_bytes)); + assert_eq!( + extract("", Some("new.docx"), &ole, &limits), + Extracted::NotInspectable(Why::Encrypted) + ); + } +} diff --git a/crates/features/src/mailflow/mod.rs b/crates/features/src/mailflow/mod.rs new file mode 100644 index 0000000..d218aac --- /dev/null +++ b/crates/features/src/mailflow/mod.rs @@ -0,0 +1,23 @@ +/* + * SPDX-FileCopyrightText: 2026 Coffey Labs + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +//! Data loss prevention and mail flow rules (dlp-and-mail-flow-rules spec). +//! +//! Pure functions over text and attachment bytes, so everything here is +//! unit-tested without a server: +//! +//! - [`detectors`]: find identifiers in text (payment cards, IBANs, +//! national ID numbers, keys), each by its published format and check +//! (§2.3); +//! - [`words`]: an organization's own word lists and patterns; +//! - [`extract`]: the text of an attachment, or why it can't be read. +//! +//! Nothing here writes what it finds anywhere: callers get counts, and the +//! matched text never leaves the evaluation (§2.7). + +pub mod detectors; +pub mod extract; +pub mod words; diff --git a/crates/features/src/mailflow/words.rs b/crates/features/src/mailflow/words.rs new file mode 100644 index 0000000..e171ff2 --- /dev/null +++ b/crates/features/src/mailflow/words.rs @@ -0,0 +1,109 @@ +/* + * SPDX-FileCopyrightText: 2026 Coffey Labs + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +//! An organization's own word lists and patterns (§2.3). Both count +//! occurrences, not distinct values: "confidential" three times is three. + +use aho_corasick::{AhoCorasick, AhoCorasickBuilder, MatchKind}; +use regex::{Regex, RegexBuilder}; + +/// How large a compiled pattern may grow. Keeps a rule someone writes from +/// making every message slow to send. +const PATTERN_SIZE_LIMIT: usize = 1 << 20; + +/// Words and phrases, matched whole and ignoring case. +#[derive(Debug, Clone)] +pub struct WordList { + matcher: AhoCorasick, +} + +impl WordList { + /// Builds a list from words or phrases; empty entries are skipped. + pub fn new(words: I) -> Result + where + I: IntoIterator, + S: AsRef, + { + let words: Vec = words + .into_iter() + .map(|w| w.as_ref().trim().to_lowercase()) + .filter(|w| !w.is_empty()) + .collect(); + if words.is_empty() { + return Err("The list has no words".into()); + } + AhoCorasickBuilder::new() + .match_kind(MatchKind::LeftmostLongest) + .build(&words) + .map(|matcher| Self { matcher }) + .map_err(|err| err.to_string()) + } + + /// How many times any word of the list appears in `text`. + pub fn count(&self, text: &str) -> usize { + let text = text.to_lowercase(); + self.matcher + .find_iter(&text) + .filter(|m| super::detectors::stands_alone(&text, m.start(), m.end())) + .count() + } +} + +/// An organization's regular expression. +#[derive(Debug, Clone)] +pub struct Pattern { + regex: Regex, +} + +impl Pattern { + /// Compiles `pattern`, or says why it can't be used. Matching ignores + /// case unless the pattern turns that off with `(?-i)`. + pub fn new(pattern: &str) -> Result { + RegexBuilder::new(pattern) + .case_insensitive(true) + .size_limit(PATTERN_SIZE_LIMIT) + .build() + .map(|regex| Self { regex }) + .map_err(|err| err.to_string()) + } + + /// How many times the pattern matches in `text`. + pub fn count(&self, text: &str) -> usize { + self.regex.find_iter(text).filter(|m| !m.is_empty()).count() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn words_whole_and_any_case() { + let list = WordList::new(["Project Falcon", "confidential", " "]).unwrap(); + assert_eq!( + list.count( + "CONFIDENTIAL: project falcon notes. Not confidentiality, not projectfalcon." + ), + 2 + ); + assert_eq!(list.count("Confidential, confidential and confidential"), 3); + // Non-ASCII case folding + let list = WordList::new(["GEHEIM", "Straße"]).unwrap(); + assert_eq!(list.count("streng geheim, STRASSE ist nicht Straße"), 2); + assert!(WordList::new(["", " "]).is_err()); + } + + #[test] + fn patterns() { + let pattern = Pattern::new(r"\bPRJ-\d{4}\b").unwrap(); + assert_eq!(pattern.count("prj-1234 and PRJ-5678, not PRJ-12"), 2); + assert!(Pattern::new("(unclosed").is_err()); + // Too large to compile within the limit + assert!(Pattern::new(r"\w{1000}\w{1000}\w{1000}").is_err()); + // Empty matches don't count + assert_eq!(Pattern::new("x*").unwrap().count("abc"), 0); + } +} From 3eb5a454fde0dacd5f533c976815b38f35f0565e Mon Sep 17 00:00:00 2001 From: John Coffey Date: Mon, 28 Sep 2026 17:01:00 -0700 Subject: [PATCH 2/2] Cargo.lock: the features crate's new dependencies --- Cargo.lock | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/Cargo.lock b/Cargo.lock index f624cd5..d14ec2d 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -3947,9 +3947,12 @@ name = "inbuxa-features" version = "0.16.22" dependencies = [ "ahash", + "aho-corasick", "base64 0.23.1", "flate2", "jmap_proto", + "quick-xml 0.41.0", + "regex", "registry", "serde", "serde_json", @@ -3961,6 +3964,7 @@ dependencies = [ "types", "utils", "xxhash-rust", + "zip", ] [[package]]