diff --git a/crates/features/src/mailflow/detectors/africa.rs b/crates/features/src/mailflow/detectors/africa.rs new file mode 100644 index 0000000..6058132 --- /dev/null +++ b/crates/features/src/mailflow/detectors/africa.rs @@ -0,0 +1,49 @@ +/* + * SPDX-FileCopyrightText: 2026 Coffey Labs + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +//! African identifiers (§2.3): South Africa's ID number. + +use super::{Detector, Findings, Region, Strength, checks, valid_short_date}; +use regex::Regex; +use std::sync::LazyLock; + +pub static DETECTORS: &[Detector] = &[Detector::new( + "za-id", + "South Africa: ID number", + Region::Africa, + Strength::Checked, + za_id, +)]; + +/// Birth date `YYMMDD`, four digits, citizenship (0, 1 or 2), 8 or 9, a Luhn +/// check digit. The date and the two fixed digits make it strong enough to +/// count alone. +static ZA_ID: LazyLock = LazyLock::new(|| { + Regex::new(r"\b(\d{2})(\d{2})(\d{2})\d{4}[012][89]\d\b").expect("detector pattern") +}); + +fn za_id(text: &str, findings: &mut Findings) { + for c in ZA_ID.captures_iter(text) { + let n = &c[0]; + let num = |s: &str| s.parse::().unwrap_or(0); + if valid_short_date(num(&c[1]), num(&c[2]), num(&c[3])) && checks::luhn(n) { + findings.insert(n); + } + } +} + +#[cfg(test)] +mod tests { + use crate::mailflow::detectors::by_id; + + #[test] + fn south_africa() { + let detector = by_id("za-id").unwrap(); + assert_eq!(detector.count("ID 8001015009087"), 1); + assert_eq!(detector.count("8001015009088"), 0); + assert_eq!(detector.count("8013015009087"), 0); + } +} diff --git a/crates/features/src/mailflow/detectors/americas.rs b/crates/features/src/mailflow/detectors/americas.rs new file mode 100644 index 0000000..2ab1c44 --- /dev/null +++ b/crates/features/src/mailflow/detectors/americas.rs @@ -0,0 +1,172 @@ +/* + * SPDX-FileCopyrightText: 2026 Coffey Labs + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +//! Identifiers from the Americas outside the US and Canada (§2.3): Brazil's +//! CPF and CNPJ, and Mexico's CURP. + +use super::{Detector, Findings, Region, Strength, digit_values, valid_short_date, word_near}; +use regex::Regex; +use std::sync::LazyLock; + +pub static DETECTORS: &[Detector] = &[ + Detector::new( + "br-cpf", + "Brazil: CPF", + Region::Americas, + Strength::Checked, + br_cpf, + ), + Detector::new( + "br-cnpj", + "Brazil: CNPJ", + Region::Americas, + Strength::Checked, + br_cnpj, + ), + Detector::new( + "mx-curp", + "Mexico: CURP", + Region::Americas, + Strength::Checked, + mx_curp, + ), +]; + +fn re(pattern: &str) -> Regex { + Regex::new(pattern).expect("detector pattern") +} + +/// Brazil's mod 11 check digit over `digits` with `weights`. +fn br_check(digits: &[u32], weights: &[u32]) -> u32 { + match digits.iter().zip(weights).map(|(a, w)| a * w).sum::() % 11 { + 0 | 1 => 0, + r => 11 - r, + } +} + +/// `111.444.777-35`, or eleven bare digits. +static CPF: LazyLock = LazyLock::new(|| re(r"\b\d{3}(\.?)\d{3}(\.?)\d{3}(-?)\d{2}\b")); + +pub fn cpf_valid(n: &str) -> bool { + let d = digit_values(n); + // A run of one digit passes the arithmetic but is never issued + d.len() == 11 + && d.iter().any(|x| *x != d[0]) + && br_check(&d[..9], &[10, 9, 8, 7, 6, 5, 4, 3, 2]) == d[9] + && br_check(&d[..10], &[11, 10, 9, 8, 7, 6, 5, 4, 3, 2]) == d[10] +} + +const CPF_WORDS: &[&str] = &[ + "cpf", + "cadastro de pessoas físicas", + "cadastro de pessoa física", +]; + +fn br_cpf(text: &str, findings: &mut Findings) { + for c in CPF.captures_iter(text) { + let whole = c.get(0).unwrap(); + let written = &c[1] == "." && &c[2] == "." && &c[3] == "-"; + let n: String = whole + .as_str() + .chars() + .filter(char::is_ascii_digit) + .collect(); + if cpf_valid(&n) && (written || word_near(text, whole.start(), whole.end(), CPF_WORDS)) { + findings.insert(n); + } + } +} + +/// `11.222.333/0001-81`, or fourteen bare digits. +static CNPJ: LazyLock = + LazyLock::new(|| re(r"\b\d{2}(\.?)\d{3}(\.?)\d{3}(/?)\d{4}(-?)\d{2}\b")); + +pub fn cnpj_valid(n: &str) -> bool { + let d = digit_values(n); + d.len() == 14 + && d.iter().any(|x| *x != d[0]) + && br_check(&d[..12], &[5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2]) == d[12] + && br_check(&d[..13], &[6, 5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2]) == d[13] +} + +const CNPJ_WORDS: &[&str] = &["cnpj", "cadastro nacional da pessoa jurídica"]; + +fn br_cnpj(text: &str, findings: &mut Findings) { + for c in CNPJ.captures_iter(text) { + let whole = c.get(0).unwrap(); + let written = &c[1] == "." && &c[2] == "." && &c[3] == "/" && &c[4] == "-"; + let n: String = whole + .as_str() + .chars() + .filter(char::is_ascii_digit) + .collect(); + if cnpj_valid(&n) && (written || word_near(text, whole.start(), whole.end(), CNPJ_WORDS)) { + findings.insert(n); + } + } +} + +/// Four letters, the birth date, sex (H, M or X), the state, three +/// consonants, a character that tells the century apart, the check digit. +static CURP: LazyLock = LazyLock::new(|| { + re(r"(?i)\b[A-Z]{4}(\d{2})(\d{2})(\d{2})[HMX][A-Z]{2}[B-DF-HJ-NP-TV-Z]{3}[A-Z0-9]\d\b") +}); + +/// RENAPO's check: each character's place in `0-9 A-N Ñ O-Z`, weighted 18 +/// down to 2; the digit is 10 minus the sum mod 10 (10 becomes 0). +pub fn curp_valid(curp: &str) -> bool { + const ALPHABET: &str = "0123456789ABCDEFGHIJKLMNÑOPQRSTUVWXYZ"; + let mut sum = 0u32; + for (i, c) in curp.chars().take(17).enumerate() { + let Some(value) = ALPHABET.chars().position(|a| a == c) else { + return false; + }; + sum += value as u32 * (18 - i as u32); + } + curp.chars().nth(17).and_then(|c| c.to_digit(10)) == Some((10 - sum % 10) % 10) +} + +fn mx_curp(text: &str, findings: &mut Findings) { + for c in CURP.captures_iter(text) { + let curp = c[0].to_ascii_uppercase(); + if valid_short_date(num(&c[1]), num(&c[2]), num(&c[3])) && curp_valid(&curp) { + findings.insert(curp); + } + } +} + +fn num(s: &str) -> u32 { + s.parse().unwrap_or(0) +} + +#[cfg(test)] +mod tests { + use crate::mailflow::detectors::by_id; + + fn count(id: &str, text: &str) -> usize { + by_id(id).unwrap().count(text) + } + + #[test] + fn brazil() { + assert_eq!(count("br-cpf", "CPF 111.444.777-35"), 1); + assert_eq!(count("br-cpf", "111.444.777-36"), 0); + assert_eq!(count("br-cpf", "pedido 11144477735"), 0); + assert_eq!(count("br-cpf", "cpf: 11144477735"), 1); + assert_eq!(count("br-cpf", "CPF 111.111.111-11"), 0); + assert_eq!(count("br-cnpj", "11.222.333/0001-81"), 1); + assert_eq!(count("br-cnpj", "11.222.333/0001-82"), 0); + assert_eq!(count("br-cnpj", "CNPJ 11222333000181"), 1); + } + + #[test] + fn mexico() { + // python-stdnum's documented example + assert_eq!(count("mx-curp", "CURP BOXW310820HNERXN09"), 1); + assert_eq!(count("mx-curp", "BOXW310820HNERXN08"), 0); + assert_eq!(count("mx-curp", "BOXW311320HNERXN09"), 0); + } +} diff --git a/crates/features/src/mailflow/detectors/any.rs b/crates/features/src/mailflow/detectors/any.rs index 1120cb7..387a0d5 100644 --- a/crates/features/src/mailflow/detectors/any.rs +++ b/crates/features/src/mailflow/detectors/any.rs @@ -6,7 +6,9 @@ //! Detectors that aren't tied to one country (§2.3, region "Any"). -use super::{Detector, Findings, Region, Strength, checks, digits, stands_alone, word_near}; +use super::{ + Detector, Findings, Region, Strength, checks, digits, stands_alone, valid_date, word_near, +}; use regex::Regex; use std::sync::LazyLock; @@ -295,19 +297,6 @@ const BIRTH_WORDS: &[&str] = &[ "nascimento", ]; -fn valid_date(year: u32, month: u32, day: u32) -> bool { - let days = match month { - 1 | 3 | 5 | 7 | 8 | 10 | 12 => 31, - 4 | 6 | 9 | 11 => 30, - 2 if year.is_multiple_of(4) && (!year.is_multiple_of(100) || year.is_multiple_of(400)) => { - 29 - } - 2 => 28, - _ => return false, - }; - (1900..=2100).contains(&year) && (1..=days).contains(&day) -} - fn month_number(name: &str) -> u32 { const MONTHS: [&str; 12] = [ "jan", "feb", "mar", "apr", "may", "jun", "jul", "aug", "sep", "oct", "nov", "dec", diff --git a/crates/features/src/mailflow/detectors/asia.rs b/crates/features/src/mailflow/detectors/asia.rs new file mode 100644 index 0000000..8f32e87 --- /dev/null +++ b/crates/features/src/mailflow/detectors/asia.rs @@ -0,0 +1,281 @@ +/* + * SPDX-FileCopyrightText: 2026 Coffey Labs + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +//! Asian identifiers (§2.3): India's Aadhaar and PAN, China's resident ID, +//! Japan's My Number, Singapore's NRIC and FIN, and South Korea's resident +//! registration number. + +use super::{ + Detector, Findings, Region, Strength, digit_values, valid_date, valid_short_date, word_near, +}; +use regex::Regex; +use std::sync::LazyLock; + +pub static DETECTORS: &[Detector] = &[ + Detector::new( + "in-aadhaar", + "India: Aadhaar", + Region::Asia, + Strength::Checked, + in_aadhaar, + ), + Detector::new( + "in-pan", + "India: PAN", + Region::Asia, + Strength::NeedsWord, + in_pan, + ), + Detector::new( + "cn-resident-id", + "China: resident ID", + Region::Asia, + Strength::Checked, + cn_resident_id, + ), + Detector::new( + "jp-my-number", + "Japan: My Number", + Region::Asia, + Strength::Checked, + jp_my_number, + ), + Detector::new( + "sg-nric", + "Singapore: NRIC and FIN", + Region::Asia, + Strength::Checked, + sg_nric, + ), + Detector::new( + "kr-rrn", + "South Korea: resident registration number", + Region::Asia, + Strength::NeedsWord, + kr_rrn, + ), +]; + +fn re(pattern: &str) -> Regex { + Regex::new(pattern).expect("detector pattern") +} + +/// Twelve digits written in fours, or bare. +static TWELVE: LazyLock = LazyLock::new(|| re(r"\b(\d{4})( ?)(\d{4})( ?)(\d{4})\b")); + +const VERHOEFF_D: [[u8; 10]; 10] = [ + [0, 1, 2, 3, 4, 5, 6, 7, 8, 9], + [1, 2, 3, 4, 0, 6, 7, 8, 9, 5], + [2, 3, 4, 0, 1, 7, 8, 9, 5, 6], + [3, 4, 0, 1, 2, 8, 9, 5, 6, 7], + [4, 0, 1, 2, 3, 9, 5, 6, 7, 8], + [5, 9, 8, 7, 6, 0, 4, 3, 2, 1], + [6, 5, 9, 8, 7, 1, 0, 4, 3, 2], + [7, 6, 5, 9, 8, 2, 1, 0, 4, 3], + [8, 7, 6, 5, 9, 3, 2, 1, 0, 4], + [9, 8, 7, 6, 5, 4, 3, 2, 1, 0], +]; +const VERHOEFF_P: [[u8; 10]; 8] = [ + [0, 1, 2, 3, 4, 5, 6, 7, 8, 9], + [1, 5, 7, 6, 2, 8, 3, 0, 9, 4], + [5, 8, 0, 3, 7, 9, 6, 1, 4, 2], + [8, 9, 1, 6, 0, 4, 3, 5, 2, 7], + [9, 4, 5, 3, 1, 2, 6, 8, 7, 0], + [4, 2, 8, 6, 5, 7, 3, 9, 0, 1], + [2, 7, 9, 3, 8, 0, 6, 4, 1, 5], + [7, 0, 4, 6, 9, 1, 3, 2, 5, 8], +]; + +/// The Verhoeff check (dihedral group D5). +pub fn verhoeff(n: &str) -> bool { + let mut c = 0u8; + for (i, b) in n.bytes().rev().enumerate() { + c = VERHOEFF_D[c as usize][VERHOEFF_P[i % 8][(b - b'0') as usize] as usize]; + } + c == 0 +} + +const AADHAAR_WORDS: &[&str] = &["aadhaar", "aadhar", "uidai", "uid"]; + +fn in_aadhaar(text: &str, findings: &mut Findings) { + for c in TWELVE.captures_iter(text) { + let whole = c.get(0).unwrap(); + let n = format!("{}{}{}", &c[1], &c[3], &c[5]); + let written = &c[2] == " " && &c[4] == " "; + // Never starts with 0 or 1 + if !n.starts_with(['0', '1']) + && verhoeff(&n) + && (written || word_near(text, whole.start(), whole.end(), AADHAAR_WORDS)) + { + findings.insert(n); + } + } +} + +/// Five letters (the fourth names the holder's type), four digits, a letter. +static PAN: LazyLock = LazyLock::new(|| re(r"\b[A-Z]{3}[ABCFGHLJPTK][A-Z]\d{4}[A-Z]\b")); + +const PAN_WORDS: &[&str] = &["pan", "pan card", "permanent account number", "income tax"]; + +fn in_pan(text: &str, findings: &mut Findings) { + for m in PAN.find_iter(text) { + if word_near(text, m.start(), m.end(), PAN_WORDS) { + findings.insert(m.as_str()); + } + } +} + +/// Region, birth date `YYYYMMDD`, sequence, then the ISO 7064 MOD 11-2 +/// check (0–9 or X). +static CN_ID: LazyLock = + LazyLock::new(|| re(r"(?i)\b[1-8]\d{5}(\d{4})(\d{2})(\d{2})\d{3}[\dX]\b")); + +pub fn cn_id_valid(id: &str) -> bool { + const WEIGHTS: [u32; 17] = [7, 9, 10, 5, 8, 4, 2, 1, 6, 3, 7, 9, 10, 5, 8, 4, 2]; + const CHECKS: &[u8] = b"10X98765432"; + let sum: u32 = digit_values(&id[..17]) + .iter() + .zip(WEIGHTS) + .map(|(a, w)| a * w) + .sum(); + CHECKS[(sum % 11) as usize] == id.as_bytes()[17].to_ascii_uppercase() +} + +fn cn_resident_id(text: &str, findings: &mut Findings) { + for c in CN_ID.captures_iter(text) { + let id = c[0].to_ascii_uppercase(); + let (y, m, d) = (num(&c[1]), num(&c[2]), num(&c[3])); + if valid_date(y, m, d) && cn_id_valid(&id) { + findings.insert(id); + } + } +} + +/// My Number: weights 2–7 then 2–6 from the right; a remainder of 0 or 1 +/// gives 0, else 11 minus it. +pub fn my_number_valid(n: &str) -> bool { + let d = digit_values(n); + let sum: u32 = (1..=11) + .map(|i| d[11 - i] * if i <= 6 { i as u32 + 1 } else { i as u32 - 5 }) + .sum(); + let check = match sum % 11 { + 0 | 1 => 0, + r => 11 - r, + }; + check == d[11] +} + +const MY_NUMBER_WORDS: &[&str] = &[ + "my number", + "mynumber", + "マイナンバー", + "個人番号", + "kojin bango", +]; + +fn jp_my_number(text: &str, findings: &mut Findings) { + for c in TWELVE.captures_iter(text) { + let whole = c.get(0).unwrap(); + let n = format!("{}{}{}", &c[1], &c[3], &c[5]); + let written = &c[2] == " " && &c[4] == " "; + if my_number_valid(&n) + && (written || word_near(text, whole.start(), whole.end(), MY_NUMBER_WORDS)) + { + findings.insert(n); + } + } +} + +static NRIC: LazyLock = LazyLock::new(|| re(r"(?i)\b([STFGM])(\d{7})([A-Z])\b")); + +/// Weights 2, 7, 6, 5, 4, 3, 2; T and G add 4, M adds 3; each series has its +/// own table of check letters. +fn nric_valid(prefix: u8, digits: &str, check: u8) -> bool { + let sum: u32 = digit_values(digits) + .iter() + .zip([2, 7, 6, 5, 4, 3, 2]) + .map(|(a, w)| a * w) + .sum::() + + match prefix { + b'T' | b'G' => 4, + b'M' => 3, + _ => 0, + }; + let table: &[u8] = match prefix { + b'S' | b'T' => b"JZIHGFEDCBA", + b'F' | b'G' => b"XWUTRQPNMLK", + _ => b"KLJNPQRTUWX", + }; + table[(sum % 11) as usize] == check +} + +fn sg_nric(text: &str, findings: &mut Findings) { + for c in NRIC.captures_iter(text) { + let id = c[0].to_ascii_uppercase(); + let bytes = id.as_bytes(); + if nric_valid(bytes[0], &c[2], bytes[8]) { + findings.insert(id); + } + } +} + +/// `YYMMDD-GNNNNNN`, the seventh digit giving sex and century. +static RRN: LazyLock = LazyLock::new(|| re(r"\b(\d{2})(\d{2})(\d{2})-?([1-8])\d{6}\b")); + +const RRN_WORDS: &[&str] = &["주민등록번호", "주민번호", "resident registration", "rrn"]; + +fn kr_rrn(text: &str, findings: &mut Findings) { + for c in RRN.captures_iter(text) { + let whole = c.get(0).unwrap(); + if valid_short_date(num(&c[1]), num(&c[2]), num(&c[3])) + && word_near(text, whole.start(), whole.end(), RRN_WORDS) + { + findings.insert(whole.as_str().replace('-', "")); + } + } +} + +fn num(s: &str) -> u32 { + s.parse().unwrap_or(0) +} + +#[cfg(test)] +mod tests { + use crate::mailflow::detectors::by_id; + + fn count(id: &str, text: &str) -> usize { + by_id(id).unwrap().count(text) + } + + #[test] + fn india() { + assert_eq!(count("in-aadhaar", "2345 6789 0124"), 1); + assert_eq!(count("in-aadhaar", "2345 6789 0125"), 0); + assert_eq!(count("in-aadhaar", "order 234567890124"), 0); + assert_eq!(count("in-aadhaar", "Aadhaar 234567890124"), 1); + assert_eq!(count("in-pan", "PAN: ABCPE1234F"), 1); + assert_eq!(count("in-pan", "ABCPE1234F"), 0); + } + + #[test] + fn china_japan() { + assert_eq!(count("cn-resident-id", "11010519491231002X"), 1); + assert_eq!(count("cn-resident-id", "110105194912310021"), 0); + assert_eq!(count("cn-resident-id", "11010519491331002X"), 0); + assert_eq!(count("jp-my-number", "1234 5678 9018"), 1); + assert_eq!(count("jp-my-number", "1234 5678 9017"), 0); + assert_eq!(count("jp-my-number", "マイナンバー 123456789018"), 1); + } + + #[test] + fn singapore_korea() { + assert_eq!(count("sg-nric", "S1234567D and T1234567J"), 2); + assert_eq!(count("sg-nric", "S1234567E"), 0); + assert_eq!(count("kr-rrn", "주민등록번호 800101-1234567"), 1); + assert_eq!(count("kr-rrn", "800101-1234567"), 0); + assert_eq!(count("kr-rrn", "RRN 801301-1234567"), 0); + } +} diff --git a/crates/features/src/mailflow/detectors/australia.rs b/crates/features/src/mailflow/detectors/australia.rs new file mode 100644 index 0000000..15812bd --- /dev/null +++ b/crates/features/src/mailflow/detectors/australia.rs @@ -0,0 +1,128 @@ +/* + * SPDX-FileCopyrightText: 2026 Coffey Labs + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +//! Australian identifiers (§2.3): the ATO's Tax File Number and the Medicare +//! card number. + +use super::{Detector, Findings, Region, Strength, word_near}; +use regex::Regex; +use std::sync::LazyLock; + +pub static DETECTORS: &[Detector] = &[ + Detector::new( + "au-tfn", + "Australian Tax File Number", + Region::Australia, + Strength::Checked, + tfn, + ), + Detector::new( + "au-medicare", + "Australian Medicare number", + Region::Australia, + Strength::Checked, + medicare, + ), +]; + +fn re(pattern: &str) -> Regex { + Regex::new(pattern).expect("detector pattern") +} + +/// `NNN NNN NNN` stands alone; bare digits (eight or nine) need a word. +static TFN: LazyLock = LazyLock::new(|| re(r"\b(\d{3})( ?)(\d{3})( ?)(\d{2,3})\b")); + +/// Weighted sum mod 11, with the ATO's weights for 9- and 8-digit numbers. +pub fn tfn_valid(n: &str) -> bool { + let weights: &[u32] = match n.len() { + 9 => &[1, 4, 3, 7, 5, 8, 6, 9, 10], + 8 => &[10, 7, 8, 4, 6, 3, 5, 1], + _ => return false, + }; + n.bytes() + .zip(weights) + .map(|(b, w)| u32::from(b - b'0') * w) + .sum::() + % 11 + == 0 +} + +const TFN_WORDS: &[&str] = &["tfn", "tax file number", "tax file no"]; + +fn tfn(text: &str, findings: &mut Findings) { + for c in TFN.captures_iter(text) { + let whole = c.get(0).unwrap(); + let n = format!("{}{}{}", &c[1], &c[3], &c[5]); + let written = n.len() == 9 && c[2] == *" " && c[4] == *" "; + if tfn_valid(&n) && (written || word_near(text, whole.start(), whole.end(), TFN_WORDS)) { + findings.insert(n); + } + } +} + +/// `NNNN NNNNN N` (and an optional issue number) stands alone; bare digits +/// need a word. +static MEDICARE: LazyLock = + LazyLock::new(|| re(r"\b([2-6]\d{3})( ?)(\d{5})( ?)(\d)(?:[ -]?\d)?\b")); + +/// The ninth digit is the weighted sum (1, 3, 7, 9, 1, 3, 7, 9) of the first +/// eight, mod 10. +pub fn medicare_valid(n: &str) -> bool { + let d: Vec = n.bytes().map(|b| u32::from(b - b'0')).collect(); + d.len() >= 9 + && d[..8] + .iter() + .zip([1, 3, 7, 9, 1, 3, 7, 9]) + .map(|(a, w)| a * w) + .sum::() + % 10 + == d[8] +} + +const MEDICARE_WORDS: &[&str] = &[ + "medicare", + "medicare card", + "medicare no", + "medicare number", +]; + +fn medicare(text: &str, findings: &mut Findings) { + for c in MEDICARE.captures_iter(text) { + let whole = c.get(0).unwrap(); + let n = format!("{}{}{}", &c[1], &c[3], &c[5]); + let written = c[2] == *" " && c[4] == *" "; + if medicare_valid(&n) + && (written || word_near(text, whole.start(), whole.end(), MEDICARE_WORDS)) + { + findings.insert(n); + } + } +} + +#[cfg(test)] +mod tests { + use crate::mailflow::detectors::by_id; + + fn count(id: &str, text: &str) -> usize { + by_id(id).unwrap().count(text) + } + + #[test] + fn tax_file_numbers() { + assert_eq!(count("au-tfn", "TFN 123 456 782"), 1); + assert_eq!(count("au-tfn", "123 456 789"), 0); + assert_eq!(count("au-tfn", "order 123456782"), 0); + assert_eq!(count("au-tfn", "tax file number 123456782"), 1); + } + + #[test] + fn medicare_numbers() { + assert_eq!(count("au-medicare", "2123 45670 1"), 1); + assert_eq!(count("au-medicare", "2123 45671 1"), 0); + assert_eq!(count("au-medicare", "ref 2123456701"), 0); + assert_eq!(count("au-medicare", "Medicare 2123456701"), 1); + } +} diff --git a/crates/features/src/mailflow/detectors/canada.rs b/crates/features/src/mailflow/detectors/canada.rs new file mode 100644 index 0000000..e2fc11d --- /dev/null +++ b/crates/features/src/mailflow/detectors/canada.rs @@ -0,0 +1,66 @@ +/* + * SPDX-FileCopyrightText: 2026 Coffey Labs + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +//! Canadian identifiers (§2.3): the Social Insurance Number. + +use super::{Detector, Findings, Region, Strength, checks, word_near}; +use regex::Regex; +use std::sync::LazyLock; + +pub static DETECTORS: &[Detector] = &[Detector::new( + "ca-sin", + "Canadian Social Insurance Number", + Region::Canada, + Strength::Checked, + sin, +)]; + +/// `NNN NNN NNN` or `NNN-NNN-NNN` stands alone; nine bare digits need a word. +static SIN: LazyLock = LazyLock::new(|| { + Regex::new(r"\b(\d{3})([ -]?)(\d{3})([ -]?)(\d{3})\b").expect("detector pattern") +}); + +const SIN_WORDS: &[&str] = &[ + "sin", + "social insurance", + "nas", + "numéro d'assurance sociale", + "assurance sociale", +]; + +fn sin(text: &str, findings: &mut Findings) { + for c in SIN.captures_iter(text) { + let whole = c.get(0).unwrap(); + let n = format!("{}{}{}", &c[1], &c[3], &c[5]); + let written = !c[2].is_empty() && c[2] == c[4]; + // 0 and 8 are never issued as a first digit + if !n.starts_with(['0', '8']) + && checks::luhn(&n) + && (written || word_near(text, whole.start(), whole.end(), SIN_WORDS)) + { + findings.insert(n); + } + } +} + +#[cfg(test)] +mod tests { + use crate::mailflow::detectors::by_id; + + fn count(text: &str) -> usize { + by_id("ca-sin").unwrap().count(text) + } + + #[test] + fn social_insurance_numbers() { + assert_eq!(count("130 692 544 and 193-456-787"), 2); + assert_eq!(count("130 692 545"), 0); + // The government's printed example starts with 0, never issued + assert_eq!(count("046 454 286"), 0); + assert_eq!(count("order 130692544"), 0); + assert_eq!(count("SIN: 130692544"), 1); + } +} diff --git a/crates/features/src/mailflow/detectors/checks.rs b/crates/features/src/mailflow/detectors/checks.rs index d0320be..860ac22 100644 --- a/crates/features/src/mailflow/detectors/checks.rs +++ b/crates/features/src/mailflow/detectors/checks.rs @@ -25,7 +25,7 @@ pub fn luhn(digits: &str) -> bool { } }) .sum(); - sum % 10 == 0 + sum.is_multiple_of(10) } /// ISO 13616 IBAN lengths, by country, from the IBAN registry. diff --git a/crates/features/src/mailflow/detectors/eu.rs b/crates/features/src/mailflow/detectors/eu.rs new file mode 100644 index 0000000..72798e8 --- /dev/null +++ b/crates/features/src/mailflow/detectors/eu.rs @@ -0,0 +1,646 @@ +/* + * SPDX-FileCopyrightText: 2026 Coffey Labs + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +//! European Union national identifiers (§2.3), each from its issuer's +//! published rules. An identifier that is only digits and whose check a +//! random number passes often (mod 10, mod 11) counts alone only in its +//! written form, and as bare digits only beside a word. + +use super::{ + Detector, Findings, Region, Strength, checks, digit_values, stands_alone, valid_short_date, + word_near, +}; +use regex::Regex; +use std::sync::LazyLock; + +pub static DETECTORS: &[Detector] = &[ + Detector::new( + "de-tax-id", + "Germany: tax ID (Steuer-ID)", + Region::Eu, + Strength::Checked, + de_tax_id, + ), + Detector::new( + "de-id-card", + "Germany: ID card number", + Region::Eu, + Strength::Checked, + de_id_card, + ), + Detector::new( + "fr-nir", + "France: social security number (NIR)", + Region::Eu, + Strength::Checked, + fr_nir, + ), + Detector::new( + "es-dni-nie", + "Spain: DNI and NIE", + Region::Eu, + Strength::Checked, + es_dni_nie, + ), + Detector::new( + "it-codice-fiscale", + "Italy: codice fiscale", + Region::Eu, + Strength::Checked, + it_codice_fiscale, + ), + Detector::new( + "nl-bsn", + "Netherlands: BSN", + Region::Eu, + Strength::Checked, + nl_bsn, + ), + Detector::new( + "be-national-number", + "Belgium: national number", + Region::Eu, + Strength::Checked, + be_national_number, + ), + Detector::new( + "pl-pesel", + "Poland: PESEL", + Region::Eu, + Strength::Checked, + pl_pesel, + ), + Detector::new( + "se-personnummer", + "Sweden: personnummer", + Region::Eu, + Strength::Checked, + se_personnummer, + ), + Detector::new( + "dk-cpr", + "Denmark: CPR number", + Region::Eu, + Strength::NeedsWord, + dk_cpr, + ), + Detector::new( + "fi-hetu", + "Finland: personal identity code", + Region::Eu, + Strength::Checked, + fi_hetu, + ), + Detector::new( + "ie-pps", + "Ireland: PPS number", + Region::Eu, + Strength::Checked, + ie_pps, + ), + Detector::new( + "pt-nif", + "Portugal: NIF", + Region::Eu, + Strength::Checked, + pt_nif, + ), + Detector::new( + "at-svnr", + "Austria: social insurance number", + Region::Eu, + Strength::Checked, + at_svnr, + ), +]; + +fn re(pattern: &str) -> Regex { + Regex::new(pattern).expect("detector pattern") +} + +fn num(s: &str) -> u32 { + s.parse().unwrap_or(0) +} + +// --- Germany -------------------------------------------------------------- + +/// Eleven digits, written `86 095 742 719` on the BZSt's letters. +static DE_TAX: LazyLock = LazyLock::new(|| re(r"\b\d{2}( ?)\d{3}( ?)\d{3}( ?)\d{3}\b")); + +/// ISO 7064 MOD 11,10; no leading zero; in the first ten digits one digit +/// appears two or three times and every other at most once. +pub fn de_tax_id_valid(n: &str) -> bool { + let d = digit_values(n); + if d.len() != 11 || d[0] == 0 { + return false; + } + let mut counts = [0u8; 10]; + for &x in &d[..10] { + counts[x as usize] += 1; + } + let repeated = counts.iter().filter(|&&c| c >= 2).count(); + if repeated != 1 || counts.iter().any(|&c| c > 3) { + return false; + } + let mut product = 10; + for &x in &d[..10] { + let mut sum = (x + product) % 10; + if sum == 0 { + sum = 10; + } + product = (2 * sum) % 11; + } + let check = match 11 - product { + 10 => 0, + c => c, + }; + check == d[10] +} + +const DE_TAX_WORDS: &[&str] = &[ + "steuer-id", + "steueridentifikationsnummer", + "steuerliche identifikationsnummer", + "idnr", + "identifikationsnummer", + "tax id", +]; + +fn de_tax_id(text: &str, findings: &mut Findings) { + for c in DE_TAX.captures_iter(text) { + let whole = c.get(0).unwrap(); + let written = [&c[1], &c[2], &c[3]].iter().all(|s| *s == " "); + let n: String = whole.as_str().replace(' ', ""); + if de_tax_id_valid(&n) + && (written || word_near(text, whole.start(), whole.end(), DE_TAX_WORDS)) + { + findings.insert(n); + } + } +} + +/// The ID card's document number: a letter from the card's alphabet, eight +/// more characters from it, then the check digit. +static DE_ID: LazyLock = + LazyLock::new(|| re(r"\b[CFGHJKLMNPRTVWXYZ][CFGHJKLMNPRTVWXYZ0-9]{8}\d\b")); + +/// ICAO 9303 check digit: weights 7, 3, 1; letters A=10 … Z=35. +pub fn icao_check(chars: &str, check: u32) -> bool { + let value = |c: char| c.to_digit(10).unwrap_or_else(|| c as u32 - 'A' as u32 + 10); + let sum: u32 = chars + .chars() + .zip([7, 3, 1].iter().cycle()) + .map(|(c, w)| value(c) * w) + .sum(); + sum % 10 == check +} + +fn de_id_card(text: &str, findings: &mut Findings) { + for m in DE_ID.find_iter(text) { + let s = m.as_str(); + if icao_check(&s[..9], num(&s[9..])) { + findings.insert(s); + } + } +} + +// --- France --------------------------------------------------------------- + +/// Sex, year, month, department (with Corsica's 2A and 2B), commune, order, +/// then the two-digit key, spaces allowed between groups. +static FR_NIR: LazyLock = LazyLock::new(|| { + re(r"\b([1-478]) ?(\d{2}) ?(\d{2}) ?(\d{2}|2[AB]) ?(\d{3}) ?(\d{3}) ?(\d{2})\b") +}); + +fn fr_nir(text: &str, findings: &mut Findings) { + for c in FR_NIR.captures_iter(text) { + let month = num(&c[3]); + if !(matches!(month, 1..=12 | 20..=42 | 50..=99)) { + continue; + } + let department = match &c[4] { + "2A" => "19", + "2B" => "18", + d => d, + }; + let body = format!( + "{}{}{}{}{}{}", + &c[1], &c[2], &c[3], department, &c[5], &c[6] + ); + let Ok(value) = body.parse::() else { + continue; + }; + if 97 - value % 97 == u64::from(num(&c[7])) { + findings.insert(format!( + "{}{}{}{}{}{}{}", + &c[1], &c[2], &c[3], &c[4], &c[5], &c[6], &c[7] + )); + } + } +} + +// --- Spain ---------------------------------------------------------------- + +static ES_ID: LazyLock = LazyLock::new(|| re(r"(?i)\b([XYZ]?)[ -]?(\d{7,8})[ -]?([A-Z])\b")); + +const DNI_LETTERS: &[u8] = b"TRWAGMYFPDXBNJZSQVHLCKE"; + +fn es_dni_nie(text: &str, findings: &mut Findings) { + for c in ES_ID.captures_iter(text) { + let prefix = c[1].to_ascii_uppercase(); + let digits = &c[2]; + // DNI: eight digits; NIE: X, Y or Z and seven digits + let number = match (prefix.as_str(), digits.len()) { + ("", 8) => digits.to_string(), + ("X", 7) => format!("0{digits}"), + ("Y", 7) => format!("1{digits}"), + ("Z", 7) => format!("2{digits}"), + _ => continue, + }; + let letter = c[3].to_ascii_uppercase(); + if DNI_LETTERS[(num(&number) % 23) as usize] == letter.as_bytes()[0] { + findings.insert(format!("{prefix}{digits}{letter}")); + } + } +} + +// --- Italy ---------------------------------------------------------------- + +/// Surname and name letters, year, month letter, day, place code, check +/// letter; digits may be replaced by letters (omocodia). +static IT_CF: LazyLock = LazyLock::new(|| { + let d = "[0-9LMNPQRSTUV]"; + re(&format!( + r"(?i)\b[A-Z]{{6}}{d}{{2}}[ABCDEHLMPRST]{d}{{2}}[A-Z]{d}{{3}}[A-Z]\b" + )) +}); + +/// The Ministry's odd-position values for 0–9 and A–Z. +const CF_ODD: [u32; 36] = [ + 1, 0, 5, 7, 9, 13, 15, 17, 19, 21, // 0-9 + 1, 0, 5, 7, 9, 13, 15, 17, 19, 21, 2, 4, 18, 20, 11, 3, 6, 8, 12, 14, 16, 10, 22, 25, 24, + 23, // A-Z +]; + +pub fn codice_fiscale_valid(cf: &str) -> bool { + let index = |c: u8| { + if c.is_ascii_digit() { + (c - b'0') as usize + } else { + (c - b'A') as usize + 10 + } + }; + let even = |c: u8| { + if c.is_ascii_digit() { + u32::from(c - b'0') + } else { + u32::from(c - b'A') + } + }; + let bytes = cf.as_bytes(); + let sum: u32 = bytes[..15] + .iter() + .enumerate() + .map(|(i, &c)| { + if i % 2 == 0 { + CF_ODD[index(c)] + } else { + even(c) + } + }) + .sum(); + u32::from(bytes[15] - b'A') == sum % 26 +} + +fn it_codice_fiscale(text: &str, findings: &mut Findings) { + for m in IT_CF.find_iter(text) { + let cf = m.as_str().to_ascii_uppercase(); + if codice_fiscale_valid(&cf) { + findings.insert(cf); + } + } +} + +// --- Netherlands ---------------------------------------------------------- + +/// Nine digits, sometimes written `1112.22.333`. +static NL_BSN: LazyLock = LazyLock::new(|| re(r"\b(\d{4})(\.?)(\d{2})(\.?)(\d{3})\b")); + +/// The eleven test: weights 9 down to 2, and −1 for the last digit. +pub fn bsn_valid(n: &str) -> bool { + let d = digit_values(n); + let sum: i64 = d[..8] + .iter() + .zip((2..=9).rev()) + .map(|(a, w)| i64::from(a * w)) + .sum::() + - i64::from(d[8]); + sum != 0 && sum % 11 == 0 +} + +const BSN_WORDS: &[&str] = &[ + "bsn", + "burgerservicenummer", + "sofinummer", + "sofi-nummer", + "citizen service number", +]; + +fn nl_bsn(text: &str, findings: &mut Findings) { + for c in NL_BSN.captures_iter(text) { + let whole = c.get(0).unwrap(); + let n = format!("{}{}{}", &c[1], &c[3], &c[5]); + let written = &c[2] == "." && &c[4] == "."; + if bsn_valid(&n) && (written || word_near(text, whole.start(), whole.end(), BSN_WORDS)) { + findings.insert(n); + } + } +} + +// --- Belgium -------------------------------------------------------------- + +/// `YY.MM.DD-XXX.CC` or eleven digits. +static BE_NN: LazyLock = + LazyLock::new(|| re(r"\b(\d{2})\.?(\d{2})\.?(\d{2})-?(\d{3})\.?(\d{2})\b")); + +fn be_national_number(text: &str, findings: &mut Findings) { + for c in BE_NN.captures_iter(text) { + let (month, day) = (num(&c[2]), num(&c[3])); + // Month 0 and day 0 mean unknown; bis numbers add 20 or 40 to the month + if !(month <= 12 || (20..=32).contains(&month) || (40..=52).contains(&month)) || day > 31 { + continue; + } + let body = format!("{}{}{}{}", &c[1], &c[2], &c[3], &c[4]); + let check = u64::from(num(&c[5])); + let before_2000 = 97 - body.parse::().unwrap_or(0) % 97; + let since_2000 = 97 - format!("2{body}").parse::().unwrap_or(0) % 97; + if check == before_2000 || check == since_2000 { + findings.insert(format!("{body}{}", &c[5])); + } + } +} + +// --- Poland --------------------------------------------------------------- + +static ELEVEN: LazyLock = LazyLock::new(|| re(r"\b\d{11}\b")); + +/// Weights 1, 3, 7, 9 repeating; the birth date encodes the century in the +/// month (+80 for the 1800s, +20 for the 2000s, and so on). +pub fn pesel_valid(n: &str) -> bool { + let d = digit_values(n); + let sum: u32 = d[..10] + .iter() + .zip([1, 3, 7, 9].iter().cycle()) + .map(|(a, w)| a * w) + .sum(); + let month = d[2] * 10 + d[3]; + let (century, month) = match month { + 81..=92 => (1800, month - 80), + 1..=12 => (1900, month), + 21..=32 => (2000, month - 20), + 41..=52 => (2100, month - 40), + _ => return false, + }; + let year = century + d[0] * 10 + d[1]; + (10 - sum % 10) % 10 == d[10] && (1..=super::days_in(year, month)).contains(&(d[4] * 10 + d[5])) +} + +const PESEL_WORDS: &[&str] = &["pesel", "numer pesel", "nr pesel"]; + +fn pl_pesel(text: &str, findings: &mut Findings) { + for m in ELEVEN.find_iter(text) { + if pesel_valid(m.as_str()) && word_near(text, m.start(), m.end(), PESEL_WORDS) { + findings.insert(m.as_str()); + } + } +} + +// --- Sweden --------------------------------------------------------------- + +/// `YYMMDD-NNNN`, `YYYYMMDD-NNNN` (`+` after 100), or the bare digits. +static SE_PNR: LazyLock = + LazyLock::new(|| re(r"\b(?:\d{2})?(\d{2})(\d{2})(\d{2})([-+]?)(\d{4})\b")); + +const SE_WORDS: &[&str] = &[ + "personnummer", + "personnr", + "person nr", + "samordningsnummer", + "pnr", +]; + +fn se_personnummer(text: &str, findings: &mut Findings) { + for c in SE_PNR.captures_iter(text) { + let whole = c.get(0).unwrap(); + let (yy, month, day) = (num(&c[1]), num(&c[2]), num(&c[3])); + // Coordination numbers add 60 to the day + let day = if day > 60 { day - 60 } else { day }; + let ten = format!("{}{}{}{}", &c[1], &c[2], &c[3], &c[5]); + let written = !c[4].is_empty(); + if valid_short_date(yy, month, day) + && checks::luhn(&ten) + && (written || word_near(text, whole.start(), whole.end(), SE_WORDS)) + { + findings.insert(ten); + } + } +} + +// --- Denmark -------------------------------------------------------------- + +static DK_CPR: LazyLock = LazyLock::new(|| re(r"\b(\d{2})(\d{2})(\d{2})-?(\d{4})\b")); + +const CPR_WORDS: &[&str] = &["cpr", "cpr-nr", "cpr nr", "cpr-nummer", "personnummer"]; + +fn dk_cpr(text: &str, findings: &mut Findings) { + for c in DK_CPR.captures_iter(text) { + let whole = c.get(0).unwrap(); + if valid_short_date(num(&c[3]), num(&c[2]), num(&c[1])) + && word_near(text, whole.start(), whole.end(), CPR_WORDS) + { + findings.insert(whole.as_str().replace('-', "")); + } + } +} + +// --- Finland -------------------------------------------------------------- + +static FI_HETU: LazyLock = + LazyLock::new(|| re(r"(?i)\b(\d{2})(\d{2})(\d{2})[-+ABCDEFYXWVU](\d{3})([0-9A-Y])\b")); + +const HETU_CHECK: &[u8] = b"0123456789ABCDEFHJKLMNPRSTUVWXY"; + +fn fi_hetu(text: &str, findings: &mut Findings) { + for c in FI_HETU.captures_iter(text) { + let (day, month, yy) = (num(&c[1]), num(&c[2]), num(&c[3])); + let n: u64 = format!("{}{}{}{}", &c[1], &c[2], &c[3], &c[4]) + .parse() + .unwrap_or(0); + let check = c[5].to_ascii_uppercase().as_bytes()[0]; + if valid_short_date(yy, month, day) && HETU_CHECK[(n % 31) as usize] == check { + findings.insert(c[0].to_ascii_uppercase()); + } + } +} + +// --- Ireland -------------------------------------------------------------- + +static IE_PPS: LazyLock = LazyLock::new(|| re(r"(?i)\b(\d{7})([A-W])([ABHW]?)\b")); + +const PPS_CHECK: &[u8] = b"WABCDEFGHIJKLMNOPQRSTUV"; + +fn ie_pps(text: &str, findings: &mut Findings) { + for c in IE_PPS.captures_iter(text) { + let mut sum: u32 = digit_values(&c[1]) + .iter() + .zip((2..=8).rev()) + .map(|(a, w)| a * w) + .sum(); + // The second letter counts, times 9; W (the old form) counts as 0 + let second = c[3].to_ascii_uppercase(); + if let Some(&letter) = second.as_bytes().first() + && letter != b'W' + { + sum += u32::from(letter - b'A' + 1) * 9; + } + let check = c[2].to_ascii_uppercase().as_bytes()[0]; + if PPS_CHECK[(sum % 23) as usize] == check { + findings.insert(c[0].to_ascii_uppercase()); + } + } +} + +// --- Portugal ------------------------------------------------------------- + +static NINE: LazyLock = LazyLock::new(|| re(r"\b\d{9}\b")); + +/// Mod 11 over weights 9 down to 2; a check of 10 or 11 becomes 0. +pub fn nif_valid(n: &str) -> bool { + let d = digit_values(n); + let sum: u32 = d[..8].iter().zip((2..=9).rev()).map(|(a, w)| a * w).sum(); + let check = match 11 - sum % 11 { + 10 | 11 => 0, + c => c, + }; + matches!(d[0], 1 | 2 | 3 | 5 | 6 | 8 | 9) && check == d[8] +} + +const NIF_WORDS: &[&str] = &[ + "nif", + "contribuinte", + "número de identificação fiscal", + "numero de contribuinte", +]; + +fn pt_nif(text: &str, findings: &mut Findings) { + for m in NINE.find_iter(text) { + if nif_valid(m.as_str()) && word_near(text, m.start(), m.end(), NIF_WORDS) { + findings.insert(m.as_str()); + } + } +} + +// --- Austria -------------------------------------------------------------- + +/// A serial and check digit, then the birth date: `1237 010180`. +static AT_SVNR: LazyLock = LazyLock::new(|| re(r"\b(\d{3})(\d)( ?)(\d{2})(\d{2})(\d{2})\b")); + +const SVNR_WORDS: &[&str] = &[ + "sozialversicherungsnummer", + "svnr", + "sv-nr", + "sv-nummer", + "versicherungsnummer", +]; + +fn at_svnr(text: &str, findings: &mut Findings) { + for c in AT_SVNR.captures_iter(text) { + let whole = c.get(0).unwrap(); + let n = format!("{}{}{}{}{}", &c[1], &c[2], &c[4], &c[5], &c[6]); + let d = digit_values(&n); + let sum: u32 = d + .iter() + .zip([3, 7, 9, 0, 5, 8, 4, 2, 1, 6]) + .map(|(a, w)| a * w) + .sum(); + let written = &c[3] == " "; + if d[0] != 0 + && sum % 11 == d[3] + && valid_short_date(num(&c[6]), num(&c[5]), num(&c[4])) + && (written || word_near(text, whole.start(), whole.end(), SVNR_WORDS)) + && stands_alone(text, whole.start(), whole.end()) + { + findings.insert(n); + } + } +} + +#[cfg(test)] +mod tests { + use crate::mailflow::detectors::by_id; + + fn count(id: &str, text: &str) -> usize { + by_id(id).unwrap().count(text) + } + + #[test] + fn germany() { + assert_eq!(count("de-tax-id", "86 095 742 719"), 1); + assert_eq!(count("de-tax-id", "Steuer-ID: 86095742719"), 1); + assert_eq!(count("de-tax-id", "Rechnung 86095742719"), 0); + assert_eq!(count("de-tax-id", "86 095 742 718"), 0); + // ICAO 9303's German specimen card + assert_eq!(count("de-id-card", "Ausweis T220001293"), 1); + assert_eq!(count("de-id-card", "T220001294"), 0); + } + + #[test] + fn france_spain_italy() { + assert_eq!(count("fr-nir", "2 55 08 14 168 025 38"), 1); + assert_eq!(count("fr-nir", "255081416802539"), 0); + assert_eq!(count("es-dni-nie", "DNI 12345678Z, NIE X-1234567-L"), 2); + assert_eq!(count("es-dni-nie", "12345678A"), 0); + assert_eq!(count("it-codice-fiscale", "CF: RSSMRA85T10A562S"), 1); + assert_eq!(count("it-codice-fiscale", "RSSMRA85T10A562T"), 0); + } + + #[test] + fn benelux() { + assert_eq!(count("nl-bsn", "1112.22.333"), 1); + assert_eq!(count("nl-bsn", "BSN 111222333"), 1); + assert_eq!(count("nl-bsn", "order 111222333"), 0); + assert_eq!(count("nl-bsn", "BSN 111222334"), 0); + assert_eq!(count("be-national-number", "85.07.30-033.28"), 1); + assert_eq!(count("be-national-number", "85073003329"), 0); + } + + #[test] + fn nordics() { + assert_eq!(count("se-personnummer", "811218-9876"), 1); + assert_eq!(count("se-personnummer", "811218-9875"), 0); + assert_eq!(count("se-personnummer", "order 8112189876"), 0); + assert_eq!(count("se-personnummer", "personnummer 198112189876"), 1); + assert_eq!(count("dk-cpr", "CPR-nr: 010170-1234"), 1); + assert_eq!(count("dk-cpr", "010170-1234"), 0); + assert_eq!(count("dk-cpr", "CPR 320170-1234"), 0); + assert_eq!(count("fi-hetu", "131052-308T"), 1); + assert_eq!(count("fi-hetu", "131052-308U"), 0); + } + + #[test] + fn poland_ireland_portugal_austria() { + assert_eq!(count("pl-pesel", "PESEL 44051401359, pesel 02070803628"), 2); + assert_eq!(count("pl-pesel", "PESEL 44051401358"), 0); + assert_eq!(count("pl-pesel", "44051401359"), 0); + assert_eq!(count("ie-pps", "PPS 1234567T and 1234567FA"), 2); + assert_eq!(count("ie-pps", "1234567U"), 0); + assert_eq!(count("pt-nif", "NIF 123456789"), 1); + assert_eq!(count("pt-nif", "NIF 123456788"), 0); + assert_eq!(count("at-svnr", "1237 010180"), 1); + assert_eq!(count("at-svnr", "SVNR 1237010180"), 1); + assert_eq!(count("at-svnr", "1238 010180"), 0); + } +} diff --git a/crates/features/src/mailflow/detectors/europe.rs b/crates/features/src/mailflow/detectors/europe.rs new file mode 100644 index 0000000..cb56341 --- /dev/null +++ b/crates/features/src/mailflow/detectors/europe.rs @@ -0,0 +1,116 @@ +/* + * SPDX-FileCopyrightText: 2026 Coffey Labs + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +//! European identifiers outside the EU (§2.3): Norway's national identity +//! number and Switzerland's AHV number. + +use super::{Detector, Findings, Region, Strength, digit_values, valid_short_date}; +use regex::Regex; +use std::sync::LazyLock; + +pub static DETECTORS: &[Detector] = &[ + Detector::new( + "no-fnr", + "Norway: national identity number", + Region::Europe, + Strength::Checked, + no_fnr, + ), + Detector::new( + "ch-ahv", + "Switzerland: AHV number", + Region::Europe, + Strength::Checked, + ch_ahv, + ), +]; + +static ELEVEN: LazyLock = + LazyLock::new(|| Regex::new(r"\b\d{6} ?\d{5}\b").expect("detector pattern")); + +/// Two mod 11 check digits over a birth date (D-numbers add 40 to the day, +/// H-numbers 40 to the month): strong enough to count alone. +pub fn fnr_valid(n: &str) -> bool { + let d = digit_values(n); + if d.len() != 11 { + return false; + } + let check = + |weights: &[u32]| match 11 - d.iter().zip(weights).map(|(a, w)| a * w).sum::() % 11 { + 11 => Some(0), + 10 => None, + c => Some(c), + }; + let day = d[0] * 10 + d[1]; + let month = d[2] * 10 + d[3]; + let day = if day > 40 { day - 40 } else { day }; + let month = if month > 40 { month - 40 } else { month }; + valid_short_date(d[4] * 10 + d[5], month, day) + && check(&[3, 7, 6, 1, 8, 9, 4, 5, 2]) == Some(d[9]) + && check(&[5, 4, 3, 2, 7, 6, 5, 4, 3, 2]) == Some(d[10]) +} + +fn no_fnr(text: &str, findings: &mut Findings) { + for m in ELEVEN.find_iter(text) { + let n = m.as_str().replace(' ', ""); + if fnr_valid(&n) { + findings.insert(n); + } + } +} + +/// `756.1234.5678.97`: the country prefix, then an EAN-13 check digit. +static AHV: LazyLock = LazyLock::new(|| { + Regex::new(r"\b756[. ]?\d{4}[. ]?\d{4}[. ]?\d{2}\b").expect("detector pattern") +}); + +pub fn ean13_valid(n: &str) -> bool { + let d = digit_values(n); + if d.len() != 13 { + return false; + } + let sum: u32 = d[..12] + .iter() + .enumerate() + .map(|(i, x)| if i % 2 == 0 { *x } else { x * 3 }) + .sum(); + (10 - sum % 10) % 10 == d[12] +} + +fn ch_ahv(text: &str, findings: &mut Findings) { + for m in AHV.find_iter(text) { + let n: String = m.as_str().chars().filter(char::is_ascii_digit).collect(); + if ean13_valid(&n) { + findings.insert(n); + } + } +} + +#[cfg(test)] +mod tests { + use crate::mailflow::detectors::by_id; + + fn count(id: &str, text: &str) -> usize { + by_id(id).unwrap().count(text) + } + + #[test] + fn norway() { + assert_eq!(count("no-fnr", "01019000083"), 1); + assert_eq!(count("no-fnr", "010190 00083"), 1); + assert_eq!(count("no-fnr", "01019000084"), 0); + // Not a date + assert_eq!(count("no-fnr", "32019000083"), 0); + } + + #[test] + fn switzerland() { + // The federal example + assert_eq!(count("ch-ahv", "AHV 756.9217.0769.85"), 1); + assert_eq!(count("ch-ahv", "7569217076985"), 1); + assert_eq!(count("ch-ahv", "756.9217.0769.86"), 0); + } +} diff --git a/crates/features/src/mailflow/detectors/mod.rs b/crates/features/src/mailflow/detectors/mod.rs index 66b1eab..9853dd6 100644 --- a/crates/features/src/mailflow/detectors/mod.rs +++ b/crates/features/src/mailflow/detectors/mod.rs @@ -19,8 +19,18 @@ //! same card number pasted twice counts once. They stay in memory: callers //! read only [`Findings::len`]. +pub mod africa; +pub mod americas; pub mod any; +pub mod asia; +pub mod australia; +pub mod canada; pub mod checks; +pub mod eu; +pub mod europe; +pub mod templates; +pub mod uk; +pub mod us; use ahash::AHashSet; @@ -108,7 +118,20 @@ impl Detector { /// Every detector, in the order the console lists them. pub fn all() -> impl Iterator { - any::DETECTORS.iter() + [ + any::DETECTORS, + us::DETECTORS, + uk::DETECTORS, + canada::DETECTORS, + australia::DETECTORS, + eu::DETECTORS, + europe::DETECTORS, + asia::DETECTORS, + americas::DETECTORS, + africa::DETECTORS, + ] + .into_iter() + .flatten() } pub fn by_id(id: &str) -> Option<&'static Detector> { @@ -159,6 +182,38 @@ pub fn stands_alone(text: &str, start: usize, end: usize) -> bool { before.is_none_or(|c| !c.is_alphanumeric()) && after.is_none_or(|c| !c.is_alphanumeric()) } +/// Days in `month` of `year` (0 for a month that doesn't exist). +pub fn days_in(year: u32, month: u32) -> u32 { + match month { + 1 | 3 | 5 | 7 | 8 | 10 | 12 => 31, + 4 | 6 | 9 | 11 => 30, + 2 if year.is_multiple_of(4) && (!year.is_multiple_of(100) || year.is_multiple_of(400)) => { + 29 + } + 2 => 28, + _ => 0, + } +} + +/// Whether `year`-`month`-`day` is a real date between 1900 and 2100. +pub fn valid_date(year: u32, month: u32, day: u32) -> bool { + (1900..=2100).contains(&year) && (1..=days_in(year, month)).contains(&day) +} + +/// Whether a two-digit year, month and day make a real date in either the +/// 1900s or the 2000s. +pub fn valid_short_date(yy: u32, month: u32, day: u32) -> bool { + valid_date(1900 + yy, month, day) || valid_date(2000 + yy, month, day) +} + +/// The value of each digit in `s`. +pub fn digit_values(s: &str) -> Vec { + s.bytes() + .filter(u8::is_ascii_digit) + .map(|b| u32::from(b - b'0')) + .collect() +} + /// The ASCII digits of `s`. pub fn digits(s: &str) -> String { s.chars().filter(char::is_ascii_digit).collect() @@ -188,6 +243,30 @@ mod tests { assert!(word_near(&text, start, start + 8, &["passport"])); } + /// An ordinary business email: order, invoice and tracking numbers, + /// dates, amounts, a street address. Nothing here is an identifier, so + /// no detector may fire, except the contact ones on the signature. + #[test] + fn ordinary_mail_finds_nothing() { + let text = "Hi Dana,\n\nThanks for order 4471-2290 placed 2026-09-14. Invoice INV-2026-00917 \ + for $12,480.00 is due 10/31/2026; PO 7731902 covers lines 1-14. Tracking \ + 1Z999AA10123456784, parcel 3 of 5, 12.5 kg, box 40x30x20 cm. Meeting moved to \ + Tuesday 9:30-10:15 in room 2B, building 1177. Ticket #5520318, case 20260914-0042. \ + Version 2026.9.28.4, build 118822, commit 5a73a118. Serial SN-88213-X. \ + Ship to 1600 Amphitheatre Pkwy, Mountain View, CA 94043. Revenue grew 18% to \ + 1,204,332 units; see figures 3.1-3.4 and table 12.\n\nBest,\nSam\n\ + Sam Rivera | +1 (415) 555-2671 | sam@example.com"; + let quiet = ["email-addresses", "phone-numbers"]; + for detector in all().filter(|d| !quiet.contains(&d.id)) { + assert_eq!( + detector.count(text), + 0, + "{} fired on ordinary mail", + detector.id + ); + } + } + #[test] fn ids_are_unique() { let mut seen = AHashSet::new(); diff --git a/crates/features/src/mailflow/detectors/templates.rs b/crates/features/src/mailflow/detectors/templates.rs new file mode 100644 index 0000000..c4fd717 --- /dev/null +++ b/crates/features/src/mailflow/detectors/templates.rs @@ -0,0 +1,97 @@ +/* + * SPDX-FileCopyrightText: 2026 Coffey Labs + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +//! Templates (§2.3): named sets of detectors, so a policy doesn't pick forty +//! one at a time. Each is named for what it finds, never for a law, and is a +//! starting point: once added to a rule, its detectors can be changed. + +pub struct Template { + pub id: &'static str, + pub name: &'static str, + pub detectors: &'static [&'static str], +} + +pub static TEMPLATES: &[Template] = &[ + Template { + id: "payment-and-bank", + name: "Payment cards and bank accounts", + detectors: &["payment-card", "iban", "swift-bic", "us-aba-routing"], + }, + Template { + id: "us-personal", + name: "US personal identifiers", + detectors: &[ + "us-ssn", + "us-itin", + "us-ein", + "us-drivers-license", + "passport", + "date-of-birth", + ], + }, + Template { + id: "uk-personal", + name: "UK personal identifiers", + detectors: &["uk-nino", "uk-utr", "uk-nhs", "passport", "date-of-birth"], + }, + Template { + id: "eu-national", + name: "EU national identifiers", + detectors: &[ + "de-tax-id", + "de-id-card", + "fr-nir", + "es-dni-nie", + "it-codice-fiscale", + "nl-bsn", + "be-national-number", + "pl-pesel", + "se-personnummer", + "dk-cpr", + "fi-hetu", + "ie-pps", + "pt-nif", + "at-svnr", + ], + }, + Template { + id: "health", + name: "Health identifiers", + detectors: &["uk-nhs", "us-mbi", "us-npi", "us-dea", "au-medicare"], + }, + Template { + id: "credentials", + name: "Credentials and keys", + detectors: &["private-key", "credentials"], + }, + Template { + id: "contact-lists", + name: "Contact lists", + detectors: &["email-addresses", "phone-numbers"], + }, +]; + +pub fn by_id(id: &str) -> Option<&'static Template> { + TEMPLATES.iter().find(|template| template.id == id) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn every_template_names_real_detectors() { + for template in TEMPLATES { + for id in template.detectors { + assert!( + super::super::by_id(id).is_some(), + "{}: no detector {id}", + template.id + ); + } + } + } +} diff --git a/crates/features/src/mailflow/detectors/uk.rs b/crates/features/src/mailflow/detectors/uk.rs new file mode 100644 index 0000000..ca4c7d3 --- /dev/null +++ b/crates/features/src/mailflow/detectors/uk.rs @@ -0,0 +1,155 @@ +/* + * SPDX-FileCopyrightText: 2026 Coffey Labs + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +//! United Kingdom identifiers (§2.3): HMRC's National Insurance number and +//! Unique Taxpayer Reference, and the NHS number. + +use super::{Detector, Findings, Region, Strength, word_near}; +use regex::Regex; +use std::sync::LazyLock; + +pub static DETECTORS: &[Detector] = &[ + Detector::new( + "uk-nino", + "UK National Insurance number", + Region::Uk, + Strength::Checked, + nino, + ), + Detector::new( + "uk-nhs", + "UK NHS number", + Region::Uk, + Strength::Checked, + nhs, + ), + Detector::new( + "uk-utr", + "UK Unique Taxpayer Reference", + Region::Uk, + Strength::NeedsWord, + utr, + ), +]; + +fn re(pattern: &str) -> Regex { + Regex::new(pattern).expect("detector pattern") +} + +/// Two letters, six digits (often in pairs), a suffix A–D. +static NINO: LazyLock = + LazyLock::new(|| re(r"(?i)\b([A-Z])([A-Z]) ?(\d{2}) ?(\d{2}) ?(\d{2}) ?([A-D])\b")); + +/// HMRC's rules: D, F, I, Q, U and V are never used; O never second; and +/// BG, GB, KN, NK, NT, TN and ZZ are never allocated. +fn nino_prefix(first: char, second: char) -> bool { + const NEVER: &str = "DFIQUV"; + let pair: String = [first, second].iter().collect(); + !NEVER.contains(first) + && !NEVER.contains(second) + && second != 'O' + && !["BG", "GB", "KN", "NK", "NT", "TN", "ZZ"].contains(&pair.as_str()) +} + +fn nino(text: &str, findings: &mut Findings) { + for c in NINO.captures_iter(text) { + let first = c[1].to_ascii_uppercase().chars().next().unwrap(); + let second = c[2].to_ascii_uppercase().chars().next().unwrap(); + if nino_prefix(first, second) { + findings.insert(format!( + "{first}{second}{}{}{}{}", + &c[3], + &c[4], + &c[5], + c[6].to_ascii_uppercase() + )); + } + } +} + +/// `NNN NNN NNNN` stands alone; ten bare digits need a word. +static NHS: LazyLock = LazyLock::new(|| re(r"\b(\d{3})([ -]?)(\d{3})([ -]?)(\d{4})\b")); + +/// Mod 11: weights 10 down to 2 over the first nine digits; the check digit +/// is 11 minus the remainder (11 becomes 0; 10 is never issued). +pub fn nhs_valid(n: &str) -> bool { + let d: Vec = n.bytes().map(|b| u32::from(b - b'0')).collect(); + let sum: u32 = d[..9].iter().zip((2..=10).rev()).map(|(a, w)| a * w).sum(); + match 11 - sum % 11 { + 11 => d[9] == 0, + 10 => false, + check => d[9] == check, + } +} + +const NHS_WORDS: &[&str] = &["nhs", "nhs number", "nhs no"]; + +fn nhs(text: &str, findings: &mut Findings) { + for c in NHS.captures_iter(text) { + let whole = c.get(0).unwrap(); + let n = format!("{}{}{}", &c[1], &c[3], &c[5]); + let written = !c[2].is_empty() && c[2] == c[4]; + if nhs_valid(&n) && (written || word_near(text, whole.start(), whole.end(), NHS_WORDS)) { + findings.insert(n); + } + } +} + +static UTR: LazyLock = LazyLock::new(|| re(r"\b\d{5} ?\d{5}\b")); + +const UTR_WORDS: &[&str] = &[ + "utr", + "unique taxpayer reference", + "tax reference", + "self assessment", +]; + +fn utr(text: &str, findings: &mut Findings) { + for m in UTR.find_iter(text) { + if word_near(text, m.start(), m.end(), UTR_WORDS) { + findings.insert(m.as_str().replace(' ', "")); + } + } +} + +#[cfg(test)] +mod tests { + use crate::mailflow::detectors::by_id; + + fn count(id: &str, text: &str) -> usize { + by_id(id).unwrap().count(text) + } + + #[test] + fn national_insurance() { + assert_eq!(count("uk-nino", "NI: AB 12 34 56 C, ce123456d"), 2); + // Letters never used, pairs never allocated, a suffix past D + for bad in [ + "QQ123456C", + "AO123456C", + "GB123456A", + "AB123456E", + "DA123456A", + ] { + assert_eq!(count("uk-nino", bad), 0, "{bad}"); + } + } + + #[test] + fn nhs_numbers() { + // The NHS's own example + assert_eq!(count("uk-nhs", "943 476 5919"), 1); + assert_eq!(count("uk-nhs", "943 476 5918"), 0); + assert_eq!(count("uk-nhs", "order 9434765919"), 0); + assert_eq!(count("uk-nhs", "NHS number 9434765919"), 1); + } + + #[test] + fn utr() { + assert_eq!(count("uk-utr", "UTR 12345 67890"), 1); + assert_eq!(count("uk-utr", "order 1234567890"), 0); + } +} diff --git a/crates/features/src/mailflow/detectors/us.rs b/crates/features/src/mailflow/detectors/us.rs new file mode 100644 index 0000000..21627e9 --- /dev/null +++ b/crates/features/src/mailflow/detectors/us.rs @@ -0,0 +1,304 @@ +/* + * SPDX-FileCopyrightText: 2026 Coffey Labs + * + * SPDX-License-Identifier: AGPL-3.0-only + */ + +//! United States identifiers (§2.3), each from its issuer's published rules: +//! the SSA (SSN), the IRS (ITIN, EIN), the ABA (routing numbers), CMS (MBI, +//! NPI) and the DEA. + +use super::{Detector, Findings, Region, Strength, checks, digits, stands_alone, word_near}; +use regex::Regex; +use std::sync::LazyLock; + +pub static DETECTORS: &[Detector] = &[ + Detector::new( + "us-ssn", + "US Social Security number", + Region::Us, + Strength::Checked, + ssn, + ), + Detector::new("us-itin", "US ITIN", Region::Us, Strength::Checked, itin), + Detector::new("us-ein", "US EIN", Region::Us, Strength::NeedsWord, ein), + Detector::new( + "us-aba-routing", + "US bank routing number", + Region::Us, + Strength::NeedsWord, + aba_routing, + ), + Detector::new( + "us-drivers-license", + "US driver's license", + Region::Us, + Strength::NeedsWord, + drivers_license, + ), + Detector::new( + "us-mbi", + "US Medicare Beneficiary Identifier", + Region::Us, + Strength::Checked, + mbi, + ), + Detector::new( + "us-npi", + "US National Provider Identifier", + Region::Us, + Strength::NeedsWord, + npi, + ), + Detector::new( + "us-dea", + "US DEA registration number", + Region::Us, + Strength::Checked, + dea, + ), +]; + +fn re(pattern: &str) -> Regex { + Regex::new(pattern).expect("detector pattern") +} + +/// `AAA-GG-SSSS` (dashes or spaces), or nine bare digits. +static NINE: LazyLock = LazyLock::new(|| re(r"\b(\d{3})([ -]?)(\d{2})([ -]?)(\d{4})\b")); + +/// Numbers the SSA has published as never valid: widely printed examples. +const SSN_EXAMPLES: &[&str] = &["078051120", "219099999"]; + +fn ssn_rules(area: u32, group: u32, serial: u32) -> bool { + area != 0 && area != 666 && area < 900 && group != 0 && serial != 0 +} + +const SSN_WORDS: &[&str] = &["ssn", "social security", "soc sec", "ss#", "ss no"]; + +fn ssn(text: &str, findings: &mut Findings) { + for c in NINE.captures_iter(text) { + let whole = c.get(0).unwrap(); + let (area, group, serial) = (num(&c[1]), num(&c[3]), num(&c[5])); + let number = format!("{}{}{}", &c[1], &c[3], &c[5]); + // Written form (with both separators, the same one) stands alone; + // nine bare digits need a word + let written = !c[2].is_empty() && c[2] == c[4]; + if ssn_rules(area, group, serial) + && !SSN_EXAMPLES.contains(&number.as_str()) + && (written || word_near(text, whole.start(), whole.end(), SSN_WORDS)) + { + findings.insert(number); + } + } +} + +/// ITINs: 9XX, then a group in the IRS's ranges. +fn itin_group(group: u32) -> bool { + matches!(group, 50..=65 | 70..=88 | 90..=92 | 94..=99) +} + +const ITIN_WORDS: &[&str] = &["itin", "taxpayer identification", "tax id"]; + +fn itin(text: &str, findings: &mut Findings) { + for c in NINE.captures_iter(text) { + let whole = c.get(0).unwrap(); + let written = !c[2].is_empty() && c[2] == c[4]; + if c[1].starts_with('9') + && itin_group(num(&c[3])) + && (written || word_near(text, whole.start(), whole.end(), ITIN_WORDS)) + { + findings.insert(format!("{}{}{}", &c[1], &c[3], &c[5])); + } + } +} + +static EIN: LazyLock = LazyLock::new(|| re(r"\b(\d{2})-?(\d{7})\b")); + +/// The prefixes the IRS assigns to its campuses and internet EINs. +fn ein_prefix(prefix: u32) -> bool { + matches!(prefix, 1..=6 | 10..=16 | 20..=27 | 30..=48 | 50..=68 | 71..=77 | 80..=88 | 90..=95 | 98 | 99) +} + +const EIN_WORDS: &[&str] = &[ + "ein", + "fein", + "employer identification", + "tax id", + "tin", + "federal tax", +]; + +fn ein(text: &str, findings: &mut Findings) { + for c in EIN.captures_iter(text) { + let whole = c.get(0).unwrap(); + if ein_prefix(num(&c[1])) && word_near(text, whole.start(), whole.end(), EIN_WORDS) { + findings.insert(format!("{}{}", &c[1], &c[2])); + } + } +} + +static ROUTING: LazyLock = LazyLock::new(|| re(r"\b\d{9}\b")); + +/// The ABA check: 3, 7 and 1 weights, mod 10; and a Federal Reserve prefix. +pub fn aba_valid(n: &str) -> bool { + let d: Vec = n.bytes().map(|b| u32::from(b - b'0')).collect(); + let prefix = d[0] * 10 + d[1]; + matches!(prefix, 0..=12 | 21..=32 | 61..=72 | 80) + && (3 * (d[0] + d[3] + d[6]) + 7 * (d[1] + d[4] + d[7]) + (d[2] + d[5] + d[8])) + .is_multiple_of(10) +} + +const ROUTING_WORDS: &[&str] = &["routing", "aba", "rtn", "routing number", "transit"]; + +fn aba_routing(text: &str, findings: &mut Findings) { + for m in ROUTING.find_iter(text) { + // One random number in ten passes the check: always needs a word + if aba_valid(m.as_str()) && word_near(text, m.start(), m.end(), ROUTING_WORDS) { + findings.insert(m.as_str()); + } + } +} + +/// The shapes states issue: up to two letters, then 5–14 digits, dashes +/// allowed (Florida and Illinois print them). +static LICENSE: LazyLock = LazyLock::new(|| re(r"\b[A-Z]{0,2}\d[\d-]{3,16}\d\b")); + +const LICENSE_WORDS: &[&str] = &[ + "driver's license", + "drivers license", + "driver license", + "driver's licence", + "dl", + "dl#", + "license number", + "lic no", + "dmv", +]; + +fn drivers_license(text: &str, findings: &mut Findings) { + for m in LICENSE.find_iter(text) { + let n = digits(m.as_str()); + if (5..=14).contains(&n.len()) && word_near(text, m.start(), m.end(), LICENSE_WORDS) { + findings.insert(m.as_str().replace('-', "")); + } + } +} + +/// CMS's MBI: 11 characters in a fixed pattern of digits, letters and +/// either, the letters S, L, O, I, B and Z never used; dashes may follow the +/// 4th and 7th. +static MBI: LazyLock = LazyLock::new(|| { + let c = "[AC-HJKMNP-RT-Y]"; + let an = "[AC-HJKMNP-RT-Y0-9]"; + re(&format!( + r"\b[1-9]{c}{an}[0-9]-?{c}{an}[0-9]-?{c}{c}[0-9][0-9]\b" + )) +}); + +fn mbi(text: &str, findings: &mut Findings) { + for m in MBI.find_iter(text) { + findings.insert(m.as_str().replace('-', "")); + } +} + +static TEN: LazyLock = LazyLock::new(|| re(r"\b[12]\d{9}\b")); + +const NPI_WORDS: &[&str] = &["npi", "national provider", "provider id", "provider number"]; + +/// NPI: Luhn over the ISO card-issuer prefix 80840 and the number. +fn npi(text: &str, findings: &mut Findings) { + for m in TEN.find_iter(text) { + if checks::luhn(&format!("80840{}", m.as_str())) + && word_near(text, m.start(), m.end(), NPI_WORDS) + { + findings.insert(m.as_str()); + } + } +} + +static DEA: LazyLock = LazyLock::new(|| re(r"\b([ABCDEFGHJKLMPRSTUX][A-Z9])(\d{7})\b")); + +/// DEA: (1st + 3rd + 5th) + 2 × (2nd + 4th + 6th) ends in the 7th digit. +fn dea(text: &str, findings: &mut Findings) { + for c in DEA.captures_iter(text) { + let d: Vec = c[2].bytes().map(|b| u32::from(b - b'0')).collect(); + if ((d[0] + d[2] + d[4]) + 2 * (d[1] + d[3] + d[5])) % 10 == d[6] { + let whole = c.get(0).unwrap(); + if stands_alone(text, whole.start(), whole.end()) { + findings.insert(whole.as_str()); + } + } + } +} + +fn num(s: &str) -> u32 { + s.parse().unwrap_or(0) +} + +#[cfg(test)] +mod tests { + use crate::mailflow::detectors::by_id; + + fn count(id: &str, text: &str) -> usize { + by_id(id).unwrap().count(text) + } + + #[test] + fn ssn() { + assert_eq!(count("us-ssn", "SSN 536-22-1234, also 536 22 1235"), 2); + // Bare digits: only with a word + assert_eq!(count("us-ssn", "ref 536221234"), 0); + assert_eq!(count("us-ssn", "social security: 536221234"), 1); + // Never issued, the SSA's printed examples, mixed separators + for bad in [ + "000-12-3456", + "666-12-3456", + "912-12-3456", + "123-00-4567", + "123-45-0000", + "078-05-1120", + "536-22 1234", + ] { + assert_eq!(count("us-ssn", bad), 0, "{bad}"); + } + } + + #[test] + fn itin_and_ein() { + assert_eq!(count("us-itin", "912-70-1234"), 1); + assert_eq!(count("us-itin", "912-69-1234"), 0); + assert_eq!(count("us-ssn", "912-70-1234"), 0); + assert_eq!(count("us-ein", "EIN: 12-3456789"), 1); + assert_eq!(count("us-ein", "part 12-3456789"), 0); + assert_eq!(count("us-ein", "EIN 07-3456789"), 0); + } + + #[test] + fn routing_needs_a_word() { + assert_eq!(count("us-aba-routing", "Routing number 011000015"), 1); + assert_eq!(count("us-aba-routing", "ABA 021000021"), 1); + assert_eq!(count("us-aba-routing", "invoice 011000015"), 0); + assert_eq!(count("us-aba-routing", "routing 011000016"), 0); + } + + #[test] + fn licenses() { + assert_eq!(count("us-drivers-license", "Driver's license: D1234567"), 1); + assert_eq!(count("us-drivers-license", "DL# S123-456-78-901-0"), 1); + assert_eq!(count("us-drivers-license", "Order D1234567"), 0); + } + + #[test] + fn health_identifiers() { + // CMS's own MBI example + assert_eq!(count("us-mbi", "Medicare 1EG4-TE5-MK73"), 1); + assert_eq!(count("us-mbi", "1EG4TE5MK73"), 1); + assert_eq!(count("us-mbi", "1EG4-TE5-MK7S"), 0); + // CMS's NPI example + assert_eq!(count("us-npi", "NPI 1234567893"), 1); + assert_eq!(count("us-npi", "NPI 1234567894"), 0); + assert_eq!(count("us-npi", "call 1234567893"), 0); + assert_eq!(count("us-dea", "DEA AB1234563"), 1); + assert_eq!(count("us-dea", "AB1234564"), 0); + } +} diff --git a/crates/features/src/mailflow/extract.rs b/crates/features/src/mailflow/extract.rs index ff233ff..761b01b 100644 --- a/crates/features/src/mailflow/extract.rs +++ b/crates/features/src/mailflow/extract.rs @@ -161,12 +161,14 @@ fn is_text(content_type: &str, extension: &str) -> bool { fn decode_text(data: &[u8]) -> String { let utf16 = |bytes: &[u8], big: bool| { let units: Vec = bytes - .chunks_exact(2) - .map(|c| { + .as_chunks::<2>() + .0 + .iter() + .map(|&c| { if big { - u16::from_be_bytes([c[0], c[1]]) + u16::from_be_bytes(c) } else { - u16::from_le_bytes([c[0], c[1]]) + u16::from_le_bytes(c) } }) .collect(); diff --git a/docs/spec/features/dlp-and-mail-flow-rules.md b/docs/spec/features/dlp-and-mail-flow-rules.md index 7a74e38..ad13146 100644 --- a/docs/spec/features/dlp-and-mail-flow-rules.md +++ b/docs/spec/features/dlp-and-mail-flow-rules.md @@ -139,6 +139,14 @@ DLP adds **detectors**. Each counts what it finds, and a rule sets a minimum either side, in the languages where the identifier is used ("passport", "Reisepass", "pasaporte"...). +A check that about one random number in ten passes (Luhn, mod 10, mod 11) is +too weak for a bare run of digits: invoice and phone numbers would match. So +a checked identifier that is only digits (SSN, SIN, NHS, TFN, Medicare…) +counts alone in the written form it's issued in (`536-22-1234`, +`130 692 544`, `943 476 5919`), and as bare digits only beside a word. ABA +routing numbers and NPIs are never written with separators, so they always +need a word. (Refinement made while building phase 2, 2026-09-28.) + The catalog (settled answer 6: the recognized, protected identifiers, not a chosen few). Each row is one table entry and one check function in `crates/features/src/mailflow/detectors/`: @@ -157,10 +165,10 @@ chosen few). Each row is one table entry and one check function in | US | Social Security number | Checked | `AAA-GG-SSSS`, or nine digits with a word; never area 000, 666 or 9xx, group 00, serial 0000 | | US | ITIN | Checked | 9XX-GG-SSSS with the IRS's group ranges | | US | EIN | Needs a word | a valid IRS prefix and seven digits | -| US | Bank routing number (ABA) | Checked | nine digits, a valid Federal Reserve prefix, the 3-7-1 checksum | +| US | Bank routing number (ABA) | Needs a word | nine digits, a valid Federal Reserve prefix, the 3-7-1 checksum | | US | Driver's license | Needs a word | each state's published format | | US | Medicare Beneficiary Identifier | Checked | CMS's 11-character pattern and excluded letters | -| US | National Provider Identifier | Checked | ten digits, Luhn over the `80840` prefix | +| US | National Provider Identifier | Needs a word | ten digits, Luhn over the `80840` prefix | | US | DEA registration number | Checked | two letters, seven digits, DEA's check digit | | UK | National Insurance number | Checked | two letters (HMRC's excluded prefixes), six digits, A–D | | UK | NHS number | Checked | ten digits, mod 11 |