DLP: regional identifiers and templates #102
@@ -0,0 +1,49 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! African identifiers (§2.3): South Africa's ID number.
|
||||
|
||||
use super::{Detector, Findings, Region, Strength, checks, valid_short_date};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[Detector::new(
|
||||
"za-id",
|
||||
"South Africa: ID number",
|
||||
Region::Africa,
|
||||
Strength::Checked,
|
||||
za_id,
|
||||
)];
|
||||
|
||||
/// Birth date `YYMMDD`, four digits, citizenship (0, 1 or 2), 8 or 9, a Luhn
|
||||
/// check digit. The date and the two fixed digits make it strong enough to
|
||||
/// count alone.
|
||||
static ZA_ID: LazyLock<Regex> = LazyLock::new(|| {
|
||||
Regex::new(r"\b(\d{2})(\d{2})(\d{2})\d{4}[012][89]\d\b").expect("detector pattern")
|
||||
});
|
||||
|
||||
fn za_id(text: &str, findings: &mut Findings) {
|
||||
for c in ZA_ID.captures_iter(text) {
|
||||
let n = &c[0];
|
||||
let num = |s: &str| s.parse::<u32>().unwrap_or(0);
|
||||
if valid_short_date(num(&c[1]), num(&c[2]), num(&c[3])) && checks::luhn(n) {
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
#[test]
|
||||
fn south_africa() {
|
||||
let detector = by_id("za-id").unwrap();
|
||||
assert_eq!(detector.count("ID 8001015009087"), 1);
|
||||
assert_eq!(detector.count("8001015009088"), 0);
|
||||
assert_eq!(detector.count("8013015009087"), 0);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,172 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Identifiers from the Americas outside the US and Canada (§2.3): Brazil's
|
||||
//! CPF and CNPJ, and Mexico's CURP.
|
||||
|
||||
use super::{Detector, Findings, Region, Strength, digit_values, valid_short_date, word_near};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[
|
||||
Detector::new(
|
||||
"br-cpf",
|
||||
"Brazil: CPF",
|
||||
Region::Americas,
|
||||
Strength::Checked,
|
||||
br_cpf,
|
||||
),
|
||||
Detector::new(
|
||||
"br-cnpj",
|
||||
"Brazil: CNPJ",
|
||||
Region::Americas,
|
||||
Strength::Checked,
|
||||
br_cnpj,
|
||||
),
|
||||
Detector::new(
|
||||
"mx-curp",
|
||||
"Mexico: CURP",
|
||||
Region::Americas,
|
||||
Strength::Checked,
|
||||
mx_curp,
|
||||
),
|
||||
];
|
||||
|
||||
fn re(pattern: &str) -> Regex {
|
||||
Regex::new(pattern).expect("detector pattern")
|
||||
}
|
||||
|
||||
/// Brazil's mod 11 check digit over `digits` with `weights`.
|
||||
fn br_check(digits: &[u32], weights: &[u32]) -> u32 {
|
||||
match digits.iter().zip(weights).map(|(a, w)| a * w).sum::<u32>() % 11 {
|
||||
0 | 1 => 0,
|
||||
r => 11 - r,
|
||||
}
|
||||
}
|
||||
|
||||
/// `111.444.777-35`, or eleven bare digits.
|
||||
static CPF: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{3}(\.?)\d{3}(\.?)\d{3}(-?)\d{2}\b"));
|
||||
|
||||
pub fn cpf_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
// A run of one digit passes the arithmetic but is never issued
|
||||
d.len() == 11
|
||||
&& d.iter().any(|x| *x != d[0])
|
||||
&& br_check(&d[..9], &[10, 9, 8, 7, 6, 5, 4, 3, 2]) == d[9]
|
||||
&& br_check(&d[..10], &[11, 10, 9, 8, 7, 6, 5, 4, 3, 2]) == d[10]
|
||||
}
|
||||
|
||||
const CPF_WORDS: &[&str] = &[
|
||||
"cpf",
|
||||
"cadastro de pessoas físicas",
|
||||
"cadastro de pessoa física",
|
||||
];
|
||||
|
||||
fn br_cpf(text: &str, findings: &mut Findings) {
|
||||
for c in CPF.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let written = &c[1] == "." && &c[2] == "." && &c[3] == "-";
|
||||
let n: String = whole
|
||||
.as_str()
|
||||
.chars()
|
||||
.filter(char::is_ascii_digit)
|
||||
.collect();
|
||||
if cpf_valid(&n) && (written || word_near(text, whole.start(), whole.end(), CPF_WORDS)) {
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// `11.222.333/0001-81`, or fourteen bare digits.
|
||||
static CNPJ: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\b\d{2}(\.?)\d{3}(\.?)\d{3}(/?)\d{4}(-?)\d{2}\b"));
|
||||
|
||||
pub fn cnpj_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
d.len() == 14
|
||||
&& d.iter().any(|x| *x != d[0])
|
||||
&& br_check(&d[..12], &[5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2]) == d[12]
|
||||
&& br_check(&d[..13], &[6, 5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2]) == d[13]
|
||||
}
|
||||
|
||||
const CNPJ_WORDS: &[&str] = &["cnpj", "cadastro nacional da pessoa jurídica"];
|
||||
|
||||
fn br_cnpj(text: &str, findings: &mut Findings) {
|
||||
for c in CNPJ.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let written = &c[1] == "." && &c[2] == "." && &c[3] == "/" && &c[4] == "-";
|
||||
let n: String = whole
|
||||
.as_str()
|
||||
.chars()
|
||||
.filter(char::is_ascii_digit)
|
||||
.collect();
|
||||
if cnpj_valid(&n) && (written || word_near(text, whole.start(), whole.end(), CNPJ_WORDS)) {
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Four letters, the birth date, sex (H, M or X), the state, three
|
||||
/// consonants, a character that tells the century apart, the check digit.
|
||||
static CURP: LazyLock<Regex> = LazyLock::new(|| {
|
||||
re(r"(?i)\b[A-Z]{4}(\d{2})(\d{2})(\d{2})[HMX][A-Z]{2}[B-DF-HJ-NP-TV-Z]{3}[A-Z0-9]\d\b")
|
||||
});
|
||||
|
||||
/// RENAPO's check: each character's place in `0-9 A-N Ñ O-Z`, weighted 18
|
||||
/// down to 2; the digit is 10 minus the sum mod 10 (10 becomes 0).
|
||||
pub fn curp_valid(curp: &str) -> bool {
|
||||
const ALPHABET: &str = "0123456789ABCDEFGHIJKLMNÑOPQRSTUVWXYZ";
|
||||
let mut sum = 0u32;
|
||||
for (i, c) in curp.chars().take(17).enumerate() {
|
||||
let Some(value) = ALPHABET.chars().position(|a| a == c) else {
|
||||
return false;
|
||||
};
|
||||
sum += value as u32 * (18 - i as u32);
|
||||
}
|
||||
curp.chars().nth(17).and_then(|c| c.to_digit(10)) == Some((10 - sum % 10) % 10)
|
||||
}
|
||||
|
||||
fn mx_curp(text: &str, findings: &mut Findings) {
|
||||
for c in CURP.captures_iter(text) {
|
||||
let curp = c[0].to_ascii_uppercase();
|
||||
if valid_short_date(num(&c[1]), num(&c[2]), num(&c[3])) && curp_valid(&curp) {
|
||||
findings.insert(curp);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn num(s: &str) -> u32 {
|
||||
s.parse().unwrap_or(0)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
fn count(id: &str, text: &str) -> usize {
|
||||
by_id(id).unwrap().count(text)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn brazil() {
|
||||
assert_eq!(count("br-cpf", "CPF 111.444.777-35"), 1);
|
||||
assert_eq!(count("br-cpf", "111.444.777-36"), 0);
|
||||
assert_eq!(count("br-cpf", "pedido 11144477735"), 0);
|
||||
assert_eq!(count("br-cpf", "cpf: 11144477735"), 1);
|
||||
assert_eq!(count("br-cpf", "CPF 111.111.111-11"), 0);
|
||||
assert_eq!(count("br-cnpj", "11.222.333/0001-81"), 1);
|
||||
assert_eq!(count("br-cnpj", "11.222.333/0001-82"), 0);
|
||||
assert_eq!(count("br-cnpj", "CNPJ 11222333000181"), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn mexico() {
|
||||
// python-stdnum's documented example
|
||||
assert_eq!(count("mx-curp", "CURP BOXW310820HNERXN09"), 1);
|
||||
assert_eq!(count("mx-curp", "BOXW310820HNERXN08"), 0);
|
||||
assert_eq!(count("mx-curp", "BOXW311320HNERXN09"), 0);
|
||||
}
|
||||
}
|
||||
@@ -6,7 +6,9 @@
|
||||
|
||||
//! Detectors that aren't tied to one country (§2.3, region "Any").
|
||||
|
||||
use super::{Detector, Findings, Region, Strength, checks, digits, stands_alone, word_near};
|
||||
use super::{
|
||||
Detector, Findings, Region, Strength, checks, digits, stands_alone, valid_date, word_near,
|
||||
};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
@@ -295,19 +297,6 @@ const BIRTH_WORDS: &[&str] = &[
|
||||
"nascimento",
|
||||
];
|
||||
|
||||
fn valid_date(year: u32, month: u32, day: u32) -> bool {
|
||||
let days = match month {
|
||||
1 | 3 | 5 | 7 | 8 | 10 | 12 => 31,
|
||||
4 | 6 | 9 | 11 => 30,
|
||||
2 if year.is_multiple_of(4) && (!year.is_multiple_of(100) || year.is_multiple_of(400)) => {
|
||||
29
|
||||
}
|
||||
2 => 28,
|
||||
_ => return false,
|
||||
};
|
||||
(1900..=2100).contains(&year) && (1..=days).contains(&day)
|
||||
}
|
||||
|
||||
fn month_number(name: &str) -> u32 {
|
||||
const MONTHS: [&str; 12] = [
|
||||
"jan", "feb", "mar", "apr", "may", "jun", "jul", "aug", "sep", "oct", "nov", "dec",
|
||||
|
||||
@@ -0,0 +1,281 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Asian identifiers (§2.3): India's Aadhaar and PAN, China's resident ID,
|
||||
//! Japan's My Number, Singapore's NRIC and FIN, and South Korea's resident
|
||||
//! registration number.
|
||||
|
||||
use super::{
|
||||
Detector, Findings, Region, Strength, digit_values, valid_date, valid_short_date, word_near,
|
||||
};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[
|
||||
Detector::new(
|
||||
"in-aadhaar",
|
||||
"India: Aadhaar",
|
||||
Region::Asia,
|
||||
Strength::Checked,
|
||||
in_aadhaar,
|
||||
),
|
||||
Detector::new(
|
||||
"in-pan",
|
||||
"India: PAN",
|
||||
Region::Asia,
|
||||
Strength::NeedsWord,
|
||||
in_pan,
|
||||
),
|
||||
Detector::new(
|
||||
"cn-resident-id",
|
||||
"China: resident ID",
|
||||
Region::Asia,
|
||||
Strength::Checked,
|
||||
cn_resident_id,
|
||||
),
|
||||
Detector::new(
|
||||
"jp-my-number",
|
||||
"Japan: My Number",
|
||||
Region::Asia,
|
||||
Strength::Checked,
|
||||
jp_my_number,
|
||||
),
|
||||
Detector::new(
|
||||
"sg-nric",
|
||||
"Singapore: NRIC and FIN",
|
||||
Region::Asia,
|
||||
Strength::Checked,
|
||||
sg_nric,
|
||||
),
|
||||
Detector::new(
|
||||
"kr-rrn",
|
||||
"South Korea: resident registration number",
|
||||
Region::Asia,
|
||||
Strength::NeedsWord,
|
||||
kr_rrn,
|
||||
),
|
||||
];
|
||||
|
||||
fn re(pattern: &str) -> Regex {
|
||||
Regex::new(pattern).expect("detector pattern")
|
||||
}
|
||||
|
||||
/// Twelve digits written in fours, or bare.
|
||||
static TWELVE: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{4})( ?)(\d{4})( ?)(\d{4})\b"));
|
||||
|
||||
const VERHOEFF_D: [[u8; 10]; 10] = [
|
||||
[0, 1, 2, 3, 4, 5, 6, 7, 8, 9],
|
||||
[1, 2, 3, 4, 0, 6, 7, 8, 9, 5],
|
||||
[2, 3, 4, 0, 1, 7, 8, 9, 5, 6],
|
||||
[3, 4, 0, 1, 2, 8, 9, 5, 6, 7],
|
||||
[4, 0, 1, 2, 3, 9, 5, 6, 7, 8],
|
||||
[5, 9, 8, 7, 6, 0, 4, 3, 2, 1],
|
||||
[6, 5, 9, 8, 7, 1, 0, 4, 3, 2],
|
||||
[7, 6, 5, 9, 8, 2, 1, 0, 4, 3],
|
||||
[8, 7, 6, 5, 9, 3, 2, 1, 0, 4],
|
||||
[9, 8, 7, 6, 5, 4, 3, 2, 1, 0],
|
||||
];
|
||||
const VERHOEFF_P: [[u8; 10]; 8] = [
|
||||
[0, 1, 2, 3, 4, 5, 6, 7, 8, 9],
|
||||
[1, 5, 7, 6, 2, 8, 3, 0, 9, 4],
|
||||
[5, 8, 0, 3, 7, 9, 6, 1, 4, 2],
|
||||
[8, 9, 1, 6, 0, 4, 3, 5, 2, 7],
|
||||
[9, 4, 5, 3, 1, 2, 6, 8, 7, 0],
|
||||
[4, 2, 8, 6, 5, 7, 3, 9, 0, 1],
|
||||
[2, 7, 9, 3, 8, 0, 6, 4, 1, 5],
|
||||
[7, 0, 4, 6, 9, 1, 3, 2, 5, 8],
|
||||
];
|
||||
|
||||
/// The Verhoeff check (dihedral group D5).
|
||||
pub fn verhoeff(n: &str) -> bool {
|
||||
let mut c = 0u8;
|
||||
for (i, b) in n.bytes().rev().enumerate() {
|
||||
c = VERHOEFF_D[c as usize][VERHOEFF_P[i % 8][(b - b'0') as usize] as usize];
|
||||
}
|
||||
c == 0
|
||||
}
|
||||
|
||||
const AADHAAR_WORDS: &[&str] = &["aadhaar", "aadhar", "uidai", "uid"];
|
||||
|
||||
fn in_aadhaar(text: &str, findings: &mut Findings) {
|
||||
for c in TWELVE.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
|
||||
let written = &c[2] == " " && &c[4] == " ";
|
||||
// Never starts with 0 or 1
|
||||
if !n.starts_with(['0', '1'])
|
||||
&& verhoeff(&n)
|
||||
&& (written || word_near(text, whole.start(), whole.end(), AADHAAR_WORDS))
|
||||
{
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Five letters (the fourth names the holder's type), four digits, a letter.
|
||||
static PAN: LazyLock<Regex> = LazyLock::new(|| re(r"\b[A-Z]{3}[ABCFGHLJPTK][A-Z]\d{4}[A-Z]\b"));
|
||||
|
||||
const PAN_WORDS: &[&str] = &["pan", "pan card", "permanent account number", "income tax"];
|
||||
|
||||
fn in_pan(text: &str, findings: &mut Findings) {
|
||||
for m in PAN.find_iter(text) {
|
||||
if word_near(text, m.start(), m.end(), PAN_WORDS) {
|
||||
findings.insert(m.as_str());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Region, birth date `YYYYMMDD`, sequence, then the ISO 7064 MOD 11-2
|
||||
/// check (0–9 or X).
|
||||
static CN_ID: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"(?i)\b[1-8]\d{5}(\d{4})(\d{2})(\d{2})\d{3}[\dX]\b"));
|
||||
|
||||
pub fn cn_id_valid(id: &str) -> bool {
|
||||
const WEIGHTS: [u32; 17] = [7, 9, 10, 5, 8, 4, 2, 1, 6, 3, 7, 9, 10, 5, 8, 4, 2];
|
||||
const CHECKS: &[u8] = b"10X98765432";
|
||||
let sum: u32 = digit_values(&id[..17])
|
||||
.iter()
|
||||
.zip(WEIGHTS)
|
||||
.map(|(a, w)| a * w)
|
||||
.sum();
|
||||
CHECKS[(sum % 11) as usize] == id.as_bytes()[17].to_ascii_uppercase()
|
||||
}
|
||||
|
||||
fn cn_resident_id(text: &str, findings: &mut Findings) {
|
||||
for c in CN_ID.captures_iter(text) {
|
||||
let id = c[0].to_ascii_uppercase();
|
||||
let (y, m, d) = (num(&c[1]), num(&c[2]), num(&c[3]));
|
||||
if valid_date(y, m, d) && cn_id_valid(&id) {
|
||||
findings.insert(id);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// My Number: weights 2–7 then 2–6 from the right; a remainder of 0 or 1
|
||||
/// gives 0, else 11 minus it.
|
||||
pub fn my_number_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
let sum: u32 = (1..=11)
|
||||
.map(|i| d[11 - i] * if i <= 6 { i as u32 + 1 } else { i as u32 - 5 })
|
||||
.sum();
|
||||
let check = match sum % 11 {
|
||||
0 | 1 => 0,
|
||||
r => 11 - r,
|
||||
};
|
||||
check == d[11]
|
||||
}
|
||||
|
||||
const MY_NUMBER_WORDS: &[&str] = &[
|
||||
"my number",
|
||||
"mynumber",
|
||||
"マイナンバー",
|
||||
"個人番号",
|
||||
"kojin bango",
|
||||
];
|
||||
|
||||
fn jp_my_number(text: &str, findings: &mut Findings) {
|
||||
for c in TWELVE.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
|
||||
let written = &c[2] == " " && &c[4] == " ";
|
||||
if my_number_valid(&n)
|
||||
&& (written || word_near(text, whole.start(), whole.end(), MY_NUMBER_WORDS))
|
||||
{
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static NRIC: LazyLock<Regex> = LazyLock::new(|| re(r"(?i)\b([STFGM])(\d{7})([A-Z])\b"));
|
||||
|
||||
/// Weights 2, 7, 6, 5, 4, 3, 2; T and G add 4, M adds 3; each series has its
|
||||
/// own table of check letters.
|
||||
fn nric_valid(prefix: u8, digits: &str, check: u8) -> bool {
|
||||
let sum: u32 = digit_values(digits)
|
||||
.iter()
|
||||
.zip([2, 7, 6, 5, 4, 3, 2])
|
||||
.map(|(a, w)| a * w)
|
||||
.sum::<u32>()
|
||||
+ match prefix {
|
||||
b'T' | b'G' => 4,
|
||||
b'M' => 3,
|
||||
_ => 0,
|
||||
};
|
||||
let table: &[u8] = match prefix {
|
||||
b'S' | b'T' => b"JZIHGFEDCBA",
|
||||
b'F' | b'G' => b"XWUTRQPNMLK",
|
||||
_ => b"KLJNPQRTUWX",
|
||||
};
|
||||
table[(sum % 11) as usize] == check
|
||||
}
|
||||
|
||||
fn sg_nric(text: &str, findings: &mut Findings) {
|
||||
for c in NRIC.captures_iter(text) {
|
||||
let id = c[0].to_ascii_uppercase();
|
||||
let bytes = id.as_bytes();
|
||||
if nric_valid(bytes[0], &c[2], bytes[8]) {
|
||||
findings.insert(id);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// `YYMMDD-GNNNNNN`, the seventh digit giving sex and century.
|
||||
static RRN: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{2})(\d{2})(\d{2})-?([1-8])\d{6}\b"));
|
||||
|
||||
const RRN_WORDS: &[&str] = &["주민등록번호", "주민번호", "resident registration", "rrn"];
|
||||
|
||||
fn kr_rrn(text: &str, findings: &mut Findings) {
|
||||
for c in RRN.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
if valid_short_date(num(&c[1]), num(&c[2]), num(&c[3]))
|
||||
&& word_near(text, whole.start(), whole.end(), RRN_WORDS)
|
||||
{
|
||||
findings.insert(whole.as_str().replace('-', ""));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn num(s: &str) -> u32 {
|
||||
s.parse().unwrap_or(0)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
fn count(id: &str, text: &str) -> usize {
|
||||
by_id(id).unwrap().count(text)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn india() {
|
||||
assert_eq!(count("in-aadhaar", "2345 6789 0124"), 1);
|
||||
assert_eq!(count("in-aadhaar", "2345 6789 0125"), 0);
|
||||
assert_eq!(count("in-aadhaar", "order 234567890124"), 0);
|
||||
assert_eq!(count("in-aadhaar", "Aadhaar 234567890124"), 1);
|
||||
assert_eq!(count("in-pan", "PAN: ABCPE1234F"), 1);
|
||||
assert_eq!(count("in-pan", "ABCPE1234F"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn china_japan() {
|
||||
assert_eq!(count("cn-resident-id", "11010519491231002X"), 1);
|
||||
assert_eq!(count("cn-resident-id", "110105194912310021"), 0);
|
||||
assert_eq!(count("cn-resident-id", "11010519491331002X"), 0);
|
||||
assert_eq!(count("jp-my-number", "1234 5678 9018"), 1);
|
||||
assert_eq!(count("jp-my-number", "1234 5678 9017"), 0);
|
||||
assert_eq!(count("jp-my-number", "マイナンバー 123456789018"), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn singapore_korea() {
|
||||
assert_eq!(count("sg-nric", "S1234567D and T1234567J"), 2);
|
||||
assert_eq!(count("sg-nric", "S1234567E"), 0);
|
||||
assert_eq!(count("kr-rrn", "주민등록번호 800101-1234567"), 1);
|
||||
assert_eq!(count("kr-rrn", "800101-1234567"), 0);
|
||||
assert_eq!(count("kr-rrn", "RRN 801301-1234567"), 0);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,128 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Australian identifiers (§2.3): the ATO's Tax File Number and the Medicare
|
||||
//! card number.
|
||||
|
||||
use super::{Detector, Findings, Region, Strength, word_near};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[
|
||||
Detector::new(
|
||||
"au-tfn",
|
||||
"Australian Tax File Number",
|
||||
Region::Australia,
|
||||
Strength::Checked,
|
||||
tfn,
|
||||
),
|
||||
Detector::new(
|
||||
"au-medicare",
|
||||
"Australian Medicare number",
|
||||
Region::Australia,
|
||||
Strength::Checked,
|
||||
medicare,
|
||||
),
|
||||
];
|
||||
|
||||
fn re(pattern: &str) -> Regex {
|
||||
Regex::new(pattern).expect("detector pattern")
|
||||
}
|
||||
|
||||
/// `NNN NNN NNN` stands alone; bare digits (eight or nine) need a word.
|
||||
static TFN: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{3})( ?)(\d{3})( ?)(\d{2,3})\b"));
|
||||
|
||||
/// Weighted sum mod 11, with the ATO's weights for 9- and 8-digit numbers.
|
||||
pub fn tfn_valid(n: &str) -> bool {
|
||||
let weights: &[u32] = match n.len() {
|
||||
9 => &[1, 4, 3, 7, 5, 8, 6, 9, 10],
|
||||
8 => &[10, 7, 8, 4, 6, 3, 5, 1],
|
||||
_ => return false,
|
||||
};
|
||||
n.bytes()
|
||||
.zip(weights)
|
||||
.map(|(b, w)| u32::from(b - b'0') * w)
|
||||
.sum::<u32>()
|
||||
% 11
|
||||
== 0
|
||||
}
|
||||
|
||||
const TFN_WORDS: &[&str] = &["tfn", "tax file number", "tax file no"];
|
||||
|
||||
fn tfn(text: &str, findings: &mut Findings) {
|
||||
for c in TFN.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
|
||||
let written = n.len() == 9 && c[2] == *" " && c[4] == *" ";
|
||||
if tfn_valid(&n) && (written || word_near(text, whole.start(), whole.end(), TFN_WORDS)) {
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// `NNNN NNNNN N` (and an optional issue number) stands alone; bare digits
|
||||
/// need a word.
|
||||
static MEDICARE: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\b([2-6]\d{3})( ?)(\d{5})( ?)(\d)(?:[ -]?\d)?\b"));
|
||||
|
||||
/// The ninth digit is the weighted sum (1, 3, 7, 9, 1, 3, 7, 9) of the first
|
||||
/// eight, mod 10.
|
||||
pub fn medicare_valid(n: &str) -> bool {
|
||||
let d: Vec<u32> = n.bytes().map(|b| u32::from(b - b'0')).collect();
|
||||
d.len() >= 9
|
||||
&& d[..8]
|
||||
.iter()
|
||||
.zip([1, 3, 7, 9, 1, 3, 7, 9])
|
||||
.map(|(a, w)| a * w)
|
||||
.sum::<u32>()
|
||||
% 10
|
||||
== d[8]
|
||||
}
|
||||
|
||||
const MEDICARE_WORDS: &[&str] = &[
|
||||
"medicare",
|
||||
"medicare card",
|
||||
"medicare no",
|
||||
"medicare number",
|
||||
];
|
||||
|
||||
fn medicare(text: &str, findings: &mut Findings) {
|
||||
for c in MEDICARE.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
|
||||
let written = c[2] == *" " && c[4] == *" ";
|
||||
if medicare_valid(&n)
|
||||
&& (written || word_near(text, whole.start(), whole.end(), MEDICARE_WORDS))
|
||||
{
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
fn count(id: &str, text: &str) -> usize {
|
||||
by_id(id).unwrap().count(text)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tax_file_numbers() {
|
||||
assert_eq!(count("au-tfn", "TFN 123 456 782"), 1);
|
||||
assert_eq!(count("au-tfn", "123 456 789"), 0);
|
||||
assert_eq!(count("au-tfn", "order 123456782"), 0);
|
||||
assert_eq!(count("au-tfn", "tax file number 123456782"), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn medicare_numbers() {
|
||||
assert_eq!(count("au-medicare", "2123 45670 1"), 1);
|
||||
assert_eq!(count("au-medicare", "2123 45671 1"), 0);
|
||||
assert_eq!(count("au-medicare", "ref 2123456701"), 0);
|
||||
assert_eq!(count("au-medicare", "Medicare 2123456701"), 1);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,66 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Canadian identifiers (§2.3): the Social Insurance Number.
|
||||
|
||||
use super::{Detector, Findings, Region, Strength, checks, word_near};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[Detector::new(
|
||||
"ca-sin",
|
||||
"Canadian Social Insurance Number",
|
||||
Region::Canada,
|
||||
Strength::Checked,
|
||||
sin,
|
||||
)];
|
||||
|
||||
/// `NNN NNN NNN` or `NNN-NNN-NNN` stands alone; nine bare digits need a word.
|
||||
static SIN: LazyLock<Regex> = LazyLock::new(|| {
|
||||
Regex::new(r"\b(\d{3})([ -]?)(\d{3})([ -]?)(\d{3})\b").expect("detector pattern")
|
||||
});
|
||||
|
||||
const SIN_WORDS: &[&str] = &[
|
||||
"sin",
|
||||
"social insurance",
|
||||
"nas",
|
||||
"numéro d'assurance sociale",
|
||||
"assurance sociale",
|
||||
];
|
||||
|
||||
fn sin(text: &str, findings: &mut Findings) {
|
||||
for c in SIN.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
|
||||
let written = !c[2].is_empty() && c[2] == c[4];
|
||||
// 0 and 8 are never issued as a first digit
|
||||
if !n.starts_with(['0', '8'])
|
||||
&& checks::luhn(&n)
|
||||
&& (written || word_near(text, whole.start(), whole.end(), SIN_WORDS))
|
||||
{
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
fn count(text: &str) -> usize {
|
||||
by_id("ca-sin").unwrap().count(text)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn social_insurance_numbers() {
|
||||
assert_eq!(count("130 692 544 and 193-456-787"), 2);
|
||||
assert_eq!(count("130 692 545"), 0);
|
||||
// The government's printed example starts with 0, never issued
|
||||
assert_eq!(count("046 454 286"), 0);
|
||||
assert_eq!(count("order 130692544"), 0);
|
||||
assert_eq!(count("SIN: 130692544"), 1);
|
||||
}
|
||||
}
|
||||
@@ -25,7 +25,7 @@ pub fn luhn(digits: &str) -> bool {
|
||||
}
|
||||
})
|
||||
.sum();
|
||||
sum % 10 == 0
|
||||
sum.is_multiple_of(10)
|
||||
}
|
||||
|
||||
/// ISO 13616 IBAN lengths, by country, from the IBAN registry.
|
||||
|
||||
@@ -0,0 +1,646 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! European Union national identifiers (§2.3), each from its issuer's
|
||||
//! published rules. An identifier that is only digits and whose check a
|
||||
//! random number passes often (mod 10, mod 11) counts alone only in its
|
||||
//! written form, and as bare digits only beside a word.
|
||||
|
||||
use super::{
|
||||
Detector, Findings, Region, Strength, checks, digit_values, stands_alone, valid_short_date,
|
||||
word_near,
|
||||
};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[
|
||||
Detector::new(
|
||||
"de-tax-id",
|
||||
"Germany: tax ID (Steuer-ID)",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
de_tax_id,
|
||||
),
|
||||
Detector::new(
|
||||
"de-id-card",
|
||||
"Germany: ID card number",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
de_id_card,
|
||||
),
|
||||
Detector::new(
|
||||
"fr-nir",
|
||||
"France: social security number (NIR)",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
fr_nir,
|
||||
),
|
||||
Detector::new(
|
||||
"es-dni-nie",
|
||||
"Spain: DNI and NIE",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
es_dni_nie,
|
||||
),
|
||||
Detector::new(
|
||||
"it-codice-fiscale",
|
||||
"Italy: codice fiscale",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
it_codice_fiscale,
|
||||
),
|
||||
Detector::new(
|
||||
"nl-bsn",
|
||||
"Netherlands: BSN",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
nl_bsn,
|
||||
),
|
||||
Detector::new(
|
||||
"be-national-number",
|
||||
"Belgium: national number",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
be_national_number,
|
||||
),
|
||||
Detector::new(
|
||||
"pl-pesel",
|
||||
"Poland: PESEL",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
pl_pesel,
|
||||
),
|
||||
Detector::new(
|
||||
"se-personnummer",
|
||||
"Sweden: personnummer",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
se_personnummer,
|
||||
),
|
||||
Detector::new(
|
||||
"dk-cpr",
|
||||
"Denmark: CPR number",
|
||||
Region::Eu,
|
||||
Strength::NeedsWord,
|
||||
dk_cpr,
|
||||
),
|
||||
Detector::new(
|
||||
"fi-hetu",
|
||||
"Finland: personal identity code",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
fi_hetu,
|
||||
),
|
||||
Detector::new(
|
||||
"ie-pps",
|
||||
"Ireland: PPS number",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
ie_pps,
|
||||
),
|
||||
Detector::new(
|
||||
"pt-nif",
|
||||
"Portugal: NIF",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
pt_nif,
|
||||
),
|
||||
Detector::new(
|
||||
"at-svnr",
|
||||
"Austria: social insurance number",
|
||||
Region::Eu,
|
||||
Strength::Checked,
|
||||
at_svnr,
|
||||
),
|
||||
];
|
||||
|
||||
fn re(pattern: &str) -> Regex {
|
||||
Regex::new(pattern).expect("detector pattern")
|
||||
}
|
||||
|
||||
fn num(s: &str) -> u32 {
|
||||
s.parse().unwrap_or(0)
|
||||
}
|
||||
|
||||
// --- Germany --------------------------------------------------------------
|
||||
|
||||
/// Eleven digits, written `86 095 742 719` on the BZSt's letters.
|
||||
static DE_TAX: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{2}( ?)\d{3}( ?)\d{3}( ?)\d{3}\b"));
|
||||
|
||||
/// ISO 7064 MOD 11,10; no leading zero; in the first ten digits one digit
|
||||
/// appears two or three times and every other at most once.
|
||||
pub fn de_tax_id_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
if d.len() != 11 || d[0] == 0 {
|
||||
return false;
|
||||
}
|
||||
let mut counts = [0u8; 10];
|
||||
for &x in &d[..10] {
|
||||
counts[x as usize] += 1;
|
||||
}
|
||||
let repeated = counts.iter().filter(|&&c| c >= 2).count();
|
||||
if repeated != 1 || counts.iter().any(|&c| c > 3) {
|
||||
return false;
|
||||
}
|
||||
let mut product = 10;
|
||||
for &x in &d[..10] {
|
||||
let mut sum = (x + product) % 10;
|
||||
if sum == 0 {
|
||||
sum = 10;
|
||||
}
|
||||
product = (2 * sum) % 11;
|
||||
}
|
||||
let check = match 11 - product {
|
||||
10 => 0,
|
||||
c => c,
|
||||
};
|
||||
check == d[10]
|
||||
}
|
||||
|
||||
const DE_TAX_WORDS: &[&str] = &[
|
||||
"steuer-id",
|
||||
"steueridentifikationsnummer",
|
||||
"steuerliche identifikationsnummer",
|
||||
"idnr",
|
||||
"identifikationsnummer",
|
||||
"tax id",
|
||||
];
|
||||
|
||||
fn de_tax_id(text: &str, findings: &mut Findings) {
|
||||
for c in DE_TAX.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let written = [&c[1], &c[2], &c[3]].iter().all(|s| *s == " ");
|
||||
let n: String = whole.as_str().replace(' ', "");
|
||||
if de_tax_id_valid(&n)
|
||||
&& (written || word_near(text, whole.start(), whole.end(), DE_TAX_WORDS))
|
||||
{
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The ID card's document number: a letter from the card's alphabet, eight
|
||||
/// more characters from it, then the check digit.
|
||||
static DE_ID: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\b[CFGHJKLMNPRTVWXYZ][CFGHJKLMNPRTVWXYZ0-9]{8}\d\b"));
|
||||
|
||||
/// ICAO 9303 check digit: weights 7, 3, 1; letters A=10 … Z=35.
|
||||
pub fn icao_check(chars: &str, check: u32) -> bool {
|
||||
let value = |c: char| c.to_digit(10).unwrap_or_else(|| c as u32 - 'A' as u32 + 10);
|
||||
let sum: u32 = chars
|
||||
.chars()
|
||||
.zip([7, 3, 1].iter().cycle())
|
||||
.map(|(c, w)| value(c) * w)
|
||||
.sum();
|
||||
sum % 10 == check
|
||||
}
|
||||
|
||||
fn de_id_card(text: &str, findings: &mut Findings) {
|
||||
for m in DE_ID.find_iter(text) {
|
||||
let s = m.as_str();
|
||||
if icao_check(&s[..9], num(&s[9..])) {
|
||||
findings.insert(s);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- France ---------------------------------------------------------------
|
||||
|
||||
/// Sex, year, month, department (with Corsica's 2A and 2B), commune, order,
|
||||
/// then the two-digit key, spaces allowed between groups.
|
||||
static FR_NIR: LazyLock<Regex> = LazyLock::new(|| {
|
||||
re(r"\b([1-478]) ?(\d{2}) ?(\d{2}) ?(\d{2}|2[AB]) ?(\d{3}) ?(\d{3}) ?(\d{2})\b")
|
||||
});
|
||||
|
||||
fn fr_nir(text: &str, findings: &mut Findings) {
|
||||
for c in FR_NIR.captures_iter(text) {
|
||||
let month = num(&c[3]);
|
||||
if !(matches!(month, 1..=12 | 20..=42 | 50..=99)) {
|
||||
continue;
|
||||
}
|
||||
let department = match &c[4] {
|
||||
"2A" => "19",
|
||||
"2B" => "18",
|
||||
d => d,
|
||||
};
|
||||
let body = format!(
|
||||
"{}{}{}{}{}{}",
|
||||
&c[1], &c[2], &c[3], department, &c[5], &c[6]
|
||||
);
|
||||
let Ok(value) = body.parse::<u64>() else {
|
||||
continue;
|
||||
};
|
||||
if 97 - value % 97 == u64::from(num(&c[7])) {
|
||||
findings.insert(format!(
|
||||
"{}{}{}{}{}{}{}",
|
||||
&c[1], &c[2], &c[3], &c[4], &c[5], &c[6], &c[7]
|
||||
));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Spain ----------------------------------------------------------------
|
||||
|
||||
static ES_ID: LazyLock<Regex> = LazyLock::new(|| re(r"(?i)\b([XYZ]?)[ -]?(\d{7,8})[ -]?([A-Z])\b"));
|
||||
|
||||
const DNI_LETTERS: &[u8] = b"TRWAGMYFPDXBNJZSQVHLCKE";
|
||||
|
||||
fn es_dni_nie(text: &str, findings: &mut Findings) {
|
||||
for c in ES_ID.captures_iter(text) {
|
||||
let prefix = c[1].to_ascii_uppercase();
|
||||
let digits = &c[2];
|
||||
// DNI: eight digits; NIE: X, Y or Z and seven digits
|
||||
let number = match (prefix.as_str(), digits.len()) {
|
||||
("", 8) => digits.to_string(),
|
||||
("X", 7) => format!("0{digits}"),
|
||||
("Y", 7) => format!("1{digits}"),
|
||||
("Z", 7) => format!("2{digits}"),
|
||||
_ => continue,
|
||||
};
|
||||
let letter = c[3].to_ascii_uppercase();
|
||||
if DNI_LETTERS[(num(&number) % 23) as usize] == letter.as_bytes()[0] {
|
||||
findings.insert(format!("{prefix}{digits}{letter}"));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Italy ----------------------------------------------------------------
|
||||
|
||||
/// Surname and name letters, year, month letter, day, place code, check
|
||||
/// letter; digits may be replaced by letters (omocodia).
|
||||
static IT_CF: LazyLock<Regex> = LazyLock::new(|| {
|
||||
let d = "[0-9LMNPQRSTUV]";
|
||||
re(&format!(
|
||||
r"(?i)\b[A-Z]{{6}}{d}{{2}}[ABCDEHLMPRST]{d}{{2}}[A-Z]{d}{{3}}[A-Z]\b"
|
||||
))
|
||||
});
|
||||
|
||||
/// The Ministry's odd-position values for 0–9 and A–Z.
|
||||
const CF_ODD: [u32; 36] = [
|
||||
1, 0, 5, 7, 9, 13, 15, 17, 19, 21, // 0-9
|
||||
1, 0, 5, 7, 9, 13, 15, 17, 19, 21, 2, 4, 18, 20, 11, 3, 6, 8, 12, 14, 16, 10, 22, 25, 24,
|
||||
23, // A-Z
|
||||
];
|
||||
|
||||
pub fn codice_fiscale_valid(cf: &str) -> bool {
|
||||
let index = |c: u8| {
|
||||
if c.is_ascii_digit() {
|
||||
(c - b'0') as usize
|
||||
} else {
|
||||
(c - b'A') as usize + 10
|
||||
}
|
||||
};
|
||||
let even = |c: u8| {
|
||||
if c.is_ascii_digit() {
|
||||
u32::from(c - b'0')
|
||||
} else {
|
||||
u32::from(c - b'A')
|
||||
}
|
||||
};
|
||||
let bytes = cf.as_bytes();
|
||||
let sum: u32 = bytes[..15]
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, &c)| {
|
||||
if i % 2 == 0 {
|
||||
CF_ODD[index(c)]
|
||||
} else {
|
||||
even(c)
|
||||
}
|
||||
})
|
||||
.sum();
|
||||
u32::from(bytes[15] - b'A') == sum % 26
|
||||
}
|
||||
|
||||
fn it_codice_fiscale(text: &str, findings: &mut Findings) {
|
||||
for m in IT_CF.find_iter(text) {
|
||||
let cf = m.as_str().to_ascii_uppercase();
|
||||
if codice_fiscale_valid(&cf) {
|
||||
findings.insert(cf);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Netherlands ----------------------------------------------------------
|
||||
|
||||
/// Nine digits, sometimes written `1112.22.333`.
|
||||
static NL_BSN: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{4})(\.?)(\d{2})(\.?)(\d{3})\b"));
|
||||
|
||||
/// The eleven test: weights 9 down to 2, and −1 for the last digit.
|
||||
pub fn bsn_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
let sum: i64 = d[..8]
|
||||
.iter()
|
||||
.zip((2..=9).rev())
|
||||
.map(|(a, w)| i64::from(a * w))
|
||||
.sum::<i64>()
|
||||
- i64::from(d[8]);
|
||||
sum != 0 && sum % 11 == 0
|
||||
}
|
||||
|
||||
const BSN_WORDS: &[&str] = &[
|
||||
"bsn",
|
||||
"burgerservicenummer",
|
||||
"sofinummer",
|
||||
"sofi-nummer",
|
||||
"citizen service number",
|
||||
];
|
||||
|
||||
fn nl_bsn(text: &str, findings: &mut Findings) {
|
||||
for c in NL_BSN.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
|
||||
let written = &c[2] == "." && &c[4] == ".";
|
||||
if bsn_valid(&n) && (written || word_near(text, whole.start(), whole.end(), BSN_WORDS)) {
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Belgium --------------------------------------------------------------
|
||||
|
||||
/// `YY.MM.DD-XXX.CC` or eleven digits.
|
||||
static BE_NN: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\b(\d{2})\.?(\d{2})\.?(\d{2})-?(\d{3})\.?(\d{2})\b"));
|
||||
|
||||
fn be_national_number(text: &str, findings: &mut Findings) {
|
||||
for c in BE_NN.captures_iter(text) {
|
||||
let (month, day) = (num(&c[2]), num(&c[3]));
|
||||
// Month 0 and day 0 mean unknown; bis numbers add 20 or 40 to the month
|
||||
if !(month <= 12 || (20..=32).contains(&month) || (40..=52).contains(&month)) || day > 31 {
|
||||
continue;
|
||||
}
|
||||
let body = format!("{}{}{}{}", &c[1], &c[2], &c[3], &c[4]);
|
||||
let check = u64::from(num(&c[5]));
|
||||
let before_2000 = 97 - body.parse::<u64>().unwrap_or(0) % 97;
|
||||
let since_2000 = 97 - format!("2{body}").parse::<u64>().unwrap_or(0) % 97;
|
||||
if check == before_2000 || check == since_2000 {
|
||||
findings.insert(format!("{body}{}", &c[5]));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Poland ---------------------------------------------------------------
|
||||
|
||||
static ELEVEN: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{11}\b"));
|
||||
|
||||
/// Weights 1, 3, 7, 9 repeating; the birth date encodes the century in the
|
||||
/// month (+80 for the 1800s, +20 for the 2000s, and so on).
|
||||
pub fn pesel_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
let sum: u32 = d[..10]
|
||||
.iter()
|
||||
.zip([1, 3, 7, 9].iter().cycle())
|
||||
.map(|(a, w)| a * w)
|
||||
.sum();
|
||||
let month = d[2] * 10 + d[3];
|
||||
let (century, month) = match month {
|
||||
81..=92 => (1800, month - 80),
|
||||
1..=12 => (1900, month),
|
||||
21..=32 => (2000, month - 20),
|
||||
41..=52 => (2100, month - 40),
|
||||
_ => return false,
|
||||
};
|
||||
let year = century + d[0] * 10 + d[1];
|
||||
(10 - sum % 10) % 10 == d[10] && (1..=super::days_in(year, month)).contains(&(d[4] * 10 + d[5]))
|
||||
}
|
||||
|
||||
const PESEL_WORDS: &[&str] = &["pesel", "numer pesel", "nr pesel"];
|
||||
|
||||
fn pl_pesel(text: &str, findings: &mut Findings) {
|
||||
for m in ELEVEN.find_iter(text) {
|
||||
if pesel_valid(m.as_str()) && word_near(text, m.start(), m.end(), PESEL_WORDS) {
|
||||
findings.insert(m.as_str());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Sweden ---------------------------------------------------------------
|
||||
|
||||
/// `YYMMDD-NNNN`, `YYYYMMDD-NNNN` (`+` after 100), or the bare digits.
|
||||
static SE_PNR: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\b(?:\d{2})?(\d{2})(\d{2})(\d{2})([-+]?)(\d{4})\b"));
|
||||
|
||||
const SE_WORDS: &[&str] = &[
|
||||
"personnummer",
|
||||
"personnr",
|
||||
"person nr",
|
||||
"samordningsnummer",
|
||||
"pnr",
|
||||
];
|
||||
|
||||
fn se_personnummer(text: &str, findings: &mut Findings) {
|
||||
for c in SE_PNR.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let (yy, month, day) = (num(&c[1]), num(&c[2]), num(&c[3]));
|
||||
// Coordination numbers add 60 to the day
|
||||
let day = if day > 60 { day - 60 } else { day };
|
||||
let ten = format!("{}{}{}{}", &c[1], &c[2], &c[3], &c[5]);
|
||||
let written = !c[4].is_empty();
|
||||
if valid_short_date(yy, month, day)
|
||||
&& checks::luhn(&ten)
|
||||
&& (written || word_near(text, whole.start(), whole.end(), SE_WORDS))
|
||||
{
|
||||
findings.insert(ten);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Denmark --------------------------------------------------------------
|
||||
|
||||
static DK_CPR: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{2})(\d{2})(\d{2})-?(\d{4})\b"));
|
||||
|
||||
const CPR_WORDS: &[&str] = &["cpr", "cpr-nr", "cpr nr", "cpr-nummer", "personnummer"];
|
||||
|
||||
fn dk_cpr(text: &str, findings: &mut Findings) {
|
||||
for c in DK_CPR.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
if valid_short_date(num(&c[3]), num(&c[2]), num(&c[1]))
|
||||
&& word_near(text, whole.start(), whole.end(), CPR_WORDS)
|
||||
{
|
||||
findings.insert(whole.as_str().replace('-', ""));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Finland --------------------------------------------------------------
|
||||
|
||||
static FI_HETU: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"(?i)\b(\d{2})(\d{2})(\d{2})[-+ABCDEFYXWVU](\d{3})([0-9A-Y])\b"));
|
||||
|
||||
const HETU_CHECK: &[u8] = b"0123456789ABCDEFHJKLMNPRSTUVWXY";
|
||||
|
||||
fn fi_hetu(text: &str, findings: &mut Findings) {
|
||||
for c in FI_HETU.captures_iter(text) {
|
||||
let (day, month, yy) = (num(&c[1]), num(&c[2]), num(&c[3]));
|
||||
let n: u64 = format!("{}{}{}{}", &c[1], &c[2], &c[3], &c[4])
|
||||
.parse()
|
||||
.unwrap_or(0);
|
||||
let check = c[5].to_ascii_uppercase().as_bytes()[0];
|
||||
if valid_short_date(yy, month, day) && HETU_CHECK[(n % 31) as usize] == check {
|
||||
findings.insert(c[0].to_ascii_uppercase());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Ireland --------------------------------------------------------------
|
||||
|
||||
static IE_PPS: LazyLock<Regex> = LazyLock::new(|| re(r"(?i)\b(\d{7})([A-W])([ABHW]?)\b"));
|
||||
|
||||
const PPS_CHECK: &[u8] = b"WABCDEFGHIJKLMNOPQRSTUV";
|
||||
|
||||
fn ie_pps(text: &str, findings: &mut Findings) {
|
||||
for c in IE_PPS.captures_iter(text) {
|
||||
let mut sum: u32 = digit_values(&c[1])
|
||||
.iter()
|
||||
.zip((2..=8).rev())
|
||||
.map(|(a, w)| a * w)
|
||||
.sum();
|
||||
// The second letter counts, times 9; W (the old form) counts as 0
|
||||
let second = c[3].to_ascii_uppercase();
|
||||
if let Some(&letter) = second.as_bytes().first()
|
||||
&& letter != b'W'
|
||||
{
|
||||
sum += u32::from(letter - b'A' + 1) * 9;
|
||||
}
|
||||
let check = c[2].to_ascii_uppercase().as_bytes()[0];
|
||||
if PPS_CHECK[(sum % 23) as usize] == check {
|
||||
findings.insert(c[0].to_ascii_uppercase());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Portugal -------------------------------------------------------------
|
||||
|
||||
static NINE: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{9}\b"));
|
||||
|
||||
/// Mod 11 over weights 9 down to 2; a check of 10 or 11 becomes 0.
|
||||
pub fn nif_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
let sum: u32 = d[..8].iter().zip((2..=9).rev()).map(|(a, w)| a * w).sum();
|
||||
let check = match 11 - sum % 11 {
|
||||
10 | 11 => 0,
|
||||
c => c,
|
||||
};
|
||||
matches!(d[0], 1 | 2 | 3 | 5 | 6 | 8 | 9) && check == d[8]
|
||||
}
|
||||
|
||||
const NIF_WORDS: &[&str] = &[
|
||||
"nif",
|
||||
"contribuinte",
|
||||
"número de identificação fiscal",
|
||||
"numero de contribuinte",
|
||||
];
|
||||
|
||||
fn pt_nif(text: &str, findings: &mut Findings) {
|
||||
for m in NINE.find_iter(text) {
|
||||
if nif_valid(m.as_str()) && word_near(text, m.start(), m.end(), NIF_WORDS) {
|
||||
findings.insert(m.as_str());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Austria --------------------------------------------------------------
|
||||
|
||||
/// A serial and check digit, then the birth date: `1237 010180`.
|
||||
static AT_SVNR: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{3})(\d)( ?)(\d{2})(\d{2})(\d{2})\b"));
|
||||
|
||||
const SVNR_WORDS: &[&str] = &[
|
||||
"sozialversicherungsnummer",
|
||||
"svnr",
|
||||
"sv-nr",
|
||||
"sv-nummer",
|
||||
"versicherungsnummer",
|
||||
];
|
||||
|
||||
fn at_svnr(text: &str, findings: &mut Findings) {
|
||||
for c in AT_SVNR.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let n = format!("{}{}{}{}{}", &c[1], &c[2], &c[4], &c[5], &c[6]);
|
||||
let d = digit_values(&n);
|
||||
let sum: u32 = d
|
||||
.iter()
|
||||
.zip([3, 7, 9, 0, 5, 8, 4, 2, 1, 6])
|
||||
.map(|(a, w)| a * w)
|
||||
.sum();
|
||||
let written = &c[3] == " ";
|
||||
if d[0] != 0
|
||||
&& sum % 11 == d[3]
|
||||
&& valid_short_date(num(&c[6]), num(&c[5]), num(&c[4]))
|
||||
&& (written || word_near(text, whole.start(), whole.end(), SVNR_WORDS))
|
||||
&& stands_alone(text, whole.start(), whole.end())
|
||||
{
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
fn count(id: &str, text: &str) -> usize {
|
||||
by_id(id).unwrap().count(text)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn germany() {
|
||||
assert_eq!(count("de-tax-id", "86 095 742 719"), 1);
|
||||
assert_eq!(count("de-tax-id", "Steuer-ID: 86095742719"), 1);
|
||||
assert_eq!(count("de-tax-id", "Rechnung 86095742719"), 0);
|
||||
assert_eq!(count("de-tax-id", "86 095 742 718"), 0);
|
||||
// ICAO 9303's German specimen card
|
||||
assert_eq!(count("de-id-card", "Ausweis T220001293"), 1);
|
||||
assert_eq!(count("de-id-card", "T220001294"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn france_spain_italy() {
|
||||
assert_eq!(count("fr-nir", "2 55 08 14 168 025 38"), 1);
|
||||
assert_eq!(count("fr-nir", "255081416802539"), 0);
|
||||
assert_eq!(count("es-dni-nie", "DNI 12345678Z, NIE X-1234567-L"), 2);
|
||||
assert_eq!(count("es-dni-nie", "12345678A"), 0);
|
||||
assert_eq!(count("it-codice-fiscale", "CF: RSSMRA85T10A562S"), 1);
|
||||
assert_eq!(count("it-codice-fiscale", "RSSMRA85T10A562T"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn benelux() {
|
||||
assert_eq!(count("nl-bsn", "1112.22.333"), 1);
|
||||
assert_eq!(count("nl-bsn", "BSN 111222333"), 1);
|
||||
assert_eq!(count("nl-bsn", "order 111222333"), 0);
|
||||
assert_eq!(count("nl-bsn", "BSN 111222334"), 0);
|
||||
assert_eq!(count("be-national-number", "85.07.30-033.28"), 1);
|
||||
assert_eq!(count("be-national-number", "85073003329"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn nordics() {
|
||||
assert_eq!(count("se-personnummer", "811218-9876"), 1);
|
||||
assert_eq!(count("se-personnummer", "811218-9875"), 0);
|
||||
assert_eq!(count("se-personnummer", "order 8112189876"), 0);
|
||||
assert_eq!(count("se-personnummer", "personnummer 198112189876"), 1);
|
||||
assert_eq!(count("dk-cpr", "CPR-nr: 010170-1234"), 1);
|
||||
assert_eq!(count("dk-cpr", "010170-1234"), 0);
|
||||
assert_eq!(count("dk-cpr", "CPR 320170-1234"), 0);
|
||||
assert_eq!(count("fi-hetu", "131052-308T"), 1);
|
||||
assert_eq!(count("fi-hetu", "131052-308U"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn poland_ireland_portugal_austria() {
|
||||
assert_eq!(count("pl-pesel", "PESEL 44051401359, pesel 02070803628"), 2);
|
||||
assert_eq!(count("pl-pesel", "PESEL 44051401358"), 0);
|
||||
assert_eq!(count("pl-pesel", "44051401359"), 0);
|
||||
assert_eq!(count("ie-pps", "PPS 1234567T and 1234567FA"), 2);
|
||||
assert_eq!(count("ie-pps", "1234567U"), 0);
|
||||
assert_eq!(count("pt-nif", "NIF 123456789"), 1);
|
||||
assert_eq!(count("pt-nif", "NIF 123456788"), 0);
|
||||
assert_eq!(count("at-svnr", "1237 010180"), 1);
|
||||
assert_eq!(count("at-svnr", "SVNR 1237010180"), 1);
|
||||
assert_eq!(count("at-svnr", "1238 010180"), 0);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,116 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! European identifiers outside the EU (§2.3): Norway's national identity
|
||||
//! number and Switzerland's AHV number.
|
||||
|
||||
use super::{Detector, Findings, Region, Strength, digit_values, valid_short_date};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[
|
||||
Detector::new(
|
||||
"no-fnr",
|
||||
"Norway: national identity number",
|
||||
Region::Europe,
|
||||
Strength::Checked,
|
||||
no_fnr,
|
||||
),
|
||||
Detector::new(
|
||||
"ch-ahv",
|
||||
"Switzerland: AHV number",
|
||||
Region::Europe,
|
||||
Strength::Checked,
|
||||
ch_ahv,
|
||||
),
|
||||
];
|
||||
|
||||
static ELEVEN: LazyLock<Regex> =
|
||||
LazyLock::new(|| Regex::new(r"\b\d{6} ?\d{5}\b").expect("detector pattern"));
|
||||
|
||||
/// Two mod 11 check digits over a birth date (D-numbers add 40 to the day,
|
||||
/// H-numbers 40 to the month): strong enough to count alone.
|
||||
pub fn fnr_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
if d.len() != 11 {
|
||||
return false;
|
||||
}
|
||||
let check =
|
||||
|weights: &[u32]| match 11 - d.iter().zip(weights).map(|(a, w)| a * w).sum::<u32>() % 11 {
|
||||
11 => Some(0),
|
||||
10 => None,
|
||||
c => Some(c),
|
||||
};
|
||||
let day = d[0] * 10 + d[1];
|
||||
let month = d[2] * 10 + d[3];
|
||||
let day = if day > 40 { day - 40 } else { day };
|
||||
let month = if month > 40 { month - 40 } else { month };
|
||||
valid_short_date(d[4] * 10 + d[5], month, day)
|
||||
&& check(&[3, 7, 6, 1, 8, 9, 4, 5, 2]) == Some(d[9])
|
||||
&& check(&[5, 4, 3, 2, 7, 6, 5, 4, 3, 2]) == Some(d[10])
|
||||
}
|
||||
|
||||
fn no_fnr(text: &str, findings: &mut Findings) {
|
||||
for m in ELEVEN.find_iter(text) {
|
||||
let n = m.as_str().replace(' ', "");
|
||||
if fnr_valid(&n) {
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// `756.1234.5678.97`: the country prefix, then an EAN-13 check digit.
|
||||
static AHV: LazyLock<Regex> = LazyLock::new(|| {
|
||||
Regex::new(r"\b756[. ]?\d{4}[. ]?\d{4}[. ]?\d{2}\b").expect("detector pattern")
|
||||
});
|
||||
|
||||
pub fn ean13_valid(n: &str) -> bool {
|
||||
let d = digit_values(n);
|
||||
if d.len() != 13 {
|
||||
return false;
|
||||
}
|
||||
let sum: u32 = d[..12]
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, x)| if i % 2 == 0 { *x } else { x * 3 })
|
||||
.sum();
|
||||
(10 - sum % 10) % 10 == d[12]
|
||||
}
|
||||
|
||||
fn ch_ahv(text: &str, findings: &mut Findings) {
|
||||
for m in AHV.find_iter(text) {
|
||||
let n: String = m.as_str().chars().filter(char::is_ascii_digit).collect();
|
||||
if ean13_valid(&n) {
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
fn count(id: &str, text: &str) -> usize {
|
||||
by_id(id).unwrap().count(text)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn norway() {
|
||||
assert_eq!(count("no-fnr", "01019000083"), 1);
|
||||
assert_eq!(count("no-fnr", "010190 00083"), 1);
|
||||
assert_eq!(count("no-fnr", "01019000084"), 0);
|
||||
// Not a date
|
||||
assert_eq!(count("no-fnr", "32019000083"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn switzerland() {
|
||||
// The federal example
|
||||
assert_eq!(count("ch-ahv", "AHV 756.9217.0769.85"), 1);
|
||||
assert_eq!(count("ch-ahv", "7569217076985"), 1);
|
||||
assert_eq!(count("ch-ahv", "756.9217.0769.86"), 0);
|
||||
}
|
||||
}
|
||||
@@ -19,8 +19,18 @@
|
||||
//! same card number pasted twice counts once. They stay in memory: callers
|
||||
//! read only [`Findings::len`].
|
||||
|
||||
pub mod africa;
|
||||
pub mod americas;
|
||||
pub mod any;
|
||||
pub mod asia;
|
||||
pub mod australia;
|
||||
pub mod canada;
|
||||
pub mod checks;
|
||||
pub mod eu;
|
||||
pub mod europe;
|
||||
pub mod templates;
|
||||
pub mod uk;
|
||||
pub mod us;
|
||||
|
||||
use ahash::AHashSet;
|
||||
|
||||
@@ -108,7 +118,20 @@ impl Detector {
|
||||
|
||||
/// Every detector, in the order the console lists them.
|
||||
pub fn all() -> impl Iterator<Item = &'static Detector> {
|
||||
any::DETECTORS.iter()
|
||||
[
|
||||
any::DETECTORS,
|
||||
us::DETECTORS,
|
||||
uk::DETECTORS,
|
||||
canada::DETECTORS,
|
||||
australia::DETECTORS,
|
||||
eu::DETECTORS,
|
||||
europe::DETECTORS,
|
||||
asia::DETECTORS,
|
||||
americas::DETECTORS,
|
||||
africa::DETECTORS,
|
||||
]
|
||||
.into_iter()
|
||||
.flatten()
|
||||
}
|
||||
|
||||
pub fn by_id(id: &str) -> Option<&'static Detector> {
|
||||
@@ -159,6 +182,38 @@ pub fn stands_alone(text: &str, start: usize, end: usize) -> bool {
|
||||
before.is_none_or(|c| !c.is_alphanumeric()) && after.is_none_or(|c| !c.is_alphanumeric())
|
||||
}
|
||||
|
||||
/// Days in `month` of `year` (0 for a month that doesn't exist).
|
||||
pub fn days_in(year: u32, month: u32) -> u32 {
|
||||
match month {
|
||||
1 | 3 | 5 | 7 | 8 | 10 | 12 => 31,
|
||||
4 | 6 | 9 | 11 => 30,
|
||||
2 if year.is_multiple_of(4) && (!year.is_multiple_of(100) || year.is_multiple_of(400)) => {
|
||||
29
|
||||
}
|
||||
2 => 28,
|
||||
_ => 0,
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether `year`-`month`-`day` is a real date between 1900 and 2100.
|
||||
pub fn valid_date(year: u32, month: u32, day: u32) -> bool {
|
||||
(1900..=2100).contains(&year) && (1..=days_in(year, month)).contains(&day)
|
||||
}
|
||||
|
||||
/// Whether a two-digit year, month and day make a real date in either the
|
||||
/// 1900s or the 2000s.
|
||||
pub fn valid_short_date(yy: u32, month: u32, day: u32) -> bool {
|
||||
valid_date(1900 + yy, month, day) || valid_date(2000 + yy, month, day)
|
||||
}
|
||||
|
||||
/// The value of each digit in `s`.
|
||||
pub fn digit_values(s: &str) -> Vec<u32> {
|
||||
s.bytes()
|
||||
.filter(u8::is_ascii_digit)
|
||||
.map(|b| u32::from(b - b'0'))
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// The ASCII digits of `s`.
|
||||
pub fn digits(s: &str) -> String {
|
||||
s.chars().filter(char::is_ascii_digit).collect()
|
||||
@@ -188,6 +243,30 @@ mod tests {
|
||||
assert!(word_near(&text, start, start + 8, &["passport"]));
|
||||
}
|
||||
|
||||
/// An ordinary business email: order, invoice and tracking numbers,
|
||||
/// dates, amounts, a street address. Nothing here is an identifier, so
|
||||
/// no detector may fire, except the contact ones on the signature.
|
||||
#[test]
|
||||
fn ordinary_mail_finds_nothing() {
|
||||
let text = "Hi Dana,\n\nThanks for order 4471-2290 placed 2026-09-14. Invoice INV-2026-00917 \
|
||||
for $12,480.00 is due 10/31/2026; PO 7731902 covers lines 1-14. Tracking \
|
||||
1Z999AA10123456784, parcel 3 of 5, 12.5 kg, box 40x30x20 cm. Meeting moved to \
|
||||
Tuesday 9:30-10:15 in room 2B, building 1177. Ticket #5520318, case 20260914-0042. \
|
||||
Version 2026.9.28.4, build 118822, commit 5a73a118. Serial SN-88213-X. \
|
||||
Ship to 1600 Amphitheatre Pkwy, Mountain View, CA 94043. Revenue grew 18% to \
|
||||
1,204,332 units; see figures 3.1-3.4 and table 12.\n\nBest,\nSam\n\
|
||||
Sam Rivera | +1 (415) 555-2671 | [email protected]";
|
||||
let quiet = ["email-addresses", "phone-numbers"];
|
||||
for detector in all().filter(|d| !quiet.contains(&d.id)) {
|
||||
assert_eq!(
|
||||
detector.count(text),
|
||||
0,
|
||||
"{} fired on ordinary mail",
|
||||
detector.id
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ids_are_unique() {
|
||||
let mut seen = AHashSet::new();
|
||||
|
||||
@@ -0,0 +1,97 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Templates (§2.3): named sets of detectors, so a policy doesn't pick forty
|
||||
//! one at a time. Each is named for what it finds, never for a law, and is a
|
||||
//! starting point: once added to a rule, its detectors can be changed.
|
||||
|
||||
pub struct Template {
|
||||
pub id: &'static str,
|
||||
pub name: &'static str,
|
||||
pub detectors: &'static [&'static str],
|
||||
}
|
||||
|
||||
pub static TEMPLATES: &[Template] = &[
|
||||
Template {
|
||||
id: "payment-and-bank",
|
||||
name: "Payment cards and bank accounts",
|
||||
detectors: &["payment-card", "iban", "swift-bic", "us-aba-routing"],
|
||||
},
|
||||
Template {
|
||||
id: "us-personal",
|
||||
name: "US personal identifiers",
|
||||
detectors: &[
|
||||
"us-ssn",
|
||||
"us-itin",
|
||||
"us-ein",
|
||||
"us-drivers-license",
|
||||
"passport",
|
||||
"date-of-birth",
|
||||
],
|
||||
},
|
||||
Template {
|
||||
id: "uk-personal",
|
||||
name: "UK personal identifiers",
|
||||
detectors: &["uk-nino", "uk-utr", "uk-nhs", "passport", "date-of-birth"],
|
||||
},
|
||||
Template {
|
||||
id: "eu-national",
|
||||
name: "EU national identifiers",
|
||||
detectors: &[
|
||||
"de-tax-id",
|
||||
"de-id-card",
|
||||
"fr-nir",
|
||||
"es-dni-nie",
|
||||
"it-codice-fiscale",
|
||||
"nl-bsn",
|
||||
"be-national-number",
|
||||
"pl-pesel",
|
||||
"se-personnummer",
|
||||
"dk-cpr",
|
||||
"fi-hetu",
|
||||
"ie-pps",
|
||||
"pt-nif",
|
||||
"at-svnr",
|
||||
],
|
||||
},
|
||||
Template {
|
||||
id: "health",
|
||||
name: "Health identifiers",
|
||||
detectors: &["uk-nhs", "us-mbi", "us-npi", "us-dea", "au-medicare"],
|
||||
},
|
||||
Template {
|
||||
id: "credentials",
|
||||
name: "Credentials and keys",
|
||||
detectors: &["private-key", "credentials"],
|
||||
},
|
||||
Template {
|
||||
id: "contact-lists",
|
||||
name: "Contact lists",
|
||||
detectors: &["email-addresses", "phone-numbers"],
|
||||
},
|
||||
];
|
||||
|
||||
pub fn by_id(id: &str) -> Option<&'static Template> {
|
||||
TEMPLATES.iter().find(|template| template.id == id)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn every_template_names_real_detectors() {
|
||||
for template in TEMPLATES {
|
||||
for id in template.detectors {
|
||||
assert!(
|
||||
super::super::by_id(id).is_some(),
|
||||
"{}: no detector {id}",
|
||||
template.id
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,155 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! United Kingdom identifiers (§2.3): HMRC's National Insurance number and
|
||||
//! Unique Taxpayer Reference, and the NHS number.
|
||||
|
||||
use super::{Detector, Findings, Region, Strength, word_near};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[
|
||||
Detector::new(
|
||||
"uk-nino",
|
||||
"UK National Insurance number",
|
||||
Region::Uk,
|
||||
Strength::Checked,
|
||||
nino,
|
||||
),
|
||||
Detector::new(
|
||||
"uk-nhs",
|
||||
"UK NHS number",
|
||||
Region::Uk,
|
||||
Strength::Checked,
|
||||
nhs,
|
||||
),
|
||||
Detector::new(
|
||||
"uk-utr",
|
||||
"UK Unique Taxpayer Reference",
|
||||
Region::Uk,
|
||||
Strength::NeedsWord,
|
||||
utr,
|
||||
),
|
||||
];
|
||||
|
||||
fn re(pattern: &str) -> Regex {
|
||||
Regex::new(pattern).expect("detector pattern")
|
||||
}
|
||||
|
||||
/// Two letters, six digits (often in pairs), a suffix A–D.
|
||||
static NINO: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"(?i)\b([A-Z])([A-Z]) ?(\d{2}) ?(\d{2}) ?(\d{2}) ?([A-D])\b"));
|
||||
|
||||
/// HMRC's rules: D, F, I, Q, U and V are never used; O never second; and
|
||||
/// BG, GB, KN, NK, NT, TN and ZZ are never allocated.
|
||||
fn nino_prefix(first: char, second: char) -> bool {
|
||||
const NEVER: &str = "DFIQUV";
|
||||
let pair: String = [first, second].iter().collect();
|
||||
!NEVER.contains(first)
|
||||
&& !NEVER.contains(second)
|
||||
&& second != 'O'
|
||||
&& !["BG", "GB", "KN", "NK", "NT", "TN", "ZZ"].contains(&pair.as_str())
|
||||
}
|
||||
|
||||
fn nino(text: &str, findings: &mut Findings) {
|
||||
for c in NINO.captures_iter(text) {
|
||||
let first = c[1].to_ascii_uppercase().chars().next().unwrap();
|
||||
let second = c[2].to_ascii_uppercase().chars().next().unwrap();
|
||||
if nino_prefix(first, second) {
|
||||
findings.insert(format!(
|
||||
"{first}{second}{}{}{}{}",
|
||||
&c[3],
|
||||
&c[4],
|
||||
&c[5],
|
||||
c[6].to_ascii_uppercase()
|
||||
));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// `NNN NNN NNNN` stands alone; ten bare digits need a word.
|
||||
static NHS: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{3})([ -]?)(\d{3})([ -]?)(\d{4})\b"));
|
||||
|
||||
/// Mod 11: weights 10 down to 2 over the first nine digits; the check digit
|
||||
/// is 11 minus the remainder (11 becomes 0; 10 is never issued).
|
||||
pub fn nhs_valid(n: &str) -> bool {
|
||||
let d: Vec<u32> = n.bytes().map(|b| u32::from(b - b'0')).collect();
|
||||
let sum: u32 = d[..9].iter().zip((2..=10).rev()).map(|(a, w)| a * w).sum();
|
||||
match 11 - sum % 11 {
|
||||
11 => d[9] == 0,
|
||||
10 => false,
|
||||
check => d[9] == check,
|
||||
}
|
||||
}
|
||||
|
||||
const NHS_WORDS: &[&str] = &["nhs", "nhs number", "nhs no"];
|
||||
|
||||
fn nhs(text: &str, findings: &mut Findings) {
|
||||
for c in NHS.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
|
||||
let written = !c[2].is_empty() && c[2] == c[4];
|
||||
if nhs_valid(&n) && (written || word_near(text, whole.start(), whole.end(), NHS_WORDS)) {
|
||||
findings.insert(n);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static UTR: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{5} ?\d{5}\b"));
|
||||
|
||||
const UTR_WORDS: &[&str] = &[
|
||||
"utr",
|
||||
"unique taxpayer reference",
|
||||
"tax reference",
|
||||
"self assessment",
|
||||
];
|
||||
|
||||
fn utr(text: &str, findings: &mut Findings) {
|
||||
for m in UTR.find_iter(text) {
|
||||
if word_near(text, m.start(), m.end(), UTR_WORDS) {
|
||||
findings.insert(m.as_str().replace(' ', ""));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
fn count(id: &str, text: &str) -> usize {
|
||||
by_id(id).unwrap().count(text)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn national_insurance() {
|
||||
assert_eq!(count("uk-nino", "NI: AB 12 34 56 C, ce123456d"), 2);
|
||||
// Letters never used, pairs never allocated, a suffix past D
|
||||
for bad in [
|
||||
"QQ123456C",
|
||||
"AO123456C",
|
||||
"GB123456A",
|
||||
"AB123456E",
|
||||
"DA123456A",
|
||||
] {
|
||||
assert_eq!(count("uk-nino", bad), 0, "{bad}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn nhs_numbers() {
|
||||
// The NHS's own example
|
||||
assert_eq!(count("uk-nhs", "943 476 5919"), 1);
|
||||
assert_eq!(count("uk-nhs", "943 476 5918"), 0);
|
||||
assert_eq!(count("uk-nhs", "order 9434765919"), 0);
|
||||
assert_eq!(count("uk-nhs", "NHS number 9434765919"), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn utr() {
|
||||
assert_eq!(count("uk-utr", "UTR 12345 67890"), 1);
|
||||
assert_eq!(count("uk-utr", "order 1234567890"), 0);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,304 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! United States identifiers (§2.3), each from its issuer's published rules:
|
||||
//! the SSA (SSN), the IRS (ITIN, EIN), the ABA (routing numbers), CMS (MBI,
|
||||
//! NPI) and the DEA.
|
||||
|
||||
use super::{Detector, Findings, Region, Strength, checks, digits, stands_alone, word_near};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[
|
||||
Detector::new(
|
||||
"us-ssn",
|
||||
"US Social Security number",
|
||||
Region::Us,
|
||||
Strength::Checked,
|
||||
ssn,
|
||||
),
|
||||
Detector::new("us-itin", "US ITIN", Region::Us, Strength::Checked, itin),
|
||||
Detector::new("us-ein", "US EIN", Region::Us, Strength::NeedsWord, ein),
|
||||
Detector::new(
|
||||
"us-aba-routing",
|
||||
"US bank routing number",
|
||||
Region::Us,
|
||||
Strength::NeedsWord,
|
||||
aba_routing,
|
||||
),
|
||||
Detector::new(
|
||||
"us-drivers-license",
|
||||
"US driver's license",
|
||||
Region::Us,
|
||||
Strength::NeedsWord,
|
||||
drivers_license,
|
||||
),
|
||||
Detector::new(
|
||||
"us-mbi",
|
||||
"US Medicare Beneficiary Identifier",
|
||||
Region::Us,
|
||||
Strength::Checked,
|
||||
mbi,
|
||||
),
|
||||
Detector::new(
|
||||
"us-npi",
|
||||
"US National Provider Identifier",
|
||||
Region::Us,
|
||||
Strength::NeedsWord,
|
||||
npi,
|
||||
),
|
||||
Detector::new(
|
||||
"us-dea",
|
||||
"US DEA registration number",
|
||||
Region::Us,
|
||||
Strength::Checked,
|
||||
dea,
|
||||
),
|
||||
];
|
||||
|
||||
fn re(pattern: &str) -> Regex {
|
||||
Regex::new(pattern).expect("detector pattern")
|
||||
}
|
||||
|
||||
/// `AAA-GG-SSSS` (dashes or spaces), or nine bare digits.
|
||||
static NINE: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{3})([ -]?)(\d{2})([ -]?)(\d{4})\b"));
|
||||
|
||||
/// Numbers the SSA has published as never valid: widely printed examples.
|
||||
const SSN_EXAMPLES: &[&str] = &["078051120", "219099999"];
|
||||
|
||||
fn ssn_rules(area: u32, group: u32, serial: u32) -> bool {
|
||||
area != 0 && area != 666 && area < 900 && group != 0 && serial != 0
|
||||
}
|
||||
|
||||
const SSN_WORDS: &[&str] = &["ssn", "social security", "soc sec", "ss#", "ss no"];
|
||||
|
||||
fn ssn(text: &str, findings: &mut Findings) {
|
||||
for c in NINE.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let (area, group, serial) = (num(&c[1]), num(&c[3]), num(&c[5]));
|
||||
let number = format!("{}{}{}", &c[1], &c[3], &c[5]);
|
||||
// Written form (with both separators, the same one) stands alone;
|
||||
// nine bare digits need a word
|
||||
let written = !c[2].is_empty() && c[2] == c[4];
|
||||
if ssn_rules(area, group, serial)
|
||||
&& !SSN_EXAMPLES.contains(&number.as_str())
|
||||
&& (written || word_near(text, whole.start(), whole.end(), SSN_WORDS))
|
||||
{
|
||||
findings.insert(number);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// ITINs: 9XX, then a group in the IRS's ranges.
|
||||
fn itin_group(group: u32) -> bool {
|
||||
matches!(group, 50..=65 | 70..=88 | 90..=92 | 94..=99)
|
||||
}
|
||||
|
||||
const ITIN_WORDS: &[&str] = &["itin", "taxpayer identification", "tax id"];
|
||||
|
||||
fn itin(text: &str, findings: &mut Findings) {
|
||||
for c in NINE.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let written = !c[2].is_empty() && c[2] == c[4];
|
||||
if c[1].starts_with('9')
|
||||
&& itin_group(num(&c[3]))
|
||||
&& (written || word_near(text, whole.start(), whole.end(), ITIN_WORDS))
|
||||
{
|
||||
findings.insert(format!("{}{}{}", &c[1], &c[3], &c[5]));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static EIN: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{2})-?(\d{7})\b"));
|
||||
|
||||
/// The prefixes the IRS assigns to its campuses and internet EINs.
|
||||
fn ein_prefix(prefix: u32) -> bool {
|
||||
matches!(prefix, 1..=6 | 10..=16 | 20..=27 | 30..=48 | 50..=68 | 71..=77 | 80..=88 | 90..=95 | 98 | 99)
|
||||
}
|
||||
|
||||
const EIN_WORDS: &[&str] = &[
|
||||
"ein",
|
||||
"fein",
|
||||
"employer identification",
|
||||
"tax id",
|
||||
"tin",
|
||||
"federal tax",
|
||||
];
|
||||
|
||||
fn ein(text: &str, findings: &mut Findings) {
|
||||
for c in EIN.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
if ein_prefix(num(&c[1])) && word_near(text, whole.start(), whole.end(), EIN_WORDS) {
|
||||
findings.insert(format!("{}{}", &c[1], &c[2]));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static ROUTING: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{9}\b"));
|
||||
|
||||
/// The ABA check: 3, 7 and 1 weights, mod 10; and a Federal Reserve prefix.
|
||||
pub fn aba_valid(n: &str) -> bool {
|
||||
let d: Vec<u32> = n.bytes().map(|b| u32::from(b - b'0')).collect();
|
||||
let prefix = d[0] * 10 + d[1];
|
||||
matches!(prefix, 0..=12 | 21..=32 | 61..=72 | 80)
|
||||
&& (3 * (d[0] + d[3] + d[6]) + 7 * (d[1] + d[4] + d[7]) + (d[2] + d[5] + d[8]))
|
||||
.is_multiple_of(10)
|
||||
}
|
||||
|
||||
const ROUTING_WORDS: &[&str] = &["routing", "aba", "rtn", "routing number", "transit"];
|
||||
|
||||
fn aba_routing(text: &str, findings: &mut Findings) {
|
||||
for m in ROUTING.find_iter(text) {
|
||||
// One random number in ten passes the check: always needs a word
|
||||
if aba_valid(m.as_str()) && word_near(text, m.start(), m.end(), ROUTING_WORDS) {
|
||||
findings.insert(m.as_str());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The shapes states issue: up to two letters, then 5–14 digits, dashes
|
||||
/// allowed (Florida and Illinois print them).
|
||||
static LICENSE: LazyLock<Regex> = LazyLock::new(|| re(r"\b[A-Z]{0,2}\d[\d-]{3,16}\d\b"));
|
||||
|
||||
const LICENSE_WORDS: &[&str] = &[
|
||||
"driver's license",
|
||||
"drivers license",
|
||||
"driver license",
|
||||
"driver's licence",
|
||||
"dl",
|
||||
"dl#",
|
||||
"license number",
|
||||
"lic no",
|
||||
"dmv",
|
||||
];
|
||||
|
||||
fn drivers_license(text: &str, findings: &mut Findings) {
|
||||
for m in LICENSE.find_iter(text) {
|
||||
let n = digits(m.as_str());
|
||||
if (5..=14).contains(&n.len()) && word_near(text, m.start(), m.end(), LICENSE_WORDS) {
|
||||
findings.insert(m.as_str().replace('-', ""));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// CMS's MBI: 11 characters in a fixed pattern of digits, letters and
|
||||
/// either, the letters S, L, O, I, B and Z never used; dashes may follow the
|
||||
/// 4th and 7th.
|
||||
static MBI: LazyLock<Regex> = LazyLock::new(|| {
|
||||
let c = "[AC-HJKMNP-RT-Y]";
|
||||
let an = "[AC-HJKMNP-RT-Y0-9]";
|
||||
re(&format!(
|
||||
r"\b[1-9]{c}{an}[0-9]-?{c}{an}[0-9]-?{c}{c}[0-9][0-9]\b"
|
||||
))
|
||||
});
|
||||
|
||||
fn mbi(text: &str, findings: &mut Findings) {
|
||||
for m in MBI.find_iter(text) {
|
||||
findings.insert(m.as_str().replace('-', ""));
|
||||
}
|
||||
}
|
||||
|
||||
static TEN: LazyLock<Regex> = LazyLock::new(|| re(r"\b[12]\d{9}\b"));
|
||||
|
||||
const NPI_WORDS: &[&str] = &["npi", "national provider", "provider id", "provider number"];
|
||||
|
||||
/// NPI: Luhn over the ISO card-issuer prefix 80840 and the number.
|
||||
fn npi(text: &str, findings: &mut Findings) {
|
||||
for m in TEN.find_iter(text) {
|
||||
if checks::luhn(&format!("80840{}", m.as_str()))
|
||||
&& word_near(text, m.start(), m.end(), NPI_WORDS)
|
||||
{
|
||||
findings.insert(m.as_str());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static DEA: LazyLock<Regex> = LazyLock::new(|| re(r"\b([ABCDEFGHJKLMPRSTUX][A-Z9])(\d{7})\b"));
|
||||
|
||||
/// DEA: (1st + 3rd + 5th) + 2 × (2nd + 4th + 6th) ends in the 7th digit.
|
||||
fn dea(text: &str, findings: &mut Findings) {
|
||||
for c in DEA.captures_iter(text) {
|
||||
let d: Vec<u32> = c[2].bytes().map(|b| u32::from(b - b'0')).collect();
|
||||
if ((d[0] + d[2] + d[4]) + 2 * (d[1] + d[3] + d[5])) % 10 == d[6] {
|
||||
let whole = c.get(0).unwrap();
|
||||
if stands_alone(text, whole.start(), whole.end()) {
|
||||
findings.insert(whole.as_str());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn num(s: &str) -> u32 {
|
||||
s.parse().unwrap_or(0)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
fn count(id: &str, text: &str) -> usize {
|
||||
by_id(id).unwrap().count(text)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ssn() {
|
||||
assert_eq!(count("us-ssn", "SSN 536-22-1234, also 536 22 1235"), 2);
|
||||
// Bare digits: only with a word
|
||||
assert_eq!(count("us-ssn", "ref 536221234"), 0);
|
||||
assert_eq!(count("us-ssn", "social security: 536221234"), 1);
|
||||
// Never issued, the SSA's printed examples, mixed separators
|
||||
for bad in [
|
||||
"000-12-3456",
|
||||
"666-12-3456",
|
||||
"912-12-3456",
|
||||
"123-00-4567",
|
||||
"123-45-0000",
|
||||
"078-05-1120",
|
||||
"536-22 1234",
|
||||
] {
|
||||
assert_eq!(count("us-ssn", bad), 0, "{bad}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn itin_and_ein() {
|
||||
assert_eq!(count("us-itin", "912-70-1234"), 1);
|
||||
assert_eq!(count("us-itin", "912-69-1234"), 0);
|
||||
assert_eq!(count("us-ssn", "912-70-1234"), 0);
|
||||
assert_eq!(count("us-ein", "EIN: 12-3456789"), 1);
|
||||
assert_eq!(count("us-ein", "part 12-3456789"), 0);
|
||||
assert_eq!(count("us-ein", "EIN 07-3456789"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn routing_needs_a_word() {
|
||||
assert_eq!(count("us-aba-routing", "Routing number 011000015"), 1);
|
||||
assert_eq!(count("us-aba-routing", "ABA 021000021"), 1);
|
||||
assert_eq!(count("us-aba-routing", "invoice 011000015"), 0);
|
||||
assert_eq!(count("us-aba-routing", "routing 011000016"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn licenses() {
|
||||
assert_eq!(count("us-drivers-license", "Driver's license: D1234567"), 1);
|
||||
assert_eq!(count("us-drivers-license", "DL# S123-456-78-901-0"), 1);
|
||||
assert_eq!(count("us-drivers-license", "Order D1234567"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn health_identifiers() {
|
||||
// CMS's own MBI example
|
||||
assert_eq!(count("us-mbi", "Medicare 1EG4-TE5-MK73"), 1);
|
||||
assert_eq!(count("us-mbi", "1EG4TE5MK73"), 1);
|
||||
assert_eq!(count("us-mbi", "1EG4-TE5-MK7S"), 0);
|
||||
// CMS's NPI example
|
||||
assert_eq!(count("us-npi", "NPI 1234567893"), 1);
|
||||
assert_eq!(count("us-npi", "NPI 1234567894"), 0);
|
||||
assert_eq!(count("us-npi", "call 1234567893"), 0);
|
||||
assert_eq!(count("us-dea", "DEA AB1234563"), 1);
|
||||
assert_eq!(count("us-dea", "AB1234564"), 0);
|
||||
}
|
||||
}
|
||||
@@ -161,12 +161,14 @@ fn is_text(content_type: &str, extension: &str) -> bool {
|
||||
fn decode_text(data: &[u8]) -> String {
|
||||
let utf16 = |bytes: &[u8], big: bool| {
|
||||
let units: Vec<u16> = bytes
|
||||
.chunks_exact(2)
|
||||
.map(|c| {
|
||||
.as_chunks::<2>()
|
||||
.0
|
||||
.iter()
|
||||
.map(|&c| {
|
||||
if big {
|
||||
u16::from_be_bytes([c[0], c[1]])
|
||||
u16::from_be_bytes(c)
|
||||
} else {
|
||||
u16::from_le_bytes([c[0], c[1]])
|
||||
u16::from_le_bytes(c)
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
|
||||
@@ -139,6 +139,14 @@ DLP adds **detectors**. Each counts what it finds, and a rule sets a minimum
|
||||
either side, in the languages where the identifier is used ("passport",
|
||||
"Reisepass", "pasaporte"...).
|
||||
|
||||
A check that about one random number in ten passes (Luhn, mod 10, mod 11) is
|
||||
too weak for a bare run of digits: invoice and phone numbers would match. So
|
||||
a checked identifier that is only digits (SSN, SIN, NHS, TFN, Medicare…)
|
||||
counts alone in the written form it's issued in (`536-22-1234`,
|
||||
`130 692 544`, `943 476 5919`), and as bare digits only beside a word. ABA
|
||||
routing numbers and NPIs are never written with separators, so they always
|
||||
need a word. (Refinement made while building phase 2, 2026-09-28.)
|
||||
|
||||
The catalog (settled answer 6: the recognized, protected identifiers, not a
|
||||
chosen few). Each row is one table entry and one check function in
|
||||
`crates/features/src/mailflow/detectors/`:
|
||||
@@ -157,10 +165,10 @@ chosen few). Each row is one table entry and one check function in
|
||||
| US | Social Security number | Checked | `AAA-GG-SSSS`, or nine digits with a word; never area 000, 666 or 9xx, group 00, serial 0000 |
|
||||
| US | ITIN | Checked | 9XX-GG-SSSS with the IRS's group ranges |
|
||||
| US | EIN | Needs a word | a valid IRS prefix and seven digits |
|
||||
| US | Bank routing number (ABA) | Checked | nine digits, a valid Federal Reserve prefix, the 3-7-1 checksum |
|
||||
| US | Bank routing number (ABA) | Needs a word | nine digits, a valid Federal Reserve prefix, the 3-7-1 checksum |
|
||||
| US | Driver's license | Needs a word | each state's published format |
|
||||
| US | Medicare Beneficiary Identifier | Checked | CMS's 11-character pattern and excluded letters |
|
||||
| US | National Provider Identifier | Checked | ten digits, Luhn over the `80840` prefix |
|
||||
| US | National Provider Identifier | Needs a word | ten digits, Luhn over the `80840` prefix |
|
||||
| US | DEA registration number | Checked | two letters, seven digits, DEA's check digit |
|
||||
| UK | National Insurance number | Checked | two letters (HMRC's excluded prefixes), six digits, A–D |
|
||||
| UK | NHS number | Checked | ten digits, mod 11 |
|
||||
|
||||
Reference in New Issue
Block a user