DLP: regional identifiers and templates
ci / fork-checks (pull_request) Successful in 1m47s
ci / build (pull_request) Successful in 7m42s

Phase 2b of the DLP and mail flow rules spec: every identifier in the
§2.3 catalog, each implemented from its issuer's published rules and
tested against published examples.

US (SSN, ITIN, EIN, ABA routing, driver's licenses, MBI, NPI, DEA), UK
(NI number, NHS number, UTR), Canada (SIN), Australia (TFN, Medicare),
the EU (Germany's tax ID and ID card, France's NIR, Spain's DNI/NIE,
Italy's codice fiscale, the Dutch BSN, Belgium's national number,
Poland's PESEL, Sweden's personnummer, Denmark's CPR, Finland's HETU,
Ireland's PPS, Portugal's NIF, Austria's SVNR), Norway, Switzerland,
India (Aadhaar, PAN), China, Japan, Singapore, South Korea, Brazil (CPF,
CNPJ), Mexico (CURP) and South Africa. 49 detectors in all, plus seven
templates named for what they find.

An identifier that is only digits and whose check about one random
number in ten passes counts alone only in its written form
(536-22-1234, 943 476 5919) and as bare digits only beside a word; ABA
routing numbers and NPIs always need one. Spec §2.3 records this.

A test runs every detector over an ordinary business email (order,
invoice and tracking numbers, dates, amounts, an address) and requires
nothing to fire but the contact detectors. 47 unit tests.
This commit is contained in:
2026-09-28 17:13:42 -07:00
parent 01f6b99631
commit 92d14fbd60
15 changed files with 2114 additions and 22 deletions
@@ -0,0 +1,49 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! African identifiers (§2.3): South Africa's ID number.
use super::{Detector, Findings, Region, Strength, checks, valid_short_date};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[Detector::new(
"za-id",
"South Africa: ID number",
Region::Africa,
Strength::Checked,
za_id,
)];
/// Birth date `YYMMDD`, four digits, citizenship (0, 1 or 2), 8 or 9, a Luhn
/// check digit. The date and the two fixed digits make it strong enough to
/// count alone.
static ZA_ID: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(r"\b(\d{2})(\d{2})(\d{2})\d{4}[012][89]\d\b").expect("detector pattern")
});
fn za_id(text: &str, findings: &mut Findings) {
for c in ZA_ID.captures_iter(text) {
let n = &c[0];
let num = |s: &str| s.parse::<u32>().unwrap_or(0);
if valid_short_date(num(&c[1]), num(&c[2]), num(&c[3])) && checks::luhn(n) {
findings.insert(n);
}
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
#[test]
fn south_africa() {
let detector = by_id("za-id").unwrap();
assert_eq!(detector.count("ID 8001015009087"), 1);
assert_eq!(detector.count("8001015009088"), 0);
assert_eq!(detector.count("8013015009087"), 0);
}
}
@@ -0,0 +1,172 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Identifiers from the Americas outside the US and Canada (§2.3): Brazil's
//! CPF and CNPJ, and Mexico's CURP.
use super::{Detector, Findings, Region, Strength, digit_values, valid_short_date, word_near};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"br-cpf",
"Brazil: CPF",
Region::Americas,
Strength::Checked,
br_cpf,
),
Detector::new(
"br-cnpj",
"Brazil: CNPJ",
Region::Americas,
Strength::Checked,
br_cnpj,
),
Detector::new(
"mx-curp",
"Mexico: CURP",
Region::Americas,
Strength::Checked,
mx_curp,
),
];
fn re(pattern: &str) -> Regex {
Regex::new(pattern).expect("detector pattern")
}
/// Brazil's mod 11 check digit over `digits` with `weights`.
fn br_check(digits: &[u32], weights: &[u32]) -> u32 {
match digits.iter().zip(weights).map(|(a, w)| a * w).sum::<u32>() % 11 {
0 | 1 => 0,
r => 11 - r,
}
}
/// `111.444.777-35`, or eleven bare digits.
static CPF: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{3}(\.?)\d{3}(\.?)\d{3}(-?)\d{2}\b"));
pub fn cpf_valid(n: &str) -> bool {
let d = digit_values(n);
// A run of one digit passes the arithmetic but is never issued
d.len() == 11
&& d.iter().any(|x| *x != d[0])
&& br_check(&d[..9], &[10, 9, 8, 7, 6, 5, 4, 3, 2]) == d[9]
&& br_check(&d[..10], &[11, 10, 9, 8, 7, 6, 5, 4, 3, 2]) == d[10]
}
const CPF_WORDS: &[&str] = &[
"cpf",
"cadastro de pessoas físicas",
"cadastro de pessoa física",
];
fn br_cpf(text: &str, findings: &mut Findings) {
for c in CPF.captures_iter(text) {
let whole = c.get(0).unwrap();
let written = &c[1] == "." && &c[2] == "." && &c[3] == "-";
let n: String = whole
.as_str()
.chars()
.filter(char::is_ascii_digit)
.collect();
if cpf_valid(&n) && (written || word_near(text, whole.start(), whole.end(), CPF_WORDS)) {
findings.insert(n);
}
}
}
/// `11.222.333/0001-81`, or fourteen bare digits.
static CNPJ: LazyLock<Regex> =
LazyLock::new(|| re(r"\b\d{2}(\.?)\d{3}(\.?)\d{3}(/?)\d{4}(-?)\d{2}\b"));
pub fn cnpj_valid(n: &str) -> bool {
let d = digit_values(n);
d.len() == 14
&& d.iter().any(|x| *x != d[0])
&& br_check(&d[..12], &[5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2]) == d[12]
&& br_check(&d[..13], &[6, 5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2]) == d[13]
}
const CNPJ_WORDS: &[&str] = &["cnpj", "cadastro nacional da pessoa jurídica"];
fn br_cnpj(text: &str, findings: &mut Findings) {
for c in CNPJ.captures_iter(text) {
let whole = c.get(0).unwrap();
let written = &c[1] == "." && &c[2] == "." && &c[3] == "/" && &c[4] == "-";
let n: String = whole
.as_str()
.chars()
.filter(char::is_ascii_digit)
.collect();
if cnpj_valid(&n) && (written || word_near(text, whole.start(), whole.end(), CNPJ_WORDS)) {
findings.insert(n);
}
}
}
/// Four letters, the birth date, sex (H, M or X), the state, three
/// consonants, a character that tells the century apart, the check digit.
static CURP: LazyLock<Regex> = LazyLock::new(|| {
re(r"(?i)\b[A-Z]{4}(\d{2})(\d{2})(\d{2})[HMX][A-Z]{2}[B-DF-HJ-NP-TV-Z]{3}[A-Z0-9]\d\b")
});
/// RENAPO's check: each character's place in `0-9 A-N Ñ O-Z`, weighted 18
/// down to 2; the digit is 10 minus the sum mod 10 (10 becomes 0).
pub fn curp_valid(curp: &str) -> bool {
const ALPHABET: &str = "0123456789ABCDEFGHIJKLMNÑOPQRSTUVWXYZ";
let mut sum = 0u32;
for (i, c) in curp.chars().take(17).enumerate() {
let Some(value) = ALPHABET.chars().position(|a| a == c) else {
return false;
};
sum += value as u32 * (18 - i as u32);
}
curp.chars().nth(17).and_then(|c| c.to_digit(10)) == Some((10 - sum % 10) % 10)
}
fn mx_curp(text: &str, findings: &mut Findings) {
for c in CURP.captures_iter(text) {
let curp = c[0].to_ascii_uppercase();
if valid_short_date(num(&c[1]), num(&c[2]), num(&c[3])) && curp_valid(&curp) {
findings.insert(curp);
}
}
}
fn num(s: &str) -> u32 {
s.parse().unwrap_or(0)
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn brazil() {
assert_eq!(count("br-cpf", "CPF 111.444.777-35"), 1);
assert_eq!(count("br-cpf", "111.444.777-36"), 0);
assert_eq!(count("br-cpf", "pedido 11144477735"), 0);
assert_eq!(count("br-cpf", "cpf: 11144477735"), 1);
assert_eq!(count("br-cpf", "CPF 111.111.111-11"), 0);
assert_eq!(count("br-cnpj", "11.222.333/0001-81"), 1);
assert_eq!(count("br-cnpj", "11.222.333/0001-82"), 0);
assert_eq!(count("br-cnpj", "CNPJ 11222333000181"), 1);
}
#[test]
fn mexico() {
// python-stdnum's documented example
assert_eq!(count("mx-curp", "CURP BOXW310820HNERXN09"), 1);
assert_eq!(count("mx-curp", "BOXW310820HNERXN08"), 0);
assert_eq!(count("mx-curp", "BOXW311320HNERXN09"), 0);
}
}
+3 -14
View File
@@ -6,7 +6,9 @@
//! Detectors that aren't tied to one country (§2.3, region "Any").
use super::{Detector, Findings, Region, Strength, checks, digits, stands_alone, word_near};
use super::{
Detector, Findings, Region, Strength, checks, digits, stands_alone, valid_date, word_near,
};
use regex::Regex;
use std::sync::LazyLock;
@@ -295,19 +297,6 @@ const BIRTH_WORDS: &[&str] = &[
"nascimento",
];
fn valid_date(year: u32, month: u32, day: u32) -> bool {
let days = match month {
1 | 3 | 5 | 7 | 8 | 10 | 12 => 31,
4 | 6 | 9 | 11 => 30,
2 if year.is_multiple_of(4) && (!year.is_multiple_of(100) || year.is_multiple_of(400)) => {
29
}
2 => 28,
_ => return false,
};
(1900..=2100).contains(&year) && (1..=days).contains(&day)
}
fn month_number(name: &str) -> u32 {
const MONTHS: [&str; 12] = [
"jan", "feb", "mar", "apr", "may", "jun", "jul", "aug", "sep", "oct", "nov", "dec",
@@ -0,0 +1,281 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Asian identifiers (§2.3): India's Aadhaar and PAN, China's resident ID,
//! Japan's My Number, Singapore's NRIC and FIN, and South Korea's resident
//! registration number.
use super::{
Detector, Findings, Region, Strength, digit_values, valid_date, valid_short_date, word_near,
};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"in-aadhaar",
"India: Aadhaar",
Region::Asia,
Strength::Checked,
in_aadhaar,
),
Detector::new(
"in-pan",
"India: PAN",
Region::Asia,
Strength::NeedsWord,
in_pan,
),
Detector::new(
"cn-resident-id",
"China: resident ID",
Region::Asia,
Strength::Checked,
cn_resident_id,
),
Detector::new(
"jp-my-number",
"Japan: My Number",
Region::Asia,
Strength::Checked,
jp_my_number,
),
Detector::new(
"sg-nric",
"Singapore: NRIC and FIN",
Region::Asia,
Strength::Checked,
sg_nric,
),
Detector::new(
"kr-rrn",
"South Korea: resident registration number",
Region::Asia,
Strength::NeedsWord,
kr_rrn,
),
];
fn re(pattern: &str) -> Regex {
Regex::new(pattern).expect("detector pattern")
}
/// Twelve digits written in fours, or bare.
static TWELVE: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{4})( ?)(\d{4})( ?)(\d{4})\b"));
const VERHOEFF_D: [[u8; 10]; 10] = [
[0, 1, 2, 3, 4, 5, 6, 7, 8, 9],
[1, 2, 3, 4, 0, 6, 7, 8, 9, 5],
[2, 3, 4, 0, 1, 7, 8, 9, 5, 6],
[3, 4, 0, 1, 2, 8, 9, 5, 6, 7],
[4, 0, 1, 2, 3, 9, 5, 6, 7, 8],
[5, 9, 8, 7, 6, 0, 4, 3, 2, 1],
[6, 5, 9, 8, 7, 1, 0, 4, 3, 2],
[7, 6, 5, 9, 8, 2, 1, 0, 4, 3],
[8, 7, 6, 5, 9, 3, 2, 1, 0, 4],
[9, 8, 7, 6, 5, 4, 3, 2, 1, 0],
];
const VERHOEFF_P: [[u8; 10]; 8] = [
[0, 1, 2, 3, 4, 5, 6, 7, 8, 9],
[1, 5, 7, 6, 2, 8, 3, 0, 9, 4],
[5, 8, 0, 3, 7, 9, 6, 1, 4, 2],
[8, 9, 1, 6, 0, 4, 3, 5, 2, 7],
[9, 4, 5, 3, 1, 2, 6, 8, 7, 0],
[4, 2, 8, 6, 5, 7, 3, 9, 0, 1],
[2, 7, 9, 3, 8, 0, 6, 4, 1, 5],
[7, 0, 4, 6, 9, 1, 3, 2, 5, 8],
];
/// The Verhoeff check (dihedral group D5).
pub fn verhoeff(n: &str) -> bool {
let mut c = 0u8;
for (i, b) in n.bytes().rev().enumerate() {
c = VERHOEFF_D[c as usize][VERHOEFF_P[i % 8][(b - b'0') as usize] as usize];
}
c == 0
}
const AADHAAR_WORDS: &[&str] = &["aadhaar", "aadhar", "uidai", "uid"];
fn in_aadhaar(text: &str, findings: &mut Findings) {
for c in TWELVE.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = &c[2] == " " && &c[4] == " ";
// Never starts with 0 or 1
if !n.starts_with(['0', '1'])
&& verhoeff(&n)
&& (written || word_near(text, whole.start(), whole.end(), AADHAAR_WORDS))
{
findings.insert(n);
}
}
}
/// Five letters (the fourth names the holder's type), four digits, a letter.
static PAN: LazyLock<Regex> = LazyLock::new(|| re(r"\b[A-Z]{3}[ABCFGHLJPTK][A-Z]\d{4}[A-Z]\b"));
const PAN_WORDS: &[&str] = &["pan", "pan card", "permanent account number", "income tax"];
fn in_pan(text: &str, findings: &mut Findings) {
for m in PAN.find_iter(text) {
if word_near(text, m.start(), m.end(), PAN_WORDS) {
findings.insert(m.as_str());
}
}
}
/// Region, birth date `YYYYMMDD`, sequence, then the ISO 7064 MOD 11-2
/// check (0–9 or X).
static CN_ID: LazyLock<Regex> =
LazyLock::new(|| re(r"(?i)\b[1-8]\d{5}(\d{4})(\d{2})(\d{2})\d{3}[\dX]\b"));
pub fn cn_id_valid(id: &str) -> bool {
const WEIGHTS: [u32; 17] = [7, 9, 10, 5, 8, 4, 2, 1, 6, 3, 7, 9, 10, 5, 8, 4, 2];
const CHECKS: &[u8] = b"10X98765432";
let sum: u32 = digit_values(&id[..17])
.iter()
.zip(WEIGHTS)
.map(|(a, w)| a * w)
.sum();
CHECKS[(sum % 11) as usize] == id.as_bytes()[17].to_ascii_uppercase()
}
fn cn_resident_id(text: &str, findings: &mut Findings) {
for c in CN_ID.captures_iter(text) {
let id = c[0].to_ascii_uppercase();
let (y, m, d) = (num(&c[1]), num(&c[2]), num(&c[3]));
if valid_date(y, m, d) && cn_id_valid(&id) {
findings.insert(id);
}
}
}
/// My Number: weights 2–7 then 2–6 from the right; a remainder of 0 or 1
/// gives 0, else 11 minus it.
pub fn my_number_valid(n: &str) -> bool {
let d = digit_values(n);
let sum: u32 = (1..=11)
.map(|i| d[11 - i] * if i <= 6 { i as u32 + 1 } else { i as u32 - 5 })
.sum();
let check = match sum % 11 {
0 | 1 => 0,
r => 11 - r,
};
check == d[11]
}
const MY_NUMBER_WORDS: &[&str] = &[
"my number",
"mynumber",
"マイナンバー",
"個人番号",
"kojin bango",
];
fn jp_my_number(text: &str, findings: &mut Findings) {
for c in TWELVE.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = &c[2] == " " && &c[4] == " ";
if my_number_valid(&n)
&& (written || word_near(text, whole.start(), whole.end(), MY_NUMBER_WORDS))
{
findings.insert(n);
}
}
}
static NRIC: LazyLock<Regex> = LazyLock::new(|| re(r"(?i)\b([STFGM])(\d{7})([A-Z])\b"));
/// Weights 2, 7, 6, 5, 4, 3, 2; T and G add 4, M adds 3; each series has its
/// own table of check letters.
fn nric_valid(prefix: u8, digits: &str, check: u8) -> bool {
let sum: u32 = digit_values(digits)
.iter()
.zip([2, 7, 6, 5, 4, 3, 2])
.map(|(a, w)| a * w)
.sum::<u32>()
+ match prefix {
b'T' | b'G' => 4,
b'M' => 3,
_ => 0,
};
let table: &[u8] = match prefix {
b'S' | b'T' => b"JZIHGFEDCBA",
b'F' | b'G' => b"XWUTRQPNMLK",
_ => b"KLJNPQRTUWX",
};
table[(sum % 11) as usize] == check
}
fn sg_nric(text: &str, findings: &mut Findings) {
for c in NRIC.captures_iter(text) {
let id = c[0].to_ascii_uppercase();
let bytes = id.as_bytes();
if nric_valid(bytes[0], &c[2], bytes[8]) {
findings.insert(id);
}
}
}
/// `YYMMDD-GNNNNNN`, the seventh digit giving sex and century.
static RRN: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{2})(\d{2})(\d{2})-?([1-8])\d{6}\b"));
const RRN_WORDS: &[&str] = &["주민등록번호", "주민번호", "resident registration", "rrn"];
fn kr_rrn(text: &str, findings: &mut Findings) {
for c in RRN.captures_iter(text) {
let whole = c.get(0).unwrap();
if valid_short_date(num(&c[1]), num(&c[2]), num(&c[3]))
&& word_near(text, whole.start(), whole.end(), RRN_WORDS)
{
findings.insert(whole.as_str().replace('-', ""));
}
}
}
fn num(s: &str) -> u32 {
s.parse().unwrap_or(0)
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn india() {
assert_eq!(count("in-aadhaar", "2345 6789 0124"), 1);
assert_eq!(count("in-aadhaar", "2345 6789 0125"), 0);
assert_eq!(count("in-aadhaar", "order 234567890124"), 0);
assert_eq!(count("in-aadhaar", "Aadhaar 234567890124"), 1);
assert_eq!(count("in-pan", "PAN: ABCPE1234F"), 1);
assert_eq!(count("in-pan", "ABCPE1234F"), 0);
}
#[test]
fn china_japan() {
assert_eq!(count("cn-resident-id", "11010519491231002X"), 1);
assert_eq!(count("cn-resident-id", "110105194912310021"), 0);
assert_eq!(count("cn-resident-id", "11010519491331002X"), 0);
assert_eq!(count("jp-my-number", "1234 5678 9018"), 1);
assert_eq!(count("jp-my-number", "1234 5678 9017"), 0);
assert_eq!(count("jp-my-number", "マイナンバー 123456789018"), 1);
}
#[test]
fn singapore_korea() {
assert_eq!(count("sg-nric", "S1234567D and T1234567J"), 2);
assert_eq!(count("sg-nric", "S1234567E"), 0);
assert_eq!(count("kr-rrn", "주민등록번호 800101-1234567"), 1);
assert_eq!(count("kr-rrn", "800101-1234567"), 0);
assert_eq!(count("kr-rrn", "RRN 801301-1234567"), 0);
}
}
@@ -0,0 +1,128 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Australian identifiers (§2.3): the ATO's Tax File Number and the Medicare
//! card number.
use super::{Detector, Findings, Region, Strength, word_near};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"au-tfn",
"Australian Tax File Number",
Region::Australia,
Strength::Checked,
tfn,
),
Detector::new(
"au-medicare",
"Australian Medicare number",
Region::Australia,
Strength::Checked,
medicare,
),
];
fn re(pattern: &str) -> Regex {
Regex::new(pattern).expect("detector pattern")
}
/// `NNN NNN NNN` stands alone; bare digits (eight or nine) need a word.
static TFN: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{3})( ?)(\d{3})( ?)(\d{2,3})\b"));
/// Weighted sum mod 11, with the ATO's weights for 9- and 8-digit numbers.
pub fn tfn_valid(n: &str) -> bool {
let weights: &[u32] = match n.len() {
9 => &[1, 4, 3, 7, 5, 8, 6, 9, 10],
8 => &[10, 7, 8, 4, 6, 3, 5, 1],
_ => return false,
};
n.bytes()
.zip(weights)
.map(|(b, w)| u32::from(b - b'0') * w)
.sum::<u32>()
% 11
== 0
}
const TFN_WORDS: &[&str] = &["tfn", "tax file number", "tax file no"];
fn tfn(text: &str, findings: &mut Findings) {
for c in TFN.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = n.len() == 9 && c[2] == *" " && c[4] == *" ";
if tfn_valid(&n) && (written || word_near(text, whole.start(), whole.end(), TFN_WORDS)) {
findings.insert(n);
}
}
}
/// `NNNN NNNNN N` (and an optional issue number) stands alone; bare digits
/// need a word.
static MEDICARE: LazyLock<Regex> =
LazyLock::new(|| re(r"\b([2-6]\d{3})( ?)(\d{5})( ?)(\d)(?:[ -]?\d)?\b"));
/// The ninth digit is the weighted sum (1, 3, 7, 9, 1, 3, 7, 9) of the first
/// eight, mod 10.
pub fn medicare_valid(n: &str) -> bool {
let d: Vec<u32> = n.bytes().map(|b| u32::from(b - b'0')).collect();
d.len() >= 9
&& d[..8]
.iter()
.zip([1, 3, 7, 9, 1, 3, 7, 9])
.map(|(a, w)| a * w)
.sum::<u32>()
% 10
== d[8]
}
const MEDICARE_WORDS: &[&str] = &[
"medicare",
"medicare card",
"medicare no",
"medicare number",
];
fn medicare(text: &str, findings: &mut Findings) {
for c in MEDICARE.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = c[2] == *" " && c[4] == *" ";
if medicare_valid(&n)
&& (written || word_near(text, whole.start(), whole.end(), MEDICARE_WORDS))
{
findings.insert(n);
}
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn tax_file_numbers() {
assert_eq!(count("au-tfn", "TFN 123 456 782"), 1);
assert_eq!(count("au-tfn", "123 456 789"), 0);
assert_eq!(count("au-tfn", "order 123456782"), 0);
assert_eq!(count("au-tfn", "tax file number 123456782"), 1);
}
#[test]
fn medicare_numbers() {
assert_eq!(count("au-medicare", "2123 45670 1"), 1);
assert_eq!(count("au-medicare", "2123 45671 1"), 0);
assert_eq!(count("au-medicare", "ref 2123456701"), 0);
assert_eq!(count("au-medicare", "Medicare 2123456701"), 1);
}
}
@@ -0,0 +1,66 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Canadian identifiers (§2.3): the Social Insurance Number.
use super::{Detector, Findings, Region, Strength, checks, word_near};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[Detector::new(
"ca-sin",
"Canadian Social Insurance Number",
Region::Canada,
Strength::Checked,
sin,
)];
/// `NNN NNN NNN` or `NNN-NNN-NNN` stands alone; nine bare digits need a word.
static SIN: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(r"\b(\d{3})([ -]?)(\d{3})([ -]?)(\d{3})\b").expect("detector pattern")
});
const SIN_WORDS: &[&str] = &[
"sin",
"social insurance",
"nas",
"numéro d'assurance sociale",
"assurance sociale",
];
fn sin(text: &str, findings: &mut Findings) {
for c in SIN.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = !c[2].is_empty() && c[2] == c[4];
// 0 and 8 are never issued as a first digit
if !n.starts_with(['0', '8'])
&& checks::luhn(&n)
&& (written || word_near(text, whole.start(), whole.end(), SIN_WORDS))
{
findings.insert(n);
}
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(text: &str) -> usize {
by_id("ca-sin").unwrap().count(text)
}
#[test]
fn social_insurance_numbers() {
assert_eq!(count("130 692 544 and 193-456-787"), 2);
assert_eq!(count("130 692 545"), 0);
// The government's printed example starts with 0, never issued
assert_eq!(count("046 454 286"), 0);
assert_eq!(count("order 130692544"), 0);
assert_eq!(count("SIN: 130692544"), 1);
}
}
@@ -25,7 +25,7 @@ pub fn luhn(digits: &str) -> bool {
}
})
.sum();
sum % 10 == 0
sum.is_multiple_of(10)
}
/// ISO 13616 IBAN lengths, by country, from the IBAN registry.
@@ -0,0 +1,646 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! European Union national identifiers (§2.3), each from its issuer's
//! published rules. An identifier that is only digits and whose check a
//! random number passes often (mod 10, mod 11) counts alone only in its
//! written form, and as bare digits only beside a word.
use super::{
Detector, Findings, Region, Strength, checks, digit_values, stands_alone, valid_short_date,
word_near,
};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"de-tax-id",
"Germany: tax ID (Steuer-ID)",
Region::Eu,
Strength::Checked,
de_tax_id,
),
Detector::new(
"de-id-card",
"Germany: ID card number",
Region::Eu,
Strength::Checked,
de_id_card,
),
Detector::new(
"fr-nir",
"France: social security number (NIR)",
Region::Eu,
Strength::Checked,
fr_nir,
),
Detector::new(
"es-dni-nie",
"Spain: DNI and NIE",
Region::Eu,
Strength::Checked,
es_dni_nie,
),
Detector::new(
"it-codice-fiscale",
"Italy: codice fiscale",
Region::Eu,
Strength::Checked,
it_codice_fiscale,
),
Detector::new(
"nl-bsn",
"Netherlands: BSN",
Region::Eu,
Strength::Checked,
nl_bsn,
),
Detector::new(
"be-national-number",
"Belgium: national number",
Region::Eu,
Strength::Checked,
be_national_number,
),
Detector::new(
"pl-pesel",
"Poland: PESEL",
Region::Eu,
Strength::Checked,
pl_pesel,
),
Detector::new(
"se-personnummer",
"Sweden: personnummer",
Region::Eu,
Strength::Checked,
se_personnummer,
),
Detector::new(
"dk-cpr",
"Denmark: CPR number",
Region::Eu,
Strength::NeedsWord,
dk_cpr,
),
Detector::new(
"fi-hetu",
"Finland: personal identity code",
Region::Eu,
Strength::Checked,
fi_hetu,
),
Detector::new(
"ie-pps",
"Ireland: PPS number",
Region::Eu,
Strength::Checked,
ie_pps,
),
Detector::new(
"pt-nif",
"Portugal: NIF",
Region::Eu,
Strength::Checked,
pt_nif,
),
Detector::new(
"at-svnr",
"Austria: social insurance number",
Region::Eu,
Strength::Checked,
at_svnr,
),
];
fn re(pattern: &str) -> Regex {
Regex::new(pattern).expect("detector pattern")
}
fn num(s: &str) -> u32 {
s.parse().unwrap_or(0)
}
// --- Germany --------------------------------------------------------------
/// Eleven digits, written `86 095 742 719` on the BZSt's letters.
static DE_TAX: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{2}( ?)\d{3}( ?)\d{3}( ?)\d{3}\b"));
/// ISO 7064 MOD 11,10; no leading zero; in the first ten digits one digit
/// appears two or three times and every other at most once.
pub fn de_tax_id_valid(n: &str) -> bool {
let d = digit_values(n);
if d.len() != 11 || d[0] == 0 {
return false;
}
let mut counts = [0u8; 10];
for &x in &d[..10] {
counts[x as usize] += 1;
}
let repeated = counts.iter().filter(|&&c| c >= 2).count();
if repeated != 1 || counts.iter().any(|&c| c > 3) {
return false;
}
let mut product = 10;
for &x in &d[..10] {
let mut sum = (x + product) % 10;
if sum == 0 {
sum = 10;
}
product = (2 * sum) % 11;
}
let check = match 11 - product {
10 => 0,
c => c,
};
check == d[10]
}
const DE_TAX_WORDS: &[&str] = &[
"steuer-id",
"steueridentifikationsnummer",
"steuerliche identifikationsnummer",
"idnr",
"identifikationsnummer",
"tax id",
];
fn de_tax_id(text: &str, findings: &mut Findings) {
for c in DE_TAX.captures_iter(text) {
let whole = c.get(0).unwrap();
let written = [&c[1], &c[2], &c[3]].iter().all(|s| *s == " ");
let n: String = whole.as_str().replace(' ', "");
if de_tax_id_valid(&n)
&& (written || word_near(text, whole.start(), whole.end(), DE_TAX_WORDS))
{
findings.insert(n);
}
}
}
/// The ID card's document number: a letter from the card's alphabet, eight
/// more characters from it, then the check digit.
static DE_ID: LazyLock<Regex> =
LazyLock::new(|| re(r"\b[CFGHJKLMNPRTVWXYZ][CFGHJKLMNPRTVWXYZ0-9]{8}\d\b"));
/// ICAO 9303 check digit: weights 7, 3, 1; letters A=10 … Z=35.
pub fn icao_check(chars: &str, check: u32) -> bool {
let value = |c: char| c.to_digit(10).unwrap_or_else(|| c as u32 - 'A' as u32 + 10);
let sum: u32 = chars
.chars()
.zip([7, 3, 1].iter().cycle())
.map(|(c, w)| value(c) * w)
.sum();
sum % 10 == check
}
fn de_id_card(text: &str, findings: &mut Findings) {
for m in DE_ID.find_iter(text) {
let s = m.as_str();
if icao_check(&s[..9], num(&s[9..])) {
findings.insert(s);
}
}
}
// --- France ---------------------------------------------------------------
/// Sex, year, month, department (with Corsica's 2A and 2B), commune, order,
/// then the two-digit key, spaces allowed between groups.
static FR_NIR: LazyLock<Regex> = LazyLock::new(|| {
re(r"\b([1-478]) ?(\d{2}) ?(\d{2}) ?(\d{2}|2[AB]) ?(\d{3}) ?(\d{3}) ?(\d{2})\b")
});
fn fr_nir(text: &str, findings: &mut Findings) {
for c in FR_NIR.captures_iter(text) {
let month = num(&c[3]);
if !(matches!(month, 1..=12 | 20..=42 | 50..=99)) {
continue;
}
let department = match &c[4] {
"2A" => "19",
"2B" => "18",
d => d,
};
let body = format!(
"{}{}{}{}{}{}",
&c[1], &c[2], &c[3], department, &c[5], &c[6]
);
let Ok(value) = body.parse::<u64>() else {
continue;
};
if 97 - value % 97 == u64::from(num(&c[7])) {
findings.insert(format!(
"{}{}{}{}{}{}{}",
&c[1], &c[2], &c[3], &c[4], &c[5], &c[6], &c[7]
));
}
}
}
// --- Spain ----------------------------------------------------------------
static ES_ID: LazyLock<Regex> = LazyLock::new(|| re(r"(?i)\b([XYZ]?)[ -]?(\d{7,8})[ -]?([A-Z])\b"));
const DNI_LETTERS: &[u8] = b"TRWAGMYFPDXBNJZSQVHLCKE";
fn es_dni_nie(text: &str, findings: &mut Findings) {
for c in ES_ID.captures_iter(text) {
let prefix = c[1].to_ascii_uppercase();
let digits = &c[2];
// DNI: eight digits; NIE: X, Y or Z and seven digits
let number = match (prefix.as_str(), digits.len()) {
("", 8) => digits.to_string(),
("X", 7) => format!("0{digits}"),
("Y", 7) => format!("1{digits}"),
("Z", 7) => format!("2{digits}"),
_ => continue,
};
let letter = c[3].to_ascii_uppercase();
if DNI_LETTERS[(num(&number) % 23) as usize] == letter.as_bytes()[0] {
findings.insert(format!("{prefix}{digits}{letter}"));
}
}
}
// --- Italy ----------------------------------------------------------------
/// Surname and name letters, year, month letter, day, place code, check
/// letter; digits may be replaced by letters (omocodia).
static IT_CF: LazyLock<Regex> = LazyLock::new(|| {
let d = "[0-9LMNPQRSTUV]";
re(&format!(
r"(?i)\b[A-Z]{{6}}{d}{{2}}[ABCDEHLMPRST]{d}{{2}}[A-Z]{d}{{3}}[A-Z]\b"
))
});
/// The Ministry's odd-position values for 0–9 and A–Z.
const CF_ODD: [u32; 36] = [
1, 0, 5, 7, 9, 13, 15, 17, 19, 21, // 0-9
1, 0, 5, 7, 9, 13, 15, 17, 19, 21, 2, 4, 18, 20, 11, 3, 6, 8, 12, 14, 16, 10, 22, 25, 24,
23, // A-Z
];
pub fn codice_fiscale_valid(cf: &str) -> bool {
let index = |c: u8| {
if c.is_ascii_digit() {
(c - b'0') as usize
} else {
(c - b'A') as usize + 10
}
};
let even = |c: u8| {
if c.is_ascii_digit() {
u32::from(c - b'0')
} else {
u32::from(c - b'A')
}
};
let bytes = cf.as_bytes();
let sum: u32 = bytes[..15]
.iter()
.enumerate()
.map(|(i, &c)| {
if i % 2 == 0 {
CF_ODD[index(c)]
} else {
even(c)
}
})
.sum();
u32::from(bytes[15] - b'A') == sum % 26
}
fn it_codice_fiscale(text: &str, findings: &mut Findings) {
for m in IT_CF.find_iter(text) {
let cf = m.as_str().to_ascii_uppercase();
if codice_fiscale_valid(&cf) {
findings.insert(cf);
}
}
}
// --- Netherlands ----------------------------------------------------------
/// Nine digits, sometimes written `1112.22.333`.
static NL_BSN: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{4})(\.?)(\d{2})(\.?)(\d{3})\b"));
/// The eleven test: weights 9 down to 2, and −1 for the last digit.
pub fn bsn_valid(n: &str) -> bool {
let d = digit_values(n);
let sum: i64 = d[..8]
.iter()
.zip((2..=9).rev())
.map(|(a, w)| i64::from(a * w))
.sum::<i64>()
- i64::from(d[8]);
sum != 0 && sum % 11 == 0
}
const BSN_WORDS: &[&str] = &[
"bsn",
"burgerservicenummer",
"sofinummer",
"sofi-nummer",
"citizen service number",
];
fn nl_bsn(text: &str, findings: &mut Findings) {
for c in NL_BSN.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = &c[2] == "." && &c[4] == ".";
if bsn_valid(&n) && (written || word_near(text, whole.start(), whole.end(), BSN_WORDS)) {
findings.insert(n);
}
}
}
// --- Belgium --------------------------------------------------------------
/// `YY.MM.DD-XXX.CC` or eleven digits.
static BE_NN: LazyLock<Regex> =
LazyLock::new(|| re(r"\b(\d{2})\.?(\d{2})\.?(\d{2})-?(\d{3})\.?(\d{2})\b"));
fn be_national_number(text: &str, findings: &mut Findings) {
for c in BE_NN.captures_iter(text) {
let (month, day) = (num(&c[2]), num(&c[3]));
// Month 0 and day 0 mean unknown; bis numbers add 20 or 40 to the month
if !(month <= 12 || (20..=32).contains(&month) || (40..=52).contains(&month)) || day > 31 {
continue;
}
let body = format!("{}{}{}{}", &c[1], &c[2], &c[3], &c[4]);
let check = u64::from(num(&c[5]));
let before_2000 = 97 - body.parse::<u64>().unwrap_or(0) % 97;
let since_2000 = 97 - format!("2{body}").parse::<u64>().unwrap_or(0) % 97;
if check == before_2000 || check == since_2000 {
findings.insert(format!("{body}{}", &c[5]));
}
}
}
// --- Poland ---------------------------------------------------------------
static ELEVEN: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{11}\b"));
/// Weights 1, 3, 7, 9 repeating; the birth date encodes the century in the
/// month (+80 for the 1800s, +20 for the 2000s, and so on).
pub fn pesel_valid(n: &str) -> bool {
let d = digit_values(n);
let sum: u32 = d[..10]
.iter()
.zip([1, 3, 7, 9].iter().cycle())
.map(|(a, w)| a * w)
.sum();
let month = d[2] * 10 + d[3];
let (century, month) = match month {
81..=92 => (1800, month - 80),
1..=12 => (1900, month),
21..=32 => (2000, month - 20),
41..=52 => (2100, month - 40),
_ => return false,
};
let year = century + d[0] * 10 + d[1];
(10 - sum % 10) % 10 == d[10] && (1..=super::days_in(year, month)).contains(&(d[4] * 10 + d[5]))
}
const PESEL_WORDS: &[&str] = &["pesel", "numer pesel", "nr pesel"];
fn pl_pesel(text: &str, findings: &mut Findings) {
for m in ELEVEN.find_iter(text) {
if pesel_valid(m.as_str()) && word_near(text, m.start(), m.end(), PESEL_WORDS) {
findings.insert(m.as_str());
}
}
}
// --- Sweden ---------------------------------------------------------------
/// `YYMMDD-NNNN`, `YYYYMMDD-NNNN` (`+` after 100), or the bare digits.
static SE_PNR: LazyLock<Regex> =
LazyLock::new(|| re(r"\b(?:\d{2})?(\d{2})(\d{2})(\d{2})([-+]?)(\d{4})\b"));
const SE_WORDS: &[&str] = &[
"personnummer",
"personnr",
"person nr",
"samordningsnummer",
"pnr",
];
fn se_personnummer(text: &str, findings: &mut Findings) {
for c in SE_PNR.captures_iter(text) {
let whole = c.get(0).unwrap();
let (yy, month, day) = (num(&c[1]), num(&c[2]), num(&c[3]));
// Coordination numbers add 60 to the day
let day = if day > 60 { day - 60 } else { day };
let ten = format!("{}{}{}{}", &c[1], &c[2], &c[3], &c[5]);
let written = !c[4].is_empty();
if valid_short_date(yy, month, day)
&& checks::luhn(&ten)
&& (written || word_near(text, whole.start(), whole.end(), SE_WORDS))
{
findings.insert(ten);
}
}
}
// --- Denmark --------------------------------------------------------------
static DK_CPR: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{2})(\d{2})(\d{2})-?(\d{4})\b"));
const CPR_WORDS: &[&str] = &["cpr", "cpr-nr", "cpr nr", "cpr-nummer", "personnummer"];
fn dk_cpr(text: &str, findings: &mut Findings) {
for c in DK_CPR.captures_iter(text) {
let whole = c.get(0).unwrap();
if valid_short_date(num(&c[3]), num(&c[2]), num(&c[1]))
&& word_near(text, whole.start(), whole.end(), CPR_WORDS)
{
findings.insert(whole.as_str().replace('-', ""));
}
}
}
// --- Finland --------------------------------------------------------------
static FI_HETU: LazyLock<Regex> =
LazyLock::new(|| re(r"(?i)\b(\d{2})(\d{2})(\d{2})[-+ABCDEFYXWVU](\d{3})([0-9A-Y])\b"));
const HETU_CHECK: &[u8] = b"0123456789ABCDEFHJKLMNPRSTUVWXY";
fn fi_hetu(text: &str, findings: &mut Findings) {
for c in FI_HETU.captures_iter(text) {
let (day, month, yy) = (num(&c[1]), num(&c[2]), num(&c[3]));
let n: u64 = format!("{}{}{}{}", &c[1], &c[2], &c[3], &c[4])
.parse()
.unwrap_or(0);
let check = c[5].to_ascii_uppercase().as_bytes()[0];
if valid_short_date(yy, month, day) && HETU_CHECK[(n % 31) as usize] == check {
findings.insert(c[0].to_ascii_uppercase());
}
}
}
// --- Ireland --------------------------------------------------------------
static IE_PPS: LazyLock<Regex> = LazyLock::new(|| re(r"(?i)\b(\d{7})([A-W])([ABHW]?)\b"));
const PPS_CHECK: &[u8] = b"WABCDEFGHIJKLMNOPQRSTUV";
fn ie_pps(text: &str, findings: &mut Findings) {
for c in IE_PPS.captures_iter(text) {
let mut sum: u32 = digit_values(&c[1])
.iter()
.zip((2..=8).rev())
.map(|(a, w)| a * w)
.sum();
// The second letter counts, times 9; W (the old form) counts as 0
let second = c[3].to_ascii_uppercase();
if let Some(&letter) = second.as_bytes().first()
&& letter != b'W'
{
sum += u32::from(letter - b'A' + 1) * 9;
}
let check = c[2].to_ascii_uppercase().as_bytes()[0];
if PPS_CHECK[(sum % 23) as usize] == check {
findings.insert(c[0].to_ascii_uppercase());
}
}
}
// --- Portugal -------------------------------------------------------------
static NINE: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{9}\b"));
/// Mod 11 over weights 9 down to 2; a check of 10 or 11 becomes 0.
pub fn nif_valid(n: &str) -> bool {
let d = digit_values(n);
let sum: u32 = d[..8].iter().zip((2..=9).rev()).map(|(a, w)| a * w).sum();
let check = match 11 - sum % 11 {
10 | 11 => 0,
c => c,
};
matches!(d[0], 1 | 2 | 3 | 5 | 6 | 8 | 9) && check == d[8]
}
const NIF_WORDS: &[&str] = &[
"nif",
"contribuinte",
"número de identificação fiscal",
"numero de contribuinte",
];
fn pt_nif(text: &str, findings: &mut Findings) {
for m in NINE.find_iter(text) {
if nif_valid(m.as_str()) && word_near(text, m.start(), m.end(), NIF_WORDS) {
findings.insert(m.as_str());
}
}
}
// --- Austria --------------------------------------------------------------
/// A serial and check digit, then the birth date: `1237 010180`.
static AT_SVNR: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{3})(\d)( ?)(\d{2})(\d{2})(\d{2})\b"));
const SVNR_WORDS: &[&str] = &[
"sozialversicherungsnummer",
"svnr",
"sv-nr",
"sv-nummer",
"versicherungsnummer",
];
fn at_svnr(text: &str, findings: &mut Findings) {
for c in AT_SVNR.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}{}{}", &c[1], &c[2], &c[4], &c[5], &c[6]);
let d = digit_values(&n);
let sum: u32 = d
.iter()
.zip([3, 7, 9, 0, 5, 8, 4, 2, 1, 6])
.map(|(a, w)| a * w)
.sum();
let written = &c[3] == " ";
if d[0] != 0
&& sum % 11 == d[3]
&& valid_short_date(num(&c[6]), num(&c[5]), num(&c[4]))
&& (written || word_near(text, whole.start(), whole.end(), SVNR_WORDS))
&& stands_alone(text, whole.start(), whole.end())
{
findings.insert(n);
}
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn germany() {
assert_eq!(count("de-tax-id", "86 095 742 719"), 1);
assert_eq!(count("de-tax-id", "Steuer-ID: 86095742719"), 1);
assert_eq!(count("de-tax-id", "Rechnung 86095742719"), 0);
assert_eq!(count("de-tax-id", "86 095 742 718"), 0);
// ICAO 9303's German specimen card
assert_eq!(count("de-id-card", "Ausweis T220001293"), 1);
assert_eq!(count("de-id-card", "T220001294"), 0);
}
#[test]
fn france_spain_italy() {
assert_eq!(count("fr-nir", "2 55 08 14 168 025 38"), 1);
assert_eq!(count("fr-nir", "255081416802539"), 0);
assert_eq!(count("es-dni-nie", "DNI 12345678Z, NIE X-1234567-L"), 2);
assert_eq!(count("es-dni-nie", "12345678A"), 0);
assert_eq!(count("it-codice-fiscale", "CF: RSSMRA85T10A562S"), 1);
assert_eq!(count("it-codice-fiscale", "RSSMRA85T10A562T"), 0);
}
#[test]
fn benelux() {
assert_eq!(count("nl-bsn", "1112.22.333"), 1);
assert_eq!(count("nl-bsn", "BSN 111222333"), 1);
assert_eq!(count("nl-bsn", "order 111222333"), 0);
assert_eq!(count("nl-bsn", "BSN 111222334"), 0);
assert_eq!(count("be-national-number", "85.07.30-033.28"), 1);
assert_eq!(count("be-national-number", "85073003329"), 0);
}
#[test]
fn nordics() {
assert_eq!(count("se-personnummer", "811218-9876"), 1);
assert_eq!(count("se-personnummer", "811218-9875"), 0);
assert_eq!(count("se-personnummer", "order 8112189876"), 0);
assert_eq!(count("se-personnummer", "personnummer 198112189876"), 1);
assert_eq!(count("dk-cpr", "CPR-nr: 010170-1234"), 1);
assert_eq!(count("dk-cpr", "010170-1234"), 0);
assert_eq!(count("dk-cpr", "CPR 320170-1234"), 0);
assert_eq!(count("fi-hetu", "131052-308T"), 1);
assert_eq!(count("fi-hetu", "131052-308U"), 0);
}
#[test]
fn poland_ireland_portugal_austria() {
assert_eq!(count("pl-pesel", "PESEL 44051401359, pesel 02070803628"), 2);
assert_eq!(count("pl-pesel", "PESEL 44051401358"), 0);
assert_eq!(count("pl-pesel", "44051401359"), 0);
assert_eq!(count("ie-pps", "PPS 1234567T and 1234567FA"), 2);
assert_eq!(count("ie-pps", "1234567U"), 0);
assert_eq!(count("pt-nif", "NIF 123456789"), 1);
assert_eq!(count("pt-nif", "NIF 123456788"), 0);
assert_eq!(count("at-svnr", "1237 010180"), 1);
assert_eq!(count("at-svnr", "SVNR 1237010180"), 1);
assert_eq!(count("at-svnr", "1238 010180"), 0);
}
}
@@ -0,0 +1,116 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! European identifiers outside the EU (§2.3): Norway's national identity
//! number and Switzerland's AHV number.
use super::{Detector, Findings, Region, Strength, digit_values, valid_short_date};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"no-fnr",
"Norway: national identity number",
Region::Europe,
Strength::Checked,
no_fnr,
),
Detector::new(
"ch-ahv",
"Switzerland: AHV number",
Region::Europe,
Strength::Checked,
ch_ahv,
),
];
static ELEVEN: LazyLock<Regex> =
LazyLock::new(|| Regex::new(r"\b\d{6} ?\d{5}\b").expect("detector pattern"));
/// Two mod 11 check digits over a birth date (D-numbers add 40 to the day,
/// H-numbers 40 to the month): strong enough to count alone.
pub fn fnr_valid(n: &str) -> bool {
let d = digit_values(n);
if d.len() != 11 {
return false;
}
let check =
|weights: &[u32]| match 11 - d.iter().zip(weights).map(|(a, w)| a * w).sum::<u32>() % 11 {
11 => Some(0),
10 => None,
c => Some(c),
};
let day = d[0] * 10 + d[1];
let month = d[2] * 10 + d[3];
let day = if day > 40 { day - 40 } else { day };
let month = if month > 40 { month - 40 } else { month };
valid_short_date(d[4] * 10 + d[5], month, day)
&& check(&[3, 7, 6, 1, 8, 9, 4, 5, 2]) == Some(d[9])
&& check(&[5, 4, 3, 2, 7, 6, 5, 4, 3, 2]) == Some(d[10])
}
fn no_fnr(text: &str, findings: &mut Findings) {
for m in ELEVEN.find_iter(text) {
let n = m.as_str().replace(' ', "");
if fnr_valid(&n) {
findings.insert(n);
}
}
}
/// `756.1234.5678.97`: the country prefix, then an EAN-13 check digit.
static AHV: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(r"\b756[. ]?\d{4}[. ]?\d{4}[. ]?\d{2}\b").expect("detector pattern")
});
pub fn ean13_valid(n: &str) -> bool {
let d = digit_values(n);
if d.len() != 13 {
return false;
}
let sum: u32 = d[..12]
.iter()
.enumerate()
.map(|(i, x)| if i % 2 == 0 { *x } else { x * 3 })
.sum();
(10 - sum % 10) % 10 == d[12]
}
fn ch_ahv(text: &str, findings: &mut Findings) {
for m in AHV.find_iter(text) {
let n: String = m.as_str().chars().filter(char::is_ascii_digit).collect();
if ean13_valid(&n) {
findings.insert(n);
}
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn norway() {
assert_eq!(count("no-fnr", "01019000083"), 1);
assert_eq!(count("no-fnr", "010190 00083"), 1);
assert_eq!(count("no-fnr", "01019000084"), 0);
// Not a date
assert_eq!(count("no-fnr", "32019000083"), 0);
}
#[test]
fn switzerland() {
// The federal example
assert_eq!(count("ch-ahv", "AHV 756.9217.0769.85"), 1);
assert_eq!(count("ch-ahv", "7569217076985"), 1);
assert_eq!(count("ch-ahv", "756.9217.0769.86"), 0);
}
}
+80 -1
View File
@@ -19,8 +19,18 @@
//! same card number pasted twice counts once. They stay in memory: callers
//! read only [`Findings::len`].
pub mod africa;
pub mod americas;
pub mod any;
pub mod asia;
pub mod australia;
pub mod canada;
pub mod checks;
pub mod eu;
pub mod europe;
pub mod templates;
pub mod uk;
pub mod us;
use ahash::AHashSet;
@@ -108,7 +118,20 @@ impl Detector {
/// Every detector, in the order the console lists them.
pub fn all() -> impl Iterator<Item = &'static Detector> {
any::DETECTORS.iter()
[
any::DETECTORS,
us::DETECTORS,
uk::DETECTORS,
canada::DETECTORS,
australia::DETECTORS,
eu::DETECTORS,
europe::DETECTORS,
asia::DETECTORS,
americas::DETECTORS,
africa::DETECTORS,
]
.into_iter()
.flatten()
}
pub fn by_id(id: &str) -> Option<&'static Detector> {
@@ -159,6 +182,38 @@ pub fn stands_alone(text: &str, start: usize, end: usize) -> bool {
before.is_none_or(|c| !c.is_alphanumeric()) && after.is_none_or(|c| !c.is_alphanumeric())
}
/// Days in `month` of `year` (0 for a month that doesn't exist).
pub fn days_in(year: u32, month: u32) -> u32 {
match month {
1 | 3 | 5 | 7 | 8 | 10 | 12 => 31,
4 | 6 | 9 | 11 => 30,
2 if year.is_multiple_of(4) && (!year.is_multiple_of(100) || year.is_multiple_of(400)) => {
29
}
2 => 28,
_ => 0,
}
}
/// Whether `year`-`month`-`day` is a real date between 1900 and 2100.
pub fn valid_date(year: u32, month: u32, day: u32) -> bool {
(1900..=2100).contains(&year) && (1..=days_in(year, month)).contains(&day)
}
/// Whether a two-digit year, month and day make a real date in either the
/// 1900s or the 2000s.
pub fn valid_short_date(yy: u32, month: u32, day: u32) -> bool {
valid_date(1900 + yy, month, day) || valid_date(2000 + yy, month, day)
}
/// The value of each digit in `s`.
pub fn digit_values(s: &str) -> Vec<u32> {
s.bytes()
.filter(u8::is_ascii_digit)
.map(|b| u32::from(b - b'0'))
.collect()
}
/// The ASCII digits of `s`.
pub fn digits(s: &str) -> String {
s.chars().filter(char::is_ascii_digit).collect()
@@ -188,6 +243,30 @@ mod tests {
assert!(word_near(&text, start, start + 8, &["passport"]));
}
/// An ordinary business email: order, invoice and tracking numbers,
/// dates, amounts, a street address. Nothing here is an identifier, so
/// no detector may fire, except the contact ones on the signature.
#[test]
fn ordinary_mail_finds_nothing() {
let text = "Hi Dana,\n\nThanks for order 4471-2290 placed 2026-09-14. Invoice INV-2026-00917 \
for $12,480.00 is due 10/31/2026; PO 7731902 covers lines 1-14. Tracking \
1Z999AA10123456784, parcel 3 of 5, 12.5 kg, box 40x30x20 cm. Meeting moved to \
Tuesday 9:30-10:15 in room 2B, building 1177. Ticket #5520318, case 20260914-0042. \
Version 2026.9.28.4, build 118822, commit 5a73a118. Serial SN-88213-X. \
Ship to 1600 Amphitheatre Pkwy, Mountain View, CA 94043. Revenue grew 18% to \
1,204,332 units; see figures 3.1-3.4 and table 12.\n\nBest,\nSam\n\
Sam Rivera | +1 (415) 555-2671 | [email protected]";
let quiet = ["email-addresses", "phone-numbers"];
for detector in all().filter(|d| !quiet.contains(&d.id)) {
assert_eq!(
detector.count(text),
0,
"{} fired on ordinary mail",
detector.id
);
}
}
#[test]
fn ids_are_unique() {
let mut seen = AHashSet::new();
@@ -0,0 +1,97 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! Templates (§2.3): named sets of detectors, so a policy doesn't pick forty
//! one at a time. Each is named for what it finds, never for a law, and is a
//! starting point: once added to a rule, its detectors can be changed.
pub struct Template {
pub id: &'static str,
pub name: &'static str,
pub detectors: &'static [&'static str],
}
pub static TEMPLATES: &[Template] = &[
Template {
id: "payment-and-bank",
name: "Payment cards and bank accounts",
detectors: &["payment-card", "iban", "swift-bic", "us-aba-routing"],
},
Template {
id: "us-personal",
name: "US personal identifiers",
detectors: &[
"us-ssn",
"us-itin",
"us-ein",
"us-drivers-license",
"passport",
"date-of-birth",
],
},
Template {
id: "uk-personal",
name: "UK personal identifiers",
detectors: &["uk-nino", "uk-utr", "uk-nhs", "passport", "date-of-birth"],
},
Template {
id: "eu-national",
name: "EU national identifiers",
detectors: &[
"de-tax-id",
"de-id-card",
"fr-nir",
"es-dni-nie",
"it-codice-fiscale",
"nl-bsn",
"be-national-number",
"pl-pesel",
"se-personnummer",
"dk-cpr",
"fi-hetu",
"ie-pps",
"pt-nif",
"at-svnr",
],
},
Template {
id: "health",
name: "Health identifiers",
detectors: &["uk-nhs", "us-mbi", "us-npi", "us-dea", "au-medicare"],
},
Template {
id: "credentials",
name: "Credentials and keys",
detectors: &["private-key", "credentials"],
},
Template {
id: "contact-lists",
name: "Contact lists",
detectors: &["email-addresses", "phone-numbers"],
},
];
pub fn by_id(id: &str) -> Option<&'static Template> {
TEMPLATES.iter().find(|template| template.id == id)
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn every_template_names_real_detectors() {
for template in TEMPLATES {
for id in template.detectors {
assert!(
super::super::by_id(id).is_some(),
"{}: no detector {id}",
template.id
);
}
}
}
}
@@ -0,0 +1,155 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! United Kingdom identifiers (§2.3): HMRC's National Insurance number and
//! Unique Taxpayer Reference, and the NHS number.
use super::{Detector, Findings, Region, Strength, word_near};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"uk-nino",
"UK National Insurance number",
Region::Uk,
Strength::Checked,
nino,
),
Detector::new(
"uk-nhs",
"UK NHS number",
Region::Uk,
Strength::Checked,
nhs,
),
Detector::new(
"uk-utr",
"UK Unique Taxpayer Reference",
Region::Uk,
Strength::NeedsWord,
utr,
),
];
fn re(pattern: &str) -> Regex {
Regex::new(pattern).expect("detector pattern")
}
/// Two letters, six digits (often in pairs), a suffix A–D.
static NINO: LazyLock<Regex> =
LazyLock::new(|| re(r"(?i)\b([A-Z])([A-Z]) ?(\d{2}) ?(\d{2}) ?(\d{2}) ?([A-D])\b"));
/// HMRC's rules: D, F, I, Q, U and V are never used; O never second; and
/// BG, GB, KN, NK, NT, TN and ZZ are never allocated.
fn nino_prefix(first: char, second: char) -> bool {
const NEVER: &str = "DFIQUV";
let pair: String = [first, second].iter().collect();
!NEVER.contains(first)
&& !NEVER.contains(second)
&& second != 'O'
&& !["BG", "GB", "KN", "NK", "NT", "TN", "ZZ"].contains(&pair.as_str())
}
fn nino(text: &str, findings: &mut Findings) {
for c in NINO.captures_iter(text) {
let first = c[1].to_ascii_uppercase().chars().next().unwrap();
let second = c[2].to_ascii_uppercase().chars().next().unwrap();
if nino_prefix(first, second) {
findings.insert(format!(
"{first}{second}{}{}{}{}",
&c[3],
&c[4],
&c[5],
c[6].to_ascii_uppercase()
));
}
}
}
/// `NNN NNN NNNN` stands alone; ten bare digits need a word.
static NHS: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{3})([ -]?)(\d{3})([ -]?)(\d{4})\b"));
/// Mod 11: weights 10 down to 2 over the first nine digits; the check digit
/// is 11 minus the remainder (11 becomes 0; 10 is never issued).
pub fn nhs_valid(n: &str) -> bool {
let d: Vec<u32> = n.bytes().map(|b| u32::from(b - b'0')).collect();
let sum: u32 = d[..9].iter().zip((2..=10).rev()).map(|(a, w)| a * w).sum();
match 11 - sum % 11 {
11 => d[9] == 0,
10 => false,
check => d[9] == check,
}
}
const NHS_WORDS: &[&str] = &["nhs", "nhs number", "nhs no"];
fn nhs(text: &str, findings: &mut Findings) {
for c in NHS.captures_iter(text) {
let whole = c.get(0).unwrap();
let n = format!("{}{}{}", &c[1], &c[3], &c[5]);
let written = !c[2].is_empty() && c[2] == c[4];
if nhs_valid(&n) && (written || word_near(text, whole.start(), whole.end(), NHS_WORDS)) {
findings.insert(n);
}
}
}
static UTR: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{5} ?\d{5}\b"));
const UTR_WORDS: &[&str] = &[
"utr",
"unique taxpayer reference",
"tax reference",
"self assessment",
];
fn utr(text: &str, findings: &mut Findings) {
for m in UTR.find_iter(text) {
if word_near(text, m.start(), m.end(), UTR_WORDS) {
findings.insert(m.as_str().replace(' ', ""));
}
}
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn national_insurance() {
assert_eq!(count("uk-nino", "NI: AB 12 34 56 C, ce123456d"), 2);
// Letters never used, pairs never allocated, a suffix past D
for bad in [
"QQ123456C",
"AO123456C",
"GB123456A",
"AB123456E",
"DA123456A",
] {
assert_eq!(count("uk-nino", bad), 0, "{bad}");
}
}
#[test]
fn nhs_numbers() {
// The NHS's own example
assert_eq!(count("uk-nhs", "943 476 5919"), 1);
assert_eq!(count("uk-nhs", "943 476 5918"), 0);
assert_eq!(count("uk-nhs", "order 9434765919"), 0);
assert_eq!(count("uk-nhs", "NHS number 9434765919"), 1);
}
#[test]
fn utr() {
assert_eq!(count("uk-utr", "UTR 12345 67890"), 1);
assert_eq!(count("uk-utr", "order 1234567890"), 0);
}
}
@@ -0,0 +1,304 @@
/*
* SPDX-FileCopyrightText: 2026 Coffey Labs
*
* SPDX-License-Identifier: AGPL-3.0-only
*/
//! United States identifiers (§2.3), each from its issuer's published rules:
//! the SSA (SSN), the IRS (ITIN, EIN), the ABA (routing numbers), CMS (MBI,
//! NPI) and the DEA.
use super::{Detector, Findings, Region, Strength, checks, digits, stands_alone, word_near};
use regex::Regex;
use std::sync::LazyLock;
pub static DETECTORS: &[Detector] = &[
Detector::new(
"us-ssn",
"US Social Security number",
Region::Us,
Strength::Checked,
ssn,
),
Detector::new("us-itin", "US ITIN", Region::Us, Strength::Checked, itin),
Detector::new("us-ein", "US EIN", Region::Us, Strength::NeedsWord, ein),
Detector::new(
"us-aba-routing",
"US bank routing number",
Region::Us,
Strength::NeedsWord,
aba_routing,
),
Detector::new(
"us-drivers-license",
"US driver's license",
Region::Us,
Strength::NeedsWord,
drivers_license,
),
Detector::new(
"us-mbi",
"US Medicare Beneficiary Identifier",
Region::Us,
Strength::Checked,
mbi,
),
Detector::new(
"us-npi",
"US National Provider Identifier",
Region::Us,
Strength::NeedsWord,
npi,
),
Detector::new(
"us-dea",
"US DEA registration number",
Region::Us,
Strength::Checked,
dea,
),
];
fn re(pattern: &str) -> Regex {
Regex::new(pattern).expect("detector pattern")
}
/// `AAA-GG-SSSS` (dashes or spaces), or nine bare digits.
static NINE: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{3})([ -]?)(\d{2})([ -]?)(\d{4})\b"));
/// Numbers the SSA has published as never valid: widely printed examples.
const SSN_EXAMPLES: &[&str] = &["078051120", "219099999"];
fn ssn_rules(area: u32, group: u32, serial: u32) -> bool {
area != 0 && area != 666 && area < 900 && group != 0 && serial != 0
}
const SSN_WORDS: &[&str] = &["ssn", "social security", "soc sec", "ss#", "ss no"];
fn ssn(text: &str, findings: &mut Findings) {
for c in NINE.captures_iter(text) {
let whole = c.get(0).unwrap();
let (area, group, serial) = (num(&c[1]), num(&c[3]), num(&c[5]));
let number = format!("{}{}{}", &c[1], &c[3], &c[5]);
// Written form (with both separators, the same one) stands alone;
// nine bare digits need a word
let written = !c[2].is_empty() && c[2] == c[4];
if ssn_rules(area, group, serial)
&& !SSN_EXAMPLES.contains(&number.as_str())
&& (written || word_near(text, whole.start(), whole.end(), SSN_WORDS))
{
findings.insert(number);
}
}
}
/// ITINs: 9XX, then a group in the IRS's ranges.
fn itin_group(group: u32) -> bool {
matches!(group, 50..=65 | 70..=88 | 90..=92 | 94..=99)
}
const ITIN_WORDS: &[&str] = &["itin", "taxpayer identification", "tax id"];
fn itin(text: &str, findings: &mut Findings) {
for c in NINE.captures_iter(text) {
let whole = c.get(0).unwrap();
let written = !c[2].is_empty() && c[2] == c[4];
if c[1].starts_with('9')
&& itin_group(num(&c[3]))
&& (written || word_near(text, whole.start(), whole.end(), ITIN_WORDS))
{
findings.insert(format!("{}{}{}", &c[1], &c[3], &c[5]));
}
}
}
static EIN: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{2})-?(\d{7})\b"));
/// The prefixes the IRS assigns to its campuses and internet EINs.
fn ein_prefix(prefix: u32) -> bool {
matches!(prefix, 1..=6 | 10..=16 | 20..=27 | 30..=48 | 50..=68 | 71..=77 | 80..=88 | 90..=95 | 98 | 99)
}
const EIN_WORDS: &[&str] = &[
"ein",
"fein",
"employer identification",
"tax id",
"tin",
"federal tax",
];
fn ein(text: &str, findings: &mut Findings) {
for c in EIN.captures_iter(text) {
let whole = c.get(0).unwrap();
if ein_prefix(num(&c[1])) && word_near(text, whole.start(), whole.end(), EIN_WORDS) {
findings.insert(format!("{}{}", &c[1], &c[2]));
}
}
}
static ROUTING: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d{9}\b"));
/// The ABA check: 3, 7 and 1 weights, mod 10; and a Federal Reserve prefix.
pub fn aba_valid(n: &str) -> bool {
let d: Vec<u32> = n.bytes().map(|b| u32::from(b - b'0')).collect();
let prefix = d[0] * 10 + d[1];
matches!(prefix, 0..=12 | 21..=32 | 61..=72 | 80)
&& (3 * (d[0] + d[3] + d[6]) + 7 * (d[1] + d[4] + d[7]) + (d[2] + d[5] + d[8]))
.is_multiple_of(10)
}
const ROUTING_WORDS: &[&str] = &["routing", "aba", "rtn", "routing number", "transit"];
fn aba_routing(text: &str, findings: &mut Findings) {
for m in ROUTING.find_iter(text) {
// One random number in ten passes the check: always needs a word
if aba_valid(m.as_str()) && word_near(text, m.start(), m.end(), ROUTING_WORDS) {
findings.insert(m.as_str());
}
}
}
/// The shapes states issue: up to two letters, then 5–14 digits, dashes
/// allowed (Florida and Illinois print them).
static LICENSE: LazyLock<Regex> = LazyLock::new(|| re(r"\b[A-Z]{0,2}\d[\d-]{3,16}\d\b"));
const LICENSE_WORDS: &[&str] = &[
"driver's license",
"drivers license",
"driver license",
"driver's licence",
"dl",
"dl#",
"license number",
"lic no",
"dmv",
];
fn drivers_license(text: &str, findings: &mut Findings) {
for m in LICENSE.find_iter(text) {
let n = digits(m.as_str());
if (5..=14).contains(&n.len()) && word_near(text, m.start(), m.end(), LICENSE_WORDS) {
findings.insert(m.as_str().replace('-', ""));
}
}
}
/// CMS's MBI: 11 characters in a fixed pattern of digits, letters and
/// either, the letters S, L, O, I, B and Z never used; dashes may follow the
/// 4th and 7th.
static MBI: LazyLock<Regex> = LazyLock::new(|| {
let c = "[AC-HJKMNP-RT-Y]";
let an = "[AC-HJKMNP-RT-Y0-9]";
re(&format!(
r"\b[1-9]{c}{an}[0-9]-?{c}{an}[0-9]-?{c}{c}[0-9][0-9]\b"
))
});
fn mbi(text: &str, findings: &mut Findings) {
for m in MBI.find_iter(text) {
findings.insert(m.as_str().replace('-', ""));
}
}
static TEN: LazyLock<Regex> = LazyLock::new(|| re(r"\b[12]\d{9}\b"));
const NPI_WORDS: &[&str] = &["npi", "national provider", "provider id", "provider number"];
/// NPI: Luhn over the ISO card-issuer prefix 80840 and the number.
fn npi(text: &str, findings: &mut Findings) {
for m in TEN.find_iter(text) {
if checks::luhn(&format!("80840{}", m.as_str()))
&& word_near(text, m.start(), m.end(), NPI_WORDS)
{
findings.insert(m.as_str());
}
}
}
static DEA: LazyLock<Regex> = LazyLock::new(|| re(r"\b([ABCDEFGHJKLMPRSTUX][A-Z9])(\d{7})\b"));
/// DEA: (1st + 3rd + 5th) + 2 × (2nd + 4th + 6th) ends in the 7th digit.
fn dea(text: &str, findings: &mut Findings) {
for c in DEA.captures_iter(text) {
let d: Vec<u32> = c[2].bytes().map(|b| u32::from(b - b'0')).collect();
if ((d[0] + d[2] + d[4]) + 2 * (d[1] + d[3] + d[5])) % 10 == d[6] {
let whole = c.get(0).unwrap();
if stands_alone(text, whole.start(), whole.end()) {
findings.insert(whole.as_str());
}
}
}
}
fn num(s: &str) -> u32 {
s.parse().unwrap_or(0)
}
#[cfg(test)]
mod tests {
use crate::mailflow::detectors::by_id;
fn count(id: &str, text: &str) -> usize {
by_id(id).unwrap().count(text)
}
#[test]
fn ssn() {
assert_eq!(count("us-ssn", "SSN 536-22-1234, also 536 22 1235"), 2);
// Bare digits: only with a word
assert_eq!(count("us-ssn", "ref 536221234"), 0);
assert_eq!(count("us-ssn", "social security: 536221234"), 1);
// Never issued, the SSA's printed examples, mixed separators
for bad in [
"000-12-3456",
"666-12-3456",
"912-12-3456",
"123-00-4567",
"123-45-0000",
"078-05-1120",
"536-22 1234",
] {
assert_eq!(count("us-ssn", bad), 0, "{bad}");
}
}
#[test]
fn itin_and_ein() {
assert_eq!(count("us-itin", "912-70-1234"), 1);
assert_eq!(count("us-itin", "912-69-1234"), 0);
assert_eq!(count("us-ssn", "912-70-1234"), 0);
assert_eq!(count("us-ein", "EIN: 12-3456789"), 1);
assert_eq!(count("us-ein", "part 12-3456789"), 0);
assert_eq!(count("us-ein", "EIN 07-3456789"), 0);
}
#[test]
fn routing_needs_a_word() {
assert_eq!(count("us-aba-routing", "Routing number 011000015"), 1);
assert_eq!(count("us-aba-routing", "ABA 021000021"), 1);
assert_eq!(count("us-aba-routing", "invoice 011000015"), 0);
assert_eq!(count("us-aba-routing", "routing 011000016"), 0);
}
#[test]
fn licenses() {
assert_eq!(count("us-drivers-license", "Driver's license: D1234567"), 1);
assert_eq!(count("us-drivers-license", "DL# S123-456-78-901-0"), 1);
assert_eq!(count("us-drivers-license", "Order D1234567"), 0);
}
#[test]
fn health_identifiers() {
// CMS's own MBI example
assert_eq!(count("us-mbi", "Medicare 1EG4-TE5-MK73"), 1);
assert_eq!(count("us-mbi", "1EG4TE5MK73"), 1);
assert_eq!(count("us-mbi", "1EG4-TE5-MK7S"), 0);
// CMS's NPI example
assert_eq!(count("us-npi", "NPI 1234567893"), 1);
assert_eq!(count("us-npi", "NPI 1234567894"), 0);
assert_eq!(count("us-npi", "call 1234567893"), 0);
assert_eq!(count("us-dea", "DEA AB1234563"), 1);
assert_eq!(count("us-dea", "AB1234564"), 0);
}
}
+6 -4
View File
@@ -161,12 +161,14 @@ fn is_text(content_type: &str, extension: &str) -> bool {
fn decode_text(data: &[u8]) -> String {
let utf16 = |bytes: &[u8], big: bool| {
let units: Vec<u16> = bytes
.chunks_exact(2)
.map(|c| {
.as_chunks::<2>()
.0
.iter()
.map(|&c| {
if big {
u16::from_be_bytes([c[0], c[1]])
u16::from_be_bytes(c)
} else {
u16::from_le_bytes([c[0], c[1]])
u16::from_le_bytes(c)
}
})
.collect();