Merge pull request 'DLP: the detector framework, the region-free detectors, word lists and attachment text' (#99) from feature/dlp-detectors into main
This commit was merged in pull request #99.
This commit is contained in:
Generated
+4
@@ -3947,9 +3947,12 @@ name = "inbuxa-features"
|
||||
version = "0.16.22"
|
||||
dependencies = [
|
||||
"ahash",
|
||||
"aho-corasick",
|
||||
"base64 0.23.1",
|
||||
"flate2",
|
||||
"jmap_proto",
|
||||
"quick-xml 0.41.0",
|
||||
"regex",
|
||||
"registry",
|
||||
"serde",
|
||||
"serde_json",
|
||||
@@ -3961,6 +3964,7 @@ dependencies = [
|
||||
"types",
|
||||
"utils",
|
||||
"xxhash-rust",
|
||||
"zip",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
|
||||
@@ -21,6 +21,11 @@ base64 = "0.23"
|
||||
sha2 = "0.11"
|
||||
flate2 = "1.1"
|
||||
tokio = { version = "1.53", features = ["sync", "rt"] }
|
||||
# inbuxa: DLP detectors and attachment text (dlp-and-mail-flow-rules spec)
|
||||
regex = "1.13.1"
|
||||
aho-corasick = "1.1"
|
||||
zip = "8.6"
|
||||
quick-xml = "0.41"
|
||||
|
||||
[dev-dependencies]
|
||||
tokio = { version = "1.53", features = ["macros", "rt"] }
|
||||
|
||||
@@ -23,6 +23,7 @@ pub mod audit;
|
||||
pub mod branding;
|
||||
pub mod hold;
|
||||
pub mod lock;
|
||||
pub mod mailflow;
|
||||
pub mod masked_email;
|
||||
pub mod privacy;
|
||||
pub mod security;
|
||||
|
||||
@@ -0,0 +1,522 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Detectors that aren't tied to one country (§2.3, region "Any").
|
||||
|
||||
use super::{Detector, Findings, Region, Strength, checks, digits, stands_alone, word_near};
|
||||
use regex::Regex;
|
||||
use std::sync::LazyLock;
|
||||
|
||||
pub static DETECTORS: &[Detector] = &[
|
||||
Detector::new(
|
||||
"payment-card",
|
||||
"Payment card number",
|
||||
Region::Any,
|
||||
Strength::Checked,
|
||||
payment_card,
|
||||
),
|
||||
Detector::new("iban", "IBAN", Region::Any, Strength::Checked, iban),
|
||||
Detector::new(
|
||||
"swift-bic",
|
||||
"SWIFT/BIC code",
|
||||
Region::Any,
|
||||
Strength::NeedsWord,
|
||||
swift_bic,
|
||||
),
|
||||
Detector::new(
|
||||
"email-addresses",
|
||||
"Email addresses",
|
||||
Region::Any,
|
||||
Strength::Checked,
|
||||
email_addresses,
|
||||
),
|
||||
Detector::new(
|
||||
"phone-numbers",
|
||||
"Phone numbers",
|
||||
Region::Any,
|
||||
Strength::NeedsWord,
|
||||
phone_numbers,
|
||||
),
|
||||
Detector::new(
|
||||
"date-of-birth",
|
||||
"Date of birth",
|
||||
Region::Any,
|
||||
Strength::NeedsWord,
|
||||
date_of_birth,
|
||||
),
|
||||
Detector::new(
|
||||
"passport",
|
||||
"Passport number",
|
||||
Region::Any,
|
||||
Strength::NeedsWord,
|
||||
passport,
|
||||
),
|
||||
Detector::new(
|
||||
"private-key",
|
||||
"Private key",
|
||||
Region::Any,
|
||||
Strength::Checked,
|
||||
private_key,
|
||||
),
|
||||
Detector::new(
|
||||
"credentials",
|
||||
"Cloud and service credentials",
|
||||
Region::Any,
|
||||
Strength::Checked,
|
||||
credentials,
|
||||
),
|
||||
];
|
||||
|
||||
fn re(pattern: &str) -> Regex {
|
||||
Regex::new(pattern).expect("detector pattern")
|
||||
}
|
||||
|
||||
// --- Payment cards --------------------------------------------------------
|
||||
|
||||
/// Issuer prefixes (ISO/IEC 7812 IINs) and the lengths each network issues.
|
||||
fn card_network(number: &str) -> bool {
|
||||
let len = number.len();
|
||||
let prefix = |n: usize| number[..n].parse::<u32>().unwrap_or(0);
|
||||
match number.as_bytes()[0] {
|
||||
// Visa
|
||||
b'4' => matches!(len, 13 | 16 | 19),
|
||||
b'5' => {
|
||||
// Mastercard 51–55; Maestro 50, 56–58
|
||||
(51..=55).contains(&prefix(2)) && len == 16
|
||||
|| matches!(prefix(2), 50 | 56..=58) && (12..=19).contains(&len)
|
||||
}
|
||||
// Mastercard 2221–2720
|
||||
b'2' => (2221..=2720).contains(&prefix(4)) && len == 16,
|
||||
b'3' => {
|
||||
// American Express 34, 37; JCB 3528–3589; Diners 300–305, 36, 38, 39
|
||||
matches!(prefix(2), 34 | 37) && len == 15
|
||||
|| (3528..=3589).contains(&prefix(4)) && (16..=19).contains(&len)
|
||||
|| ((300..=305).contains(&prefix(3)) || matches!(prefix(2), 36 | 38 | 39))
|
||||
&& (14..=19).contains(&len)
|
||||
}
|
||||
// Discover 6011, 644–649, 65; UnionPay 62; Maestro 6x
|
||||
b'6' => (12..=19).contains(&len),
|
||||
_ => false,
|
||||
}
|
||||
}
|
||||
|
||||
fn is_card(number: &str) -> bool {
|
||||
(12..=19).contains(&number.len()) && card_network(number) && checks::luhn(number)
|
||||
}
|
||||
|
||||
static CARD: LazyLock<Regex> = LazyLock::new(|| re(r"\b\d(?:[ -]?\d){11,18}\b"));
|
||||
|
||||
fn payment_card(text: &str, findings: &mut Findings) {
|
||||
for m in CARD.find_iter(text) {
|
||||
if !stands_alone(text, m.start(), m.end()) {
|
||||
continue;
|
||||
}
|
||||
let whole = digits(m.as_str());
|
||||
if is_card(&whole) {
|
||||
findings.insert(whole);
|
||||
continue;
|
||||
}
|
||||
// Two numbers side by side ("4242 4242 4242 4242 2031"): try each
|
||||
// run of whole groups
|
||||
let groups: Vec<String> = m.as_str().split([' ', '-']).map(digits).collect();
|
||||
'runs: for from in 0..groups.len() {
|
||||
let mut number = String::new();
|
||||
for group in &groups[from..] {
|
||||
number.push_str(group);
|
||||
if is_card(&number) {
|
||||
findings.insert(number);
|
||||
break 'runs;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- IBAN -----------------------------------------------------------------
|
||||
|
||||
static IBAN: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\b[A-Za-z]{2}\d{2}(?:[ ]?[A-Za-z0-9]){11,30}"));
|
||||
|
||||
fn iban(text: &str, findings: &mut Findings) {
|
||||
// The pattern can run on into the next words, even the next IBAN: after
|
||||
// each hit, look again from where that IBAN ended
|
||||
let mut from = 0;
|
||||
while let Some(m) = IBAN.find_at(text, from) {
|
||||
from = m.start() + 1;
|
||||
let compact = m.as_str().replace(' ', "").to_ascii_uppercase();
|
||||
let Some(len) = checks::iban_length(&compact[..2]) else {
|
||||
continue;
|
||||
};
|
||||
if compact.len() < len {
|
||||
continue;
|
||||
}
|
||||
// Where the country's length ends in the text, spaces counted
|
||||
let mut seen = 0;
|
||||
let Some(end) = m
|
||||
.as_str()
|
||||
.char_indices()
|
||||
.find(|(_, c)| {
|
||||
if *c != ' ' {
|
||||
seen += 1;
|
||||
}
|
||||
seen == len
|
||||
})
|
||||
.map(|(i, c)| m.start() + i + c.len_utf8())
|
||||
else {
|
||||
continue;
|
||||
};
|
||||
let candidate = &compact[..len];
|
||||
if stands_alone(text, m.start(), end) && checks::iban(candidate) {
|
||||
findings.insert(candidate);
|
||||
from = end;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- SWIFT/BIC ------------------------------------------------------------
|
||||
|
||||
static BIC: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\b[A-Z]{4}[A-Z]{2}[A-Z0-9]{2}(?:[A-Z0-9]{3})?\b"));
|
||||
|
||||
const BIC_WORDS: &[&str] = &[
|
||||
"swift",
|
||||
"bic",
|
||||
"swift/bic",
|
||||
"bank",
|
||||
"banque",
|
||||
"bankverbindung",
|
||||
];
|
||||
|
||||
fn swift_bic(text: &str, findings: &mut Findings) {
|
||||
for m in BIC.find_iter(text) {
|
||||
let code = m.as_str();
|
||||
if checks::is_country(&code[4..6]) && word_near(text, m.start(), m.end(), BIC_WORDS) {
|
||||
findings.insert(code);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Contact lists --------------------------------------------------------
|
||||
|
||||
static EMAIL: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"(?i)\b[a-z0-9._%+-]+@[a-z0-9-]+(?:\.[a-z0-9-]+)*\.[a-z]{2,}\b"));
|
||||
|
||||
fn email_addresses(text: &str, findings: &mut Findings) {
|
||||
for m in EMAIL.find_iter(text) {
|
||||
findings.insert(m.as_str().to_lowercase());
|
||||
}
|
||||
}
|
||||
|
||||
/// International form: found alone. National form: only with a word.
|
||||
static PHONE_INTL: LazyLock<Regex> = LazyLock::new(|| re(r"\+\d{1,3}(?:[ .-]?\(?\d{1,4}\)?){2,5}"));
|
||||
static PHONE_NATIONAL: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\(?\d{2,4}\)?[ .-]\d{3,4}[ .-]\d{3,4}"));
|
||||
|
||||
const PHONE_WORDS: &[&str] = &[
|
||||
"phone",
|
||||
"tel",
|
||||
"telephone",
|
||||
"mobile",
|
||||
"cell",
|
||||
"fax",
|
||||
"telefon",
|
||||
"téléphone",
|
||||
"teléfono",
|
||||
"telefono",
|
||||
"handy",
|
||||
"portable",
|
||||
"móvil",
|
||||
"cellulare",
|
||||
"mobiel",
|
||||
];
|
||||
|
||||
fn phone_numbers(text: &str, findings: &mut Findings) {
|
||||
let mut international = Vec::new();
|
||||
for m in PHONE_INTL.find_iter(text) {
|
||||
let number = digits(m.as_str());
|
||||
if (8..=15).contains(&number.len()) && stands_alone(text, m.start() + 1, m.end()) {
|
||||
findings.insert(number);
|
||||
international.push(m.range());
|
||||
}
|
||||
}
|
||||
for m in PHONE_NATIONAL.find_iter(text) {
|
||||
let number = digits(m.as_str());
|
||||
// Not the tail of an international number already counted
|
||||
if international.iter().any(|r| r.contains(&m.start())) {
|
||||
continue;
|
||||
}
|
||||
if (9..=11).contains(&number.len())
|
||||
&& stands_alone(text, m.start(), m.end())
|
||||
&& !text[..m.start()].ends_with('+')
|
||||
&& word_near(text, m.start(), m.end(), PHONE_WORDS)
|
||||
{
|
||||
findings.insert(number);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Date of birth --------------------------------------------------------
|
||||
|
||||
static DATE_ISO: LazyLock<Regex> = LazyLock::new(|| re(r"\b(\d{4})-(\d{2})-(\d{2})\b"));
|
||||
static DATE_NUMERIC: LazyLock<Regex> =
|
||||
LazyLock::new(|| re(r"\b(\d{1,2})[./-](\d{1,2})[./-](\d{4})\b"));
|
||||
static DATE_WORDS: LazyLock<Regex> = LazyLock::new(|| {
|
||||
re(
|
||||
r"(?i)\b(?:(\d{1,2})\s+(jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)[a-z]*\.?,?\s+(\d{4})|(jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)[a-z]*\.?\s+(\d{1,2}),?\s+(\d{4}))\b",
|
||||
)
|
||||
});
|
||||
|
||||
const BIRTH_WORDS: &[&str] = &[
|
||||
"born",
|
||||
"birth",
|
||||
"dob",
|
||||
"d.o.b",
|
||||
"birthday",
|
||||
"birthdate",
|
||||
"geburtsdatum",
|
||||
"geboren",
|
||||
"naissance",
|
||||
"né le",
|
||||
"née le",
|
||||
"nacimiento",
|
||||
"nacido",
|
||||
"nacida",
|
||||
"nascita",
|
||||
"nato il",
|
||||
"nata il",
|
||||
"geboortedatum",
|
||||
"födelsedatum",
|
||||
"fødselsdato",
|
||||
"syntymäaika",
|
||||
"urodzenia",
|
||||
"nascimento",
|
||||
];
|
||||
|
||||
fn valid_date(year: u32, month: u32, day: u32) -> bool {
|
||||
let days = match month {
|
||||
1 | 3 | 5 | 7 | 8 | 10 | 12 => 31,
|
||||
4 | 6 | 9 | 11 => 30,
|
||||
2 if year.is_multiple_of(4) && (!year.is_multiple_of(100) || year.is_multiple_of(400)) => {
|
||||
29
|
||||
}
|
||||
2 => 28,
|
||||
_ => return false,
|
||||
};
|
||||
(1900..=2100).contains(&year) && (1..=days).contains(&day)
|
||||
}
|
||||
|
||||
fn month_number(name: &str) -> u32 {
|
||||
const MONTHS: [&str; 12] = [
|
||||
"jan", "feb", "mar", "apr", "may", "jun", "jul", "aug", "sep", "oct", "nov", "dec",
|
||||
];
|
||||
let name = name.to_lowercase();
|
||||
MONTHS
|
||||
.iter()
|
||||
.position(|m| *m == name)
|
||||
.map_or(0, |i| i as u32 + 1)
|
||||
}
|
||||
|
||||
fn date_of_birth(text: &str, findings: &mut Findings) {
|
||||
let mut add = |start: usize, end: usize, key: String| {
|
||||
if word_near(text, start, end, BIRTH_WORDS) {
|
||||
findings.insert(key);
|
||||
}
|
||||
};
|
||||
let num = |s: &str| s.parse::<u32>().unwrap_or(0);
|
||||
for c in DATE_ISO.captures_iter(text) {
|
||||
let (y, m, d) = (num(&c[1]), num(&c[2]), num(&c[3]));
|
||||
let whole = c.get(0).unwrap();
|
||||
if valid_date(y, m, d) {
|
||||
add(whole.start(), whole.end(), format!("{y:04}{m:02}{d:02}"));
|
||||
}
|
||||
}
|
||||
for c in DATE_NUMERIC.captures_iter(text) {
|
||||
let (a, b, y) = (num(&c[1]), num(&c[2]), num(&c[3]));
|
||||
let whole = c.get(0).unwrap();
|
||||
// Day first or month first: either reading that is a real date
|
||||
if valid_date(y, b, a) || valid_date(y, a, b) {
|
||||
add(whole.start(), whole.end(), whole.as_str().to_string());
|
||||
}
|
||||
}
|
||||
for c in DATE_WORDS.captures_iter(text) {
|
||||
let whole = c.get(0).unwrap();
|
||||
let (d, m, y) = match (c.get(1), c.get(4)) {
|
||||
(Some(d), _) => (num(d.as_str()), month_number(&c[2]), num(&c[3])),
|
||||
(_, Some(m)) => (num(&c[5]), month_number(m.as_str()), num(&c[6])),
|
||||
_ => continue,
|
||||
};
|
||||
if valid_date(y, m, d) {
|
||||
add(whole.start(), whole.end(), format!("{y:04}{m:02}{d:02}"));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Passport -------------------------------------------------------------
|
||||
|
||||
static PASSPORT: LazyLock<Regex> = LazyLock::new(|| re(r"\b[A-Z0-9]{6,9}\b"));
|
||||
|
||||
const PASSPORT_WORDS: &[&str] = &[
|
||||
"passport",
|
||||
"passeport",
|
||||
"reisepass",
|
||||
"pasaporte",
|
||||
"passaporto",
|
||||
"paspoort",
|
||||
"passnummer",
|
||||
"pass-nr",
|
||||
"passport no",
|
||||
"pasaporte n.º",
|
||||
"passaporte",
|
||||
];
|
||||
|
||||
fn passport(text: &str, findings: &mut Findings) {
|
||||
for m in PASSPORT.find_iter(text) {
|
||||
let value = m.as_str();
|
||||
if value.bytes().filter(u8::is_ascii_digit).count() >= 5
|
||||
&& word_near(text, m.start(), m.end(), PASSPORT_WORDS)
|
||||
{
|
||||
findings.insert(value);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// --- Keys and credentials -------------------------------------------------
|
||||
|
||||
static PRIVATE_KEY: LazyLock<Regex> = LazyLock::new(|| {
|
||||
re(
|
||||
r"-----BEGIN (?:(?:RSA|EC|DSA|OPENSSH|ENCRYPTED|PGP) )?PRIVATE KEY(?: BLOCK)?-----\s*([A-Za-z0-9+/=:\s-]{0,64})",
|
||||
)
|
||||
});
|
||||
|
||||
fn private_key(text: &str, findings: &mut Findings) {
|
||||
for c in PRIVATE_KEY.captures_iter(text) {
|
||||
// Each key once, by the start of its body
|
||||
let body: String = c[1].chars().filter(|c| !c.is_whitespace()).collect();
|
||||
let whole = c.get(0).unwrap();
|
||||
findings.insert(if body.is_empty() {
|
||||
format!("@{}", whole.start())
|
||||
} else {
|
||||
body
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
/// Published token formats: AWS access key IDs, GitHub tokens, Slack
|
||||
/// tokens, Stripe live secret and restricted keys, Google API keys.
|
||||
static CREDENTIAL: LazyLock<Regex> = LazyLock::new(|| {
|
||||
re(concat!(
|
||||
r"\b(?:",
|
||||
r"(?:AKIA|ASIA|ABIA|ACCA)[A-Z0-9]{16}",
|
||||
r"|gh[pousr]_[A-Za-z0-9]{36}",
|
||||
r"|github_pat_[A-Za-z0-9_]{82}",
|
||||
r"|xox[abposr]-[A-Za-z0-9-]{10,72}",
|
||||
r"|(?:sk|rk)_live_[A-Za-z0-9]{24,99}",
|
||||
r"|AIza[0-9A-Za-z_-]{35}",
|
||||
r")\b"
|
||||
))
|
||||
});
|
||||
|
||||
fn credentials(text: &str, findings: &mut Findings) {
|
||||
for m in CREDENTIAL.find_iter(text) {
|
||||
findings.insert(m.as_str());
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::mailflow::detectors::by_id;
|
||||
|
||||
fn count(id: &str, text: &str) -> usize {
|
||||
by_id(id).unwrap().count(text)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn payment_cards() {
|
||||
// Networks' and processors' published test numbers
|
||||
let text = "Visa 4242 4242 4242 4242, MC 5555-5555-5555-4444, Amex 378282246310005, \
|
||||
Discover 6011111111111117, JCB 3566002020360505, Diners 30569309025904, \
|
||||
UnionPay 6200000000000005, Mastercard 2-series 2223003122003222";
|
||||
assert_eq!(count("payment-card", text), 8);
|
||||
// Luhn fails, wrong network length, inside a longer number
|
||||
assert_eq!(count("payment-card", "4242424242424241"), 0);
|
||||
assert_eq!(count("payment-card", "378282246310005 0"), 1);
|
||||
assert_eq!(count("payment-card", "order 94242424242424242 shipped"), 0);
|
||||
// The same number twice counts once
|
||||
assert_eq!(
|
||||
count("payment-card", "4242424242424242 and 4242-4242-4242-4242"),
|
||||
1
|
||||
);
|
||||
// A card followed by a year
|
||||
assert_eq!(count("payment-card", "card 4242 4242 4242 4242 2031"), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ibans() {
|
||||
let text =
|
||||
"Pay GB29 NWBK 6016 1331 9268 19 or de89370400440532013000 (NL91ABNA0417164300).";
|
||||
assert_eq!(count("iban", text), 3);
|
||||
assert_eq!(count("iban", "GB29 NWBK 6016 1331 9268 18"), 0);
|
||||
// Runs into the next word: still found at the country's length
|
||||
assert_eq!(count("iban", "IBAN NL91ABNA0417164300 BIC ABNANL2A"), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn swift_codes_need_a_word() {
|
||||
assert_eq!(count("swift-bic", "SWIFT: DEUTDEFF500"), 1);
|
||||
assert_eq!(count("swift-bic", "BIC NWBKGB2L"), 1);
|
||||
assert_eq!(count("swift-bic", "HAPPYDAYS DEUTDEFF"), 0);
|
||||
// Not a country in positions 5–6
|
||||
assert_eq!(count("swift-bic", "BIC DEUTZZFF"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn email_and_phone_lists() {
|
||||
let list = "[email protected], [email protected], [email protected], [email protected]";
|
||||
assert_eq!(count("email-addresses", list), 3);
|
||||
assert_eq!(
|
||||
count("phone-numbers", "+44 20 7946 0958, +1 (415) 555-2671"),
|
||||
2
|
||||
);
|
||||
assert_eq!(count("phone-numbers", "call 020 7946 0958"), 0);
|
||||
assert_eq!(count("phone-numbers", "Tel: 020 7946 0958"), 1);
|
||||
assert_eq!(count("phone-numbers", "invoice 020 7946 0958"), 0);
|
||||
// One number, not also its national tail
|
||||
assert_eq!(count("phone-numbers", "Tel: +44 20 7946 0958"), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn dates_of_birth() {
|
||||
assert_eq!(count("date-of-birth", "DOB: 1984-02-29"), 1);
|
||||
assert_eq!(count("date-of-birth", "Geburtsdatum 31.12.1970"), 1);
|
||||
assert_eq!(count("date-of-birth", "born on March 3, 1962"), 1);
|
||||
assert_eq!(count("date-of-birth", "date of birth 3 Mar 1962"), 1);
|
||||
// Not a real date, no word, a meeting
|
||||
assert_eq!(count("date-of-birth", "DOB: 1985-02-29"), 0);
|
||||
assert_eq!(count("date-of-birth", "invoice 1984-02-29"), 0);
|
||||
assert_eq!(count("date-of-birth", "Meeting on 12/05/2026"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn passports_need_a_word() {
|
||||
assert_eq!(count("passport", "Passport number: 533380006"), 1);
|
||||
assert_eq!(count("passport", "Reisepass C01X00T47"), 1);
|
||||
assert_eq!(count("passport", "Order 533380006 shipped"), 0);
|
||||
// Mostly letters: a word, not a number
|
||||
assert_eq!(count("passport", "passport PASSWORD"), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn keys_and_credentials() {
|
||||
let key = "-----BEGIN OPENSSH PRIVATE KEY-----\nb3BlbnNzaC1rZXktdjEAAAAABG5vbmUAAAAEbm9uZQ\n-----END OPENSSH PRIVATE KEY-----";
|
||||
assert_eq!(count("private-key", key), 1);
|
||||
assert_eq!(count("private-key", "-----BEGIN PUBLIC KEY-----\nMFkw"), 0);
|
||||
// Documentation examples of each format
|
||||
let tokens = "AKIAIOSFODNN7EXAMPLE ghp_0123456789abcdefghijklmnopqrstuvwxyz \
|
||||
AIzaSyA-0123456789abcdefghijklmnopqrstu";
|
||||
assert_eq!(count("credentials", tokens), 3);
|
||||
assert_eq!(count("credentials", "AKIA123 ghp_short"), 0);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,227 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Check-digit algorithms, each from its public definition.
|
||||
|
||||
/// The Luhn check (ISO/IEC 7812-1, Annex B) over a string of ASCII digits.
|
||||
pub fn luhn(digits: &str) -> bool {
|
||||
if digits.len() < 2 || !digits.bytes().all(|b| b.is_ascii_digit()) {
|
||||
return false;
|
||||
}
|
||||
let sum: u32 = digits
|
||||
.bytes()
|
||||
.rev()
|
||||
.enumerate()
|
||||
.map(|(i, b)| {
|
||||
let d = u32::from(b - b'0');
|
||||
if i % 2 == 1 {
|
||||
let d = d * 2;
|
||||
if d > 9 { d - 9 } else { d }
|
||||
} else {
|
||||
d
|
||||
}
|
||||
})
|
||||
.sum();
|
||||
sum % 10 == 0
|
||||
}
|
||||
|
||||
/// ISO 13616 IBAN lengths, by country, from the IBAN registry.
|
||||
const IBAN_LENGTHS: &[(&str, usize)] = &[
|
||||
("AD", 24),
|
||||
("AE", 23),
|
||||
("AL", 28),
|
||||
("AT", 20),
|
||||
("AZ", 28),
|
||||
("BA", 20),
|
||||
("BE", 16),
|
||||
("BG", 22),
|
||||
("BH", 22),
|
||||
("BI", 27),
|
||||
("BR", 29),
|
||||
("BY", 28),
|
||||
("CH", 21),
|
||||
("CR", 22),
|
||||
("CY", 28),
|
||||
("CZ", 24),
|
||||
("DE", 22),
|
||||
("DJ", 27),
|
||||
("DK", 18),
|
||||
("DO", 28),
|
||||
("EE", 20),
|
||||
("EG", 29),
|
||||
("ES", 24),
|
||||
("FI", 18),
|
||||
("FK", 18),
|
||||
("FO", 18),
|
||||
("FR", 27),
|
||||
("GB", 22),
|
||||
("GE", 22),
|
||||
("GI", 23),
|
||||
("GL", 18),
|
||||
("GR", 27),
|
||||
("GT", 28),
|
||||
("HN", 28),
|
||||
("HR", 21),
|
||||
("HU", 28),
|
||||
("IE", 22),
|
||||
("IL", 23),
|
||||
("IQ", 23),
|
||||
("IS", 26),
|
||||
("IT", 27),
|
||||
("JO", 30),
|
||||
("KW", 30),
|
||||
("KZ", 20),
|
||||
("LB", 28),
|
||||
("LC", 32),
|
||||
("LI", 21),
|
||||
("LT", 20),
|
||||
("LU", 20),
|
||||
("LV", 21),
|
||||
("LY", 25),
|
||||
("MC", 27),
|
||||
("MD", 24),
|
||||
("ME", 22),
|
||||
("MK", 19),
|
||||
("MN", 20),
|
||||
("MR", 27),
|
||||
("MT", 31),
|
||||
("MU", 30),
|
||||
("NI", 28),
|
||||
("NL", 18),
|
||||
("NO", 15),
|
||||
("OM", 23),
|
||||
("PK", 24),
|
||||
("PL", 28),
|
||||
("PS", 29),
|
||||
("PT", 25),
|
||||
("QA", 29),
|
||||
("RO", 24),
|
||||
("RS", 22),
|
||||
("RU", 33),
|
||||
("SA", 24),
|
||||
("SC", 31),
|
||||
("SD", 18),
|
||||
("SE", 24),
|
||||
("SI", 19),
|
||||
("SK", 24),
|
||||
("SM", 27),
|
||||
("SO", 23),
|
||||
("ST", 25),
|
||||
("SV", 28),
|
||||
("TL", 23),
|
||||
("TN", 24),
|
||||
("TR", 26),
|
||||
("UA", 29),
|
||||
("VA", 22),
|
||||
("VG", 24),
|
||||
("XK", 20),
|
||||
("YE", 30),
|
||||
];
|
||||
|
||||
/// The IBAN length for a country code, if the country uses IBANs.
|
||||
pub fn iban_length(country: &str) -> Option<usize> {
|
||||
IBAN_LENGTHS
|
||||
.iter()
|
||||
.find(|(code, _)| *code == country)
|
||||
.map(|(_, len)| *len)
|
||||
}
|
||||
|
||||
/// ISO 13616 / ISO 7064 MOD 97-10 over an IBAN with no spaces, upper case:
|
||||
/// move the first four characters to the end, turn letters into 10–35, and
|
||||
/// the number mod 97 must be 1. Also checks the country's length.
|
||||
pub fn iban(iban: &str) -> bool {
|
||||
if iban.len() < 5
|
||||
|| !iban
|
||||
.bytes()
|
||||
.all(|b| b.is_ascii_uppercase() || b.is_ascii_digit())
|
||||
{
|
||||
return false;
|
||||
}
|
||||
if iban_length(&iban[..2]) != Some(iban.len())
|
||||
|| !iban[2..4].bytes().all(|b| b.is_ascii_digit())
|
||||
{
|
||||
return false;
|
||||
}
|
||||
let mut remainder: u32 = 0;
|
||||
for b in iban[4..].bytes().chain(iban[..4].bytes()) {
|
||||
let value = if b.is_ascii_digit() {
|
||||
u32::from(b - b'0')
|
||||
} else {
|
||||
u32::from(b - b'A') + 10
|
||||
};
|
||||
remainder = if value >= 10 {
|
||||
(remainder * 100 + value) % 97
|
||||
} else {
|
||||
(remainder * 10 + value) % 97
|
||||
};
|
||||
}
|
||||
remainder == 1
|
||||
}
|
||||
|
||||
/// ISO 3166-1 alpha-2 country codes, for SWIFT/BIC positions 5–6.
|
||||
const COUNTRIES: &str = "AD AE AF AG AI AL AM AO AQ AR AS AT AU AW AX AZ BA BB BD BE BF BG BH BI BJ \
|
||||
BL BM BN BO BQ BR BS BT BV BW BY BZ CA CC CD CF CG CH CI CK CL CM CN CO CR CU CV CW CX CY CZ DE DJ \
|
||||
DK DM DO DZ EC EE EG EH ER ES ET FI FJ FK FM FO FR GA GB GD GE GF GG GH GI GL GM GN GP GQ GR GS GT \
|
||||
GU GW GY HK HM HN HR HT HU ID IE IL IM IN IO IQ IR IS IT JE JM JO JP KE KG KH KI KM KN KP KR KW KY \
|
||||
KZ LA LB LC LI LK LR LS LT LU LV LY MA MC MD ME MF MG MH MK ML MM MN MO MP MQ MR MS MT MU MV MW MX \
|
||||
MY MZ NA NC NE NF NG NI NL NO NP NR NU NZ OM PA PE PF PG PH PK PL PM PN PR PS PT PW PY QA RE RO RS \
|
||||
RU RW SA SB SC SD SE SG SH SI SJ SK SL SM SN SO SR SS ST SV SX SY SZ TC TD TF TG TH TJ TK TL TM TN \
|
||||
TO TR TT TV TW TZ UA UG UM US UY UZ VA VC VE VG VI VN VU WF WS XK YE YT ZA ZM ZW";
|
||||
|
||||
pub fn is_country(code: &str) -> bool {
|
||||
code.len() == 2 && COUNTRIES.split(' ').any(|c| c == code)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn luhn_known_numbers() {
|
||||
// Published test card numbers
|
||||
for good in [
|
||||
"4242424242424242",
|
||||
"5555555555554444",
|
||||
"378282246310005",
|
||||
"79927398713",
|
||||
] {
|
||||
assert!(luhn(good), "{good}");
|
||||
}
|
||||
for bad in ["4242424242424241", "79927398710", "1", "12a4"] {
|
||||
assert!(!luhn(bad), "{bad}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn iban_registry_examples() {
|
||||
// The IBAN registry's own examples
|
||||
for good in [
|
||||
"GB29NWBK60161331926819",
|
||||
"DE89370400440532013000",
|
||||
"FR1420041010050500013M02606",
|
||||
"NL91ABNA0417164300",
|
||||
"BE68539007547034",
|
||||
"NO9386011117947",
|
||||
"CH9300762011623852957",
|
||||
] {
|
||||
assert!(iban(good), "{good}");
|
||||
}
|
||||
for bad in [
|
||||
"GB29NWBK60161331926818", // check fails
|
||||
"GB29NWBK6016133192681", // too short for GB
|
||||
"ZZ29NWBK60161331926819", // no such country
|
||||
"DE8937040044053201300A", // letters where DE has none still fail mod 97
|
||||
] {
|
||||
assert!(!iban(bad), "{bad}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn countries() {
|
||||
assert!(is_country("DE") && is_country("US") && is_country("XK"));
|
||||
assert!(!is_country("ZZ") && !is_country("D"));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,199 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Detectors (dlp-and-mail-flow-rules spec, §2.3): each finds one kind of
|
||||
//! identifier in text and reports the distinct ones it found.
|
||||
//!
|
||||
//! A detector is one of two strengths:
|
||||
//!
|
||||
//! - **Checked**: the identifier carries a published check digit or
|
||||
//! checksum, so a random number rarely passes; found on its own.
|
||||
//! - **Needs a word**: the format alone is too common, so a candidate counts
|
||||
//! only with a corroborating word within [`WINDOW`] characters either
|
||||
//! side.
|
||||
//!
|
||||
//! Findings are distinct normalized values (digits only, upper case), so the
|
||||
//! same card number pasted twice counts once. They stay in memory: callers
|
||||
//! read only [`Findings::len`].
|
||||
|
||||
pub mod any;
|
||||
pub mod checks;
|
||||
|
||||
use ahash::AHashSet;
|
||||
|
||||
/// How far, in characters, a corroborating word may be from a candidate.
|
||||
pub const WINDOW: usize = 50;
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum Strength {
|
||||
Checked,
|
||||
NeedsWord,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum Region {
|
||||
Any,
|
||||
Us,
|
||||
Uk,
|
||||
Canada,
|
||||
Australia,
|
||||
Eu,
|
||||
Europe,
|
||||
Asia,
|
||||
Americas,
|
||||
Africa,
|
||||
}
|
||||
|
||||
/// The distinct values one detector found.
|
||||
#[derive(Debug, Default)]
|
||||
pub struct Findings(AHashSet<String>);
|
||||
|
||||
impl Findings {
|
||||
pub fn insert(&mut self, value: impl Into<String>) {
|
||||
self.0.insert(value.into());
|
||||
}
|
||||
|
||||
pub fn len(&self) -> usize {
|
||||
self.0.len()
|
||||
}
|
||||
|
||||
pub fn is_empty(&self) -> bool {
|
||||
self.0.is_empty()
|
||||
}
|
||||
}
|
||||
|
||||
pub struct Detector {
|
||||
/// Stable id, stored in rules: `payment-card`, `iban`, `us-ssn`.
|
||||
pub id: &'static str,
|
||||
pub name: &'static str,
|
||||
pub region: Region,
|
||||
pub strength: Strength,
|
||||
find: fn(&str, &mut Findings),
|
||||
}
|
||||
|
||||
impl Detector {
|
||||
pub const fn new(
|
||||
id: &'static str,
|
||||
name: &'static str,
|
||||
region: Region,
|
||||
strength: Strength,
|
||||
find: fn(&str, &mut Findings),
|
||||
) -> Self {
|
||||
Self {
|
||||
id,
|
||||
name,
|
||||
region,
|
||||
strength,
|
||||
find,
|
||||
}
|
||||
}
|
||||
|
||||
/// Adds what this detector finds in `text` to `findings`. Call once per
|
||||
/// piece of text (subject, each part, each attachment) with the same
|
||||
/// `findings`, then read its length.
|
||||
pub fn find(&self, text: &str, findings: &mut Findings) {
|
||||
(self.find)(text, findings)
|
||||
}
|
||||
|
||||
/// The distinct values found in one text.
|
||||
pub fn count(&self, text: &str) -> usize {
|
||||
let mut findings = Findings::default();
|
||||
self.find(text, &mut findings);
|
||||
findings.len()
|
||||
}
|
||||
}
|
||||
|
||||
/// Every detector, in the order the console lists them.
|
||||
pub fn all() -> impl Iterator<Item = &'static Detector> {
|
||||
any::DETECTORS.iter()
|
||||
}
|
||||
|
||||
pub fn by_id(id: &str) -> Option<&'static Detector> {
|
||||
all().find(|detector| detector.id == id)
|
||||
}
|
||||
|
||||
/// Whether one of `words` appears, as a whole word and ignoring case, within
|
||||
/// [`WINDOW`] characters before `start` or after `end` (byte offsets of the
|
||||
/// candidate in `text`). The window is widened by the longest word, so a
|
||||
/// word that reaches into it still counts whole.
|
||||
pub fn word_near(text: &str, start: usize, end: usize, words: &[&str]) -> bool {
|
||||
let reach = WINDOW + words.iter().map(|w| w.chars().count()).max().unwrap_or(0);
|
||||
let before = text[..start]
|
||||
.char_indices()
|
||||
.rev()
|
||||
.nth(reach - 1)
|
||||
.map_or(0, |(i, _)| i);
|
||||
let after = text[end..]
|
||||
.char_indices()
|
||||
.nth(reach)
|
||||
.map_or(text.len(), |(i, _)| end + i);
|
||||
let window = text[before..after].to_lowercase();
|
||||
words.iter().any(|word| contains_word(&window, word))
|
||||
}
|
||||
|
||||
/// Whether `word` (lower case) appears in `haystack` (lower case) with no
|
||||
/// letter or digit on either side.
|
||||
pub fn contains_word(haystack: &str, word: &str) -> bool {
|
||||
haystack.match_indices(word).any(|(i, _)| {
|
||||
let before_ok = haystack[..i]
|
||||
.chars()
|
||||
.next_back()
|
||||
.is_none_or(|c| !c.is_alphanumeric());
|
||||
let after_ok = haystack[i + word.len()..]
|
||||
.chars()
|
||||
.next()
|
||||
.is_none_or(|c| !c.is_alphanumeric());
|
||||
before_ok && after_ok
|
||||
})
|
||||
}
|
||||
|
||||
/// Whether the match at `start..end` stands alone: no digit or letter
|
||||
/// directly before or after it, so `123-45-6789` isn't found inside a
|
||||
/// longer run of digits.
|
||||
pub fn stands_alone(text: &str, start: usize, end: usize) -> bool {
|
||||
let before = text[..start].chars().next_back();
|
||||
let after = text[end..].chars().next();
|
||||
before.is_none_or(|c| !c.is_alphanumeric()) && after.is_none_or(|c| !c.is_alphanumeric())
|
||||
}
|
||||
|
||||
/// The ASCII digits of `s`.
|
||||
pub fn digits(s: &str) -> String {
|
||||
s.chars().filter(char::is_ascii_digit).collect()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn words_are_whole_and_near() {
|
||||
let text = "Your passport number is X1234567, thanks";
|
||||
let start = text.find("X123").unwrap();
|
||||
assert!(word_near(text, start, start + 8, &["passport"]));
|
||||
assert!(!word_near(text, start, start + 8, &["pass"]));
|
||||
let far = format!("passport{}X1234567", " ".repeat(60));
|
||||
let start = far.find("X123").unwrap();
|
||||
assert!(!word_near(&far, start, start + 8, &["passport"]));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn near_counts_characters_not_bytes() {
|
||||
// 45 two-byte characters between the word and the candidate: within
|
||||
// 50 characters, though over 50 bytes
|
||||
let text = format!("passport {} X1234567", "é".repeat(45));
|
||||
let start = text.find("X123").unwrap();
|
||||
assert!(word_near(&text, start, start + 8, &["passport"]));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ids_are_unique() {
|
||||
let mut seen = AHashSet::new();
|
||||
for detector in all() {
|
||||
assert!(seen.insert(detector.id), "duplicate id {}", detector.id);
|
||||
assert!(by_id(detector.id).is_some());
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,692 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! The text of an attachment, for the detectors (§2.3), or why there isn't
|
||||
//! one.
|
||||
//!
|
||||
//! Read: text files (plain, CSV, JSON, XML, HTML), Office Open XML (DOCX,
|
||||
//! XLSX, PPTX) and OpenDocument (ODT, ODS, ODP) documents, and ZIP archives
|
||||
//! one level deep. **Can't be inspected**: encrypted or password-protected
|
||||
//! files, PDF (settled answer 2), the older binary Office formats, archives
|
||||
//! inside archives, and anything past the limits. Everything else (images,
|
||||
//! audio, programs) has no text to read and is neither.
|
||||
//!
|
||||
//! Office files are ZIP archives of XML, read here with the `zip` and
|
||||
//! `quick-xml` crates the server already uses: no outside converter runs.
|
||||
|
||||
use quick_xml::{Reader, XmlVersion, events::Event};
|
||||
use std::io::{Cursor, Read};
|
||||
|
||||
/// How much may be unpacked from one attachment, and from how many entries.
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub struct Limits {
|
||||
pub max_unpacked: u64,
|
||||
pub max_entries: usize,
|
||||
}
|
||||
|
||||
impl Default for Limits {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
max_unpacked: 50 * 1024 * 1024,
|
||||
max_entries: 10_000,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub enum Extracted {
|
||||
/// The text to check.
|
||||
Text(String),
|
||||
/// A kind of file with no text in it: nothing to check, nothing missed.
|
||||
NoText,
|
||||
/// A file that may hold text the detectors couldn't read.
|
||||
NotInspectable(Why),
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum Why {
|
||||
Encrypted,
|
||||
Pdf,
|
||||
LegacyOffice,
|
||||
NestedArchive,
|
||||
TooLarge,
|
||||
Damaged,
|
||||
}
|
||||
|
||||
impl Why {
|
||||
pub fn as_str(&self) -> &'static str {
|
||||
match self {
|
||||
Why::Encrypted => "encrypted",
|
||||
Why::Pdf => "pdf",
|
||||
Why::LegacyOffice => "legacy-office",
|
||||
Why::NestedArchive => "nested-archive",
|
||||
Why::TooLarge => "too-large",
|
||||
Why::Damaged => "damaged",
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const OLE_MAGIC: &[u8] = &[0xD0, 0xCF, 0x11, 0xE0, 0xA1, 0xB1, 0x1A, 0xE1];
|
||||
const ZIP_MAGIC: &[u8] = b"PK\x03\x04";
|
||||
|
||||
/// What an attachment says, from its declared type, its file name and, above
|
||||
/// all, its first bytes.
|
||||
pub fn extract(
|
||||
content_type: &str,
|
||||
file_name: Option<&str>,
|
||||
data: &[u8],
|
||||
limits: &Limits,
|
||||
) -> Extracted {
|
||||
extract_at(content_type, file_name, data, limits, 0)
|
||||
}
|
||||
|
||||
fn extract_at(
|
||||
content_type: &str,
|
||||
file_name: Option<&str>,
|
||||
data: &[u8],
|
||||
limits: &Limits,
|
||||
depth: u8,
|
||||
) -> Extracted {
|
||||
let content_type = content_type.to_ascii_lowercase();
|
||||
let extension = file_name
|
||||
.and_then(|name| name.rsplit_once('.'))
|
||||
.map(|(_, ext)| ext.to_ascii_lowercase())
|
||||
.unwrap_or_default();
|
||||
|
||||
if data.len() as u64 > limits.max_unpacked {
|
||||
return Extracted::NotInspectable(Why::TooLarge);
|
||||
}
|
||||
if data.starts_with(b"%PDF-") || content_type == "application/pdf" || extension == "pdf" {
|
||||
return Extracted::NotInspectable(Why::Pdf);
|
||||
}
|
||||
if data.starts_with(OLE_MAGIC) {
|
||||
// An encrypted OOXML file is an OLE container holding the encrypted
|
||||
// package; any other OLE file is a legacy .doc, .xls or .ppt
|
||||
return Extracted::NotInspectable(if has_utf16(data, "EncryptedPackage") {
|
||||
Why::Encrypted
|
||||
} else {
|
||||
Why::LegacyOffice
|
||||
});
|
||||
}
|
||||
if data.starts_with(ZIP_MAGIC) {
|
||||
if depth > 0 {
|
||||
return Extracted::NotInspectable(Why::NestedArchive);
|
||||
}
|
||||
return zip(data, limits);
|
||||
}
|
||||
if is_text(&content_type, &extension) {
|
||||
let text = decode_text(data);
|
||||
return Extracted::Text(
|
||||
if content_type == "text/html" || matches!(extension.as_str(), "html" | "htm") {
|
||||
strip_html(&text)
|
||||
} else {
|
||||
text
|
||||
},
|
||||
);
|
||||
}
|
||||
Extracted::NoText
|
||||
}
|
||||
|
||||
fn is_text(content_type: &str, extension: &str) -> bool {
|
||||
content_type.starts_with("text/")
|
||||
|| matches!(
|
||||
content_type,
|
||||
"application/json"
|
||||
| "application/xml"
|
||||
| "application/csv"
|
||||
| "application/x-csv"
|
||||
| "message/rfc822"
|
||||
)
|
||||
|| matches!(
|
||||
extension,
|
||||
"txt"
|
||||
| "csv"
|
||||
| "tsv"
|
||||
| "json"
|
||||
| "xml"
|
||||
| "md"
|
||||
| "log"
|
||||
| "html"
|
||||
| "htm"
|
||||
| "eml"
|
||||
| "ics"
|
||||
| "vcf"
|
||||
)
|
||||
}
|
||||
|
||||
/// UTF-16 with a byte order mark, else UTF-8 (lossy).
|
||||
fn decode_text(data: &[u8]) -> String {
|
||||
let utf16 = |bytes: &[u8], big: bool| {
|
||||
let units: Vec<u16> = bytes
|
||||
.chunks_exact(2)
|
||||
.map(|c| {
|
||||
if big {
|
||||
u16::from_be_bytes([c[0], c[1]])
|
||||
} else {
|
||||
u16::from_le_bytes([c[0], c[1]])
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
String::from_utf16_lossy(&units)
|
||||
};
|
||||
match data {
|
||||
[0xFF, 0xFE, rest @ ..] => utf16(rest, false),
|
||||
[0xFE, 0xFF, rest @ ..] => utf16(rest, true),
|
||||
[0xEF, 0xBB, 0xBF, rest @ ..] => String::from_utf8_lossy(rest).into_owned(),
|
||||
_ => String::from_utf8_lossy(data).into_owned(),
|
||||
}
|
||||
}
|
||||
|
||||
fn has_utf16(data: &[u8], needle: &str) -> bool {
|
||||
let needle: Vec<u8> = needle.encode_utf16().flat_map(u16::to_le_bytes).collect();
|
||||
data.windows(needle.len()).any(|w| w == needle.as_slice())
|
||||
}
|
||||
|
||||
/// Tags out, the common entities decoded, block ends as new lines.
|
||||
fn strip_html(html: &str) -> String {
|
||||
let mut out = String::with_capacity(html.len());
|
||||
let mut in_tag = false;
|
||||
let mut skip_until: Option<&str> = None;
|
||||
let lower = html.to_ascii_lowercase();
|
||||
let mut i = 0;
|
||||
let bytes = html.as_bytes();
|
||||
while i < bytes.len() {
|
||||
if let Some(end) = skip_until {
|
||||
match lower[i..].find(end) {
|
||||
Some(at) => {
|
||||
i += at + end.len();
|
||||
skip_until = None;
|
||||
}
|
||||
None => break,
|
||||
}
|
||||
continue;
|
||||
}
|
||||
let c = bytes[i];
|
||||
if in_tag {
|
||||
if c == b'>' {
|
||||
in_tag = false;
|
||||
}
|
||||
i += 1;
|
||||
continue;
|
||||
}
|
||||
if c == b'<' {
|
||||
if lower[i..].starts_with("<script") {
|
||||
skip_until = Some("</script>");
|
||||
} else if lower[i..].starts_with("<style") {
|
||||
skip_until = Some("</style>");
|
||||
} else {
|
||||
if [
|
||||
"<br", "<p", "</p", "<div", "</div", "<tr", "<li", "<td", "<th",
|
||||
]
|
||||
.iter()
|
||||
.any(|t| lower[i..].starts_with(t))
|
||||
{
|
||||
out.push(
|
||||
if lower[i..].starts_with("<td") || lower[i..].starts_with("<th") {
|
||||
'\t'
|
||||
} else {
|
||||
'\n'
|
||||
},
|
||||
);
|
||||
}
|
||||
in_tag = true;
|
||||
}
|
||||
i += 1;
|
||||
continue;
|
||||
}
|
||||
// Copy up to the next tag
|
||||
let next = html[i..].find('<').map_or(html.len(), |at| i + at);
|
||||
out.push_str(&html[i..next]);
|
||||
i = next;
|
||||
}
|
||||
for (entity, text) in [
|
||||
(" ", " "),
|
||||
("<", "<"),
|
||||
(">", ">"),
|
||||
(""", "\""),
|
||||
("'", "'"),
|
||||
("&", "&"),
|
||||
] {
|
||||
out = out.replace(entity, text);
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
/// A ZIP file: an Office document, an OpenDocument, or an archive.
|
||||
fn zip(data: &[u8], limits: &Limits) -> Extracted {
|
||||
let Ok(mut archive) = zip::ZipArchive::new(Cursor::new(data)) else {
|
||||
return Extracted::NotInspectable(Why::Damaged);
|
||||
};
|
||||
if archive.len() > limits.max_entries {
|
||||
return Extracted::NotInspectable(Why::TooLarge);
|
||||
}
|
||||
let mut names = Vec::with_capacity(archive.len());
|
||||
let mut declared: u64 = 0;
|
||||
for i in 0..archive.len() {
|
||||
let Ok(entry) = archive.by_index_raw(i) else {
|
||||
return Extracted::NotInspectable(Why::Damaged);
|
||||
};
|
||||
if entry.encrypted() {
|
||||
return Extracted::NotInspectable(Why::Encrypted);
|
||||
}
|
||||
declared = declared.saturating_add(entry.size());
|
||||
names.push(entry.name().to_string());
|
||||
}
|
||||
if declared > limits.max_unpacked {
|
||||
return Extracted::NotInspectable(Why::TooLarge);
|
||||
}
|
||||
let mut budget = limits.max_unpacked;
|
||||
let mut read =
|
||||
|archive: &mut zip::ZipArchive<Cursor<&[u8]>>, name: &str| -> Result<Vec<u8>, Why> {
|
||||
let entry = archive.by_name(name).map_err(|_| Why::Damaged)?;
|
||||
let mut bytes = Vec::new();
|
||||
// Declared sizes can lie: stop at the budget whatever they say
|
||||
entry
|
||||
.take(budget + 1)
|
||||
.read_to_end(&mut bytes)
|
||||
.map_err(|_| Why::Damaged)?;
|
||||
if bytes.len() as u64 > budget {
|
||||
return Err(Why::TooLarge);
|
||||
}
|
||||
budget -= bytes.len() as u64;
|
||||
Ok(bytes)
|
||||
};
|
||||
|
||||
let has = |name: &str| names.iter().any(|n| n == name);
|
||||
let mut text = String::new();
|
||||
let result: Result<(), Why> = (|| {
|
||||
if has("[Content_Types].xml") {
|
||||
// Office Open XML: the parts that hold what a person wrote
|
||||
let mut shared = Vec::new();
|
||||
if has("xl/sharedStrings.xml") {
|
||||
shared = xml_strings(&read(&mut archive, "xl/sharedStrings.xml")?, "si");
|
||||
}
|
||||
for name in names.iter().filter(|n| ooxml_text_part(n)) {
|
||||
let xml = read(&mut archive, name)?;
|
||||
if name.starts_with("xl/worksheets/") {
|
||||
xlsx_sheet(&xml, &mut text);
|
||||
} else {
|
||||
xml_text(&xml, &mut text);
|
||||
}
|
||||
text.push('\n');
|
||||
}
|
||||
text.extend(shared.iter().map(|s| format!("{s}\n")));
|
||||
} else if names.first().is_some_and(|n| n == "mimetype")
|
||||
&& read(&mut archive, "mimetype")?.starts_with(b"application/vnd.oasis.opendocument")
|
||||
{
|
||||
// OpenDocument: an encrypted one says so in its manifest
|
||||
if has("META-INF/manifest.xml")
|
||||
&& contains(
|
||||
&read(&mut archive, "META-INF/manifest.xml")?,
|
||||
b"encryption-data",
|
||||
)
|
||||
{
|
||||
return Err(Why::Encrypted);
|
||||
}
|
||||
for name in ["content.xml", "styles.xml"] {
|
||||
if has(name) {
|
||||
xml_text(&read(&mut archive, name)?, &mut text);
|
||||
text.push('\n');
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// An archive: each file inside, one level deep
|
||||
for name in names.iter().filter(|n| !n.ends_with('/')) {
|
||||
let bytes = read(&mut archive, name)?;
|
||||
match extract_at("", Some(name), &bytes, limits, 1) {
|
||||
Extracted::Text(inner) => {
|
||||
text.push_str(&inner);
|
||||
text.push('\n');
|
||||
}
|
||||
Extracted::NoText => {}
|
||||
Extracted::NotInspectable(why) => return Err(why),
|
||||
}
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
})();
|
||||
match result {
|
||||
Ok(()) => Extracted::Text(text),
|
||||
Err(why) => Extracted::NotInspectable(why),
|
||||
}
|
||||
}
|
||||
|
||||
fn ooxml_text_part(name: &str) -> bool {
|
||||
let xml = name.ends_with(".xml");
|
||||
xml && (name == "word/document.xml"
|
||||
|| [
|
||||
"word/header",
|
||||
"word/footer",
|
||||
"word/footnotes",
|
||||
"word/endnotes",
|
||||
"word/comments",
|
||||
]
|
||||
.iter()
|
||||
.any(|p| name.starts_with(p))
|
||||
|| name.starts_with("xl/worksheets/sheet")
|
||||
|| name.starts_with("ppt/slides/slide")
|
||||
|| name.starts_with("ppt/notesSlides/"))
|
||||
}
|
||||
|
||||
fn contains(haystack: &[u8], needle: &[u8]) -> bool {
|
||||
haystack.windows(needle.len()).any(|w| w == needle)
|
||||
}
|
||||
|
||||
/// The local name of a tag, without its namespace prefix.
|
||||
fn local(name: &[u8]) -> &[u8] {
|
||||
name.rsplit(|b| *b == b':').next().unwrap_or(name)
|
||||
}
|
||||
|
||||
fn push_entity(entity: &[u8], out: &mut String) {
|
||||
match entity {
|
||||
b"lt" => out.push('<'),
|
||||
b"gt" => out.push('>'),
|
||||
b"amp" => out.push('&'),
|
||||
b"apos" => out.push('\''),
|
||||
b"quot" => out.push('"'),
|
||||
_ => {
|
||||
let code = match entity {
|
||||
[b'#', b'x' | b'X', hex @ ..] => std::str::from_utf8(hex)
|
||||
.ok()
|
||||
.and_then(|h| u32::from_str_radix(h, 16).ok()),
|
||||
[b'#', dec @ ..] => std::str::from_utf8(dec).ok().and_then(|d| d.parse().ok()),
|
||||
_ => None,
|
||||
};
|
||||
if let Some(c) = code.and_then(char::from_u32) {
|
||||
out.push(c);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Every text node, runs joined as written, a new line after each paragraph
|
||||
/// or row and a tab after each cell, so a number split across runs is whole
|
||||
/// again.
|
||||
fn xml_text(xml: &[u8], out: &mut String) {
|
||||
let mut reader = Reader::from_reader(xml);
|
||||
let mut buf = Vec::new();
|
||||
loop {
|
||||
match reader.read_event_into(&mut buf) {
|
||||
Ok(Event::Text(t)) => {
|
||||
if let Ok(text) = t.xml_content(XmlVersion::Implicit1_0) {
|
||||
out.push_str(&text);
|
||||
}
|
||||
}
|
||||
Ok(Event::CData(t)) => out.push_str(&String::from_utf8_lossy(&t)),
|
||||
Ok(Event::GeneralRef(entity)) => push_entity(&entity, out),
|
||||
Ok(Event::End(e)) => match local(e.name().as_ref()) {
|
||||
b"p" | b"h" | b"tr" | b"row" | b"table-row" | b"br" => out.push('\n'),
|
||||
b"tc" | b"c" | b"table-cell" | b"tab" => out.push('\t'),
|
||||
_ => {}
|
||||
},
|
||||
Ok(Event::Empty(e)) => match local(e.name().as_ref()) {
|
||||
b"br" | b"line-break" => out.push('\n'),
|
||||
b"tab" | b"s" => out.push(' '),
|
||||
_ => {}
|
||||
},
|
||||
Ok(Event::Eof) | Err(_) => break,
|
||||
_ => {}
|
||||
}
|
||||
buf.clear();
|
||||
}
|
||||
}
|
||||
|
||||
/// The text of each `item` element (a shared string in XLSX).
|
||||
fn xml_strings(xml: &[u8], item: &str) -> Vec<String> {
|
||||
let mut reader = Reader::from_reader(xml);
|
||||
let mut buf = Vec::new();
|
||||
let mut items = Vec::new();
|
||||
let mut current: Option<String> = None;
|
||||
loop {
|
||||
match reader.read_event_into(&mut buf) {
|
||||
Ok(Event::Start(e)) if local(e.name().as_ref()) == item.as_bytes() => {
|
||||
current = Some(String::new())
|
||||
}
|
||||
Ok(Event::End(e)) if local(e.name().as_ref()) == item.as_bytes() => {
|
||||
items.extend(current.take());
|
||||
}
|
||||
Ok(Event::Text(t)) => {
|
||||
if let (Some(s), Ok(text)) =
|
||||
(current.as_mut(), t.xml_content(XmlVersion::Implicit1_0))
|
||||
{
|
||||
s.push_str(&text);
|
||||
}
|
||||
}
|
||||
Ok(Event::GeneralRef(entity)) => {
|
||||
if let Some(s) = current.as_mut() {
|
||||
push_entity(&entity, s);
|
||||
}
|
||||
}
|
||||
Ok(Event::Eof) | Err(_) => break,
|
||||
_ => {}
|
||||
}
|
||||
buf.clear();
|
||||
}
|
||||
items
|
||||
}
|
||||
|
||||
/// A worksheet's cell values: numbers and inline strings. Cells holding a
|
||||
/// shared string are skipped here; the shared strings are read whole.
|
||||
fn xlsx_sheet(xml: &[u8], out: &mut String) {
|
||||
let mut reader = Reader::from_reader(xml);
|
||||
let mut buf = Vec::new();
|
||||
let mut shared_cell = false;
|
||||
let mut in_value = false;
|
||||
loop {
|
||||
match reader.read_event_into(&mut buf) {
|
||||
Ok(Event::Start(e)) => match local(e.name().as_ref()) {
|
||||
b"c" => {
|
||||
shared_cell = e
|
||||
.attributes()
|
||||
.flatten()
|
||||
.any(|a| a.key.as_ref() == b"t" && a.value.as_ref() == b"s");
|
||||
}
|
||||
b"v" | b"t" => in_value = true,
|
||||
_ => {}
|
||||
},
|
||||
Ok(Event::End(e)) => match local(e.name().as_ref()) {
|
||||
b"v" | b"t" => in_value = false,
|
||||
b"c" => out.push('\t'),
|
||||
b"row" => out.push('\n'),
|
||||
_ => {}
|
||||
},
|
||||
// A shared string's cell holds only its index: the string itself
|
||||
// is added with the shared strings
|
||||
Ok(Event::Text(t)) if in_value && !shared_cell => {
|
||||
if let Ok(text) = t.xml_content(XmlVersion::Implicit1_0) {
|
||||
out.push_str(&text);
|
||||
}
|
||||
}
|
||||
Ok(Event::Eof) | Err(_) => break,
|
||||
_ => {}
|
||||
}
|
||||
buf.clear();
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use std::io::Write;
|
||||
use zip::{ZipWriter, write::SimpleFileOptions};
|
||||
|
||||
fn zip_of(files: &[(&str, &str)]) -> Vec<u8> {
|
||||
let mut zip = ZipWriter::new(Cursor::new(Vec::new()));
|
||||
for (name, body) in files {
|
||||
zip.start_file(*name, SimpleFileOptions::default()).unwrap();
|
||||
zip.write_all(body.as_bytes()).unwrap();
|
||||
}
|
||||
zip.finish().unwrap().into_inner()
|
||||
}
|
||||
|
||||
fn text_of(extracted: Extracted) -> String {
|
||||
match extracted {
|
||||
Extracted::Text(text) => text,
|
||||
other => panic!("expected text, got {other:?}"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn plain_text_and_html() {
|
||||
let limits = Limits::default();
|
||||
assert_eq!(
|
||||
text_of(extract("text/plain", None, b"card 4242", &limits)),
|
||||
"card 4242"
|
||||
);
|
||||
let utf16: Vec<u8> = [0xFF, 0xFE]
|
||||
.into_iter()
|
||||
.chain("héllo".encode_utf16().flat_map(u16::to_le_bytes))
|
||||
.collect();
|
||||
assert_eq!(
|
||||
text_of(extract(
|
||||
"application/octet-stream",
|
||||
Some("a.csv"),
|
||||
&utf16,
|
||||
&limits
|
||||
)),
|
||||
"héllo"
|
||||
);
|
||||
let html = "<html><style>p{}</style><p>Card 4242</p><script>x()</script><td>a</td><td>b</td></html>";
|
||||
let text = text_of(extract("text/html", None, html.as_bytes(), &limits));
|
||||
assert!(
|
||||
text.contains("Card 4242") && !text.contains("x()") && !text.contains("p{}"),
|
||||
"{text:?}"
|
||||
);
|
||||
assert_eq!(
|
||||
extract("image/png", Some("a.png"), b"\x89PNG....", &limits),
|
||||
Extracted::NoText
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn docx_joins_split_runs() {
|
||||
let doc = r#"<w:document xmlns:w="w"><w:body><w:p><w:r><w:t>Card 4242 42</w:t></w:r><w:r><w:t>42 4242 4242</w:t></w:r></w:p><w:p><w:r><w:t>A & B</w:t></w:r></w:p></w:body></w:document>"#;
|
||||
let docx = zip_of(&[
|
||||
("[Content_Types].xml", "<Types/>"),
|
||||
("word/document.xml", doc),
|
||||
]);
|
||||
let text = text_of(extract(
|
||||
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
||||
Some("a.docx"),
|
||||
&docx,
|
||||
&Limits::default(),
|
||||
));
|
||||
assert!(text.contains("Card 4242 4242 4242 4242\nA & B"), "{text:?}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn xlsx_numbers_and_shared_strings() {
|
||||
let sheet = r#"<worksheet><sheetData><row><c r="A1" t="s"><v>0</v></c><c r="B1"><v>4242424242424242</v></c></row></sheetData></worksheet>"#;
|
||||
let shared = r#"<sst><si><t>IBAN GB29 NWBK 6016 1331 9268 19</t></si></sst>"#;
|
||||
let xlsx = zip_of(&[
|
||||
("[Content_Types].xml", "<Types/>"),
|
||||
("xl/sharedStrings.xml", shared),
|
||||
("xl/worksheets/sheet1.xml", sheet),
|
||||
]);
|
||||
let text = text_of(extract("", Some("book.xlsx"), &xlsx, &Limits::default()));
|
||||
assert!(
|
||||
text.contains("4242424242424242") && text.contains("GB29 NWBK 6016 1331 9268 19"),
|
||||
"{text:?}"
|
||||
);
|
||||
// The shared string's index isn't read as a value
|
||||
assert!(
|
||||
!text.contains("\t0\t") && !text.starts_with('0'),
|
||||
"{text:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn opendocument_and_encrypted_opendocument() {
|
||||
let content = r#"<office:document-content xmlns:text="t"><text:p>SSN 078-05-1120</text:p></office:document-content>"#;
|
||||
let odt = zip_of(&[
|
||||
("mimetype", "application/vnd.oasis.opendocument.text"),
|
||||
("content.xml", content),
|
||||
]);
|
||||
assert!(
|
||||
text_of(extract("", Some("a.odt"), &odt, &Limits::default()))
|
||||
.contains("SSN 078-05-1120")
|
||||
);
|
||||
let manifest = r#"<manifest:manifest><manifest:file-entry><manifest:encryption-data/></manifest:file-entry></manifest:manifest>"#;
|
||||
let locked = zip_of(&[
|
||||
("mimetype", "application/vnd.oasis.opendocument.text"),
|
||||
("META-INF/manifest.xml", manifest),
|
||||
("content.xml", "x"),
|
||||
]);
|
||||
assert_eq!(
|
||||
extract("", Some("a.odt"), &locked, &Limits::default()),
|
||||
Extracted::NotInspectable(Why::Encrypted)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn archives() {
|
||||
let limits = Limits::default();
|
||||
let archive = zip_of(&[
|
||||
("notes/a.txt", "card 4242424242424242"),
|
||||
("b.png", "\u{89}PNG"),
|
||||
]);
|
||||
assert!(
|
||||
text_of(extract("application/zip", Some("x.zip"), &archive, &limits))
|
||||
.contains("4242424242424242")
|
||||
);
|
||||
let nested = zip_of(&[(
|
||||
"inner.zip",
|
||||
std::str::from_utf8(&[b'P', b'K', 3, 4]).unwrap(),
|
||||
)]);
|
||||
assert_eq!(
|
||||
extract("application/zip", Some("x.zip"), &nested, &limits),
|
||||
Extracted::NotInspectable(Why::NestedArchive)
|
||||
);
|
||||
|
||||
// Password-protected
|
||||
let mut zip = ZipWriter::new(Cursor::new(Vec::new()));
|
||||
zip.start_file(
|
||||
"secret.txt",
|
||||
SimpleFileOptions::default().with_aes_encryption(zip::AesMode::Aes256, "pw"),
|
||||
)
|
||||
.unwrap();
|
||||
zip.write_all(b"4242424242424242").unwrap();
|
||||
let locked = zip.finish().unwrap().into_inner();
|
||||
assert_eq!(
|
||||
extract("application/zip", Some("x.zip"), &locked, &limits),
|
||||
Extracted::NotInspectable(Why::Encrypted)
|
||||
);
|
||||
|
||||
// Past the limits
|
||||
let small = Limits {
|
||||
max_unpacked: 10,
|
||||
max_entries: 1,
|
||||
};
|
||||
assert_eq!(
|
||||
extract("application/zip", Some("x.zip"), &archive, &small),
|
||||
Extracted::NotInspectable(Why::TooLarge)
|
||||
);
|
||||
assert_eq!(
|
||||
extract("application/zip", None, b"PK\x03\x04garbage", &limits),
|
||||
Extracted::NotInspectable(Why::Damaged)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn not_inspectable_kinds() {
|
||||
let limits = Limits::default();
|
||||
assert_eq!(
|
||||
extract("application/octet-stream", None, b"%PDF-1.7 ...", &limits),
|
||||
Extracted::NotInspectable(Why::Pdf)
|
||||
);
|
||||
let mut ole = OLE_MAGIC.to_vec();
|
||||
ole.extend(std::iter::repeat_n(0, 64));
|
||||
assert_eq!(
|
||||
extract("", Some("old.doc"), &ole, &limits),
|
||||
Extracted::NotInspectable(Why::LegacyOffice)
|
||||
);
|
||||
ole.extend("EncryptedPackage".encode_utf16().flat_map(u16::to_le_bytes));
|
||||
assert_eq!(
|
||||
extract("", Some("new.docx"), &ole, &limits),
|
||||
Extracted::NotInspectable(Why::Encrypted)
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,23 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! Data loss prevention and mail flow rules (dlp-and-mail-flow-rules spec).
|
||||
//!
|
||||
//! Pure functions over text and attachment bytes, so everything here is
|
||||
//! unit-tested without a server:
|
||||
//!
|
||||
//! - [`detectors`]: find identifiers in text (payment cards, IBANs,
|
||||
//! national ID numbers, keys), each by its published format and check
|
||||
//! (§2.3);
|
||||
//! - [`words`]: an organization's own word lists and patterns;
|
||||
//! - [`extract`]: the text of an attachment, or why it can't be read.
|
||||
//!
|
||||
//! Nothing here writes what it finds anywhere: callers get counts, and the
|
||||
//! matched text never leaves the evaluation (§2.7).
|
||||
|
||||
pub mod detectors;
|
||||
pub mod extract;
|
||||
pub mod words;
|
||||
@@ -0,0 +1,109 @@
|
||||
/*
|
||||
* SPDX-FileCopyrightText: 2026 Coffey Labs
|
||||
*
|
||||
* SPDX-License-Identifier: AGPL-3.0-only
|
||||
*/
|
||||
|
||||
//! An organization's own word lists and patterns (§2.3). Both count
|
||||
//! occurrences, not distinct values: "confidential" three times is three.
|
||||
|
||||
use aho_corasick::{AhoCorasick, AhoCorasickBuilder, MatchKind};
|
||||
use regex::{Regex, RegexBuilder};
|
||||
|
||||
/// How large a compiled pattern may grow. Keeps a rule someone writes from
|
||||
/// making every message slow to send.
|
||||
const PATTERN_SIZE_LIMIT: usize = 1 << 20;
|
||||
|
||||
/// Words and phrases, matched whole and ignoring case.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct WordList {
|
||||
matcher: AhoCorasick,
|
||||
}
|
||||
|
||||
impl WordList {
|
||||
/// Builds a list from words or phrases; empty entries are skipped.
|
||||
pub fn new<I, S>(words: I) -> Result<Self, String>
|
||||
where
|
||||
I: IntoIterator<Item = S>,
|
||||
S: AsRef<str>,
|
||||
{
|
||||
let words: Vec<String> = words
|
||||
.into_iter()
|
||||
.map(|w| w.as_ref().trim().to_lowercase())
|
||||
.filter(|w| !w.is_empty())
|
||||
.collect();
|
||||
if words.is_empty() {
|
||||
return Err("The list has no words".into());
|
||||
}
|
||||
AhoCorasickBuilder::new()
|
||||
.match_kind(MatchKind::LeftmostLongest)
|
||||
.build(&words)
|
||||
.map(|matcher| Self { matcher })
|
||||
.map_err(|err| err.to_string())
|
||||
}
|
||||
|
||||
/// How many times any word of the list appears in `text`.
|
||||
pub fn count(&self, text: &str) -> usize {
|
||||
let text = text.to_lowercase();
|
||||
self.matcher
|
||||
.find_iter(&text)
|
||||
.filter(|m| super::detectors::stands_alone(&text, m.start(), m.end()))
|
||||
.count()
|
||||
}
|
||||
}
|
||||
|
||||
/// An organization's regular expression.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct Pattern {
|
||||
regex: Regex,
|
||||
}
|
||||
|
||||
impl Pattern {
|
||||
/// Compiles `pattern`, or says why it can't be used. Matching ignores
|
||||
/// case unless the pattern turns that off with `(?-i)`.
|
||||
pub fn new(pattern: &str) -> Result<Self, String> {
|
||||
RegexBuilder::new(pattern)
|
||||
.case_insensitive(true)
|
||||
.size_limit(PATTERN_SIZE_LIMIT)
|
||||
.build()
|
||||
.map(|regex| Self { regex })
|
||||
.map_err(|err| err.to_string())
|
||||
}
|
||||
|
||||
/// How many times the pattern matches in `text`.
|
||||
pub fn count(&self, text: &str) -> usize {
|
||||
self.regex.find_iter(text).filter(|m| !m.is_empty()).count()
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn words_whole_and_any_case() {
|
||||
let list = WordList::new(["Project Falcon", "confidential", " "]).unwrap();
|
||||
assert_eq!(
|
||||
list.count(
|
||||
"CONFIDENTIAL: project falcon notes. Not confidentiality, not projectfalcon."
|
||||
),
|
||||
2
|
||||
);
|
||||
assert_eq!(list.count("Confidential, confidential and confidential"), 3);
|
||||
// Non-ASCII case folding
|
||||
let list = WordList::new(["GEHEIM", "Straße"]).unwrap();
|
||||
assert_eq!(list.count("streng geheim, STRASSE ist nicht Straße"), 2);
|
||||
assert!(WordList::new(["", " "]).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn patterns() {
|
||||
let pattern = Pattern::new(r"\bPRJ-\d{4}\b").unwrap();
|
||||
assert_eq!(pattern.count("prj-1234 and PRJ-5678, not PRJ-12"), 2);
|
||||
assert!(Pattern::new("(unclosed").is_err());
|
||||
// Too large to compile within the limit
|
||||
assert!(Pattern::new(r"\w{1000}\w{1000}\w{1000}").is_err());
|
||||
// Empty matches don't count
|
||||
assert_eq!(Pattern::new("x*").unwrap().count("abc"), 0);
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user