PostgreSQL search: find words inside URLs and file names in body text #52
@@ -51,7 +51,9 @@ impl PostgresStore {
|
|||||||
let all_fields = index.all_fields();
|
let all_fields = index.all_fields();
|
||||||
let fields = document.fields;
|
let fields = document.fields;
|
||||||
// inbuxa: keyword text (addresses, contact fields, ...) is split into
|
// inbuxa: keyword text (addresses, contact fields, ...) is split into
|
||||||
// words before it reaches the text parser, see keyword_terms().
|
// words before it reaches the text parser, see keyword_terms();
|
||||||
|
// language text gets the words inside its URLs, host names and
|
||||||
|
// file names added, see url_terms().
|
||||||
let keywords = primary_keys
|
let keywords = primary_keys
|
||||||
.iter()
|
.iter()
|
||||||
.chain(all_fields)
|
.chain(all_fields)
|
||||||
@@ -60,6 +62,9 @@ impl PostgresStore {
|
|||||||
value,
|
value,
|
||||||
language: Language::None,
|
language: Language::None,
|
||||||
}) if field.is_text() => Some(keyword_terms(value)),
|
}) if field.is_text() => Some(keyword_terms(value)),
|
||||||
|
Some(SearchValue::Text { value, .. }) if field.is_text() => {
|
||||||
|
url_terms(value)
|
||||||
|
}
|
||||||
_ => None,
|
_ => None,
|
||||||
})
|
})
|
||||||
.collect::<Vec<_>>();
|
.collect::<Vec<_>>();
|
||||||
@@ -290,14 +295,36 @@ impl PostgresStore {
|
|||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
|
// inbuxa: a query word written as a URL, host,
|
||||||
|
// file or hyphenated word also matches as its word
|
||||||
|
// parts, which url_terms() indexes
|
||||||
|
let parts = match value {
|
||||||
|
SearchValue::Text { value, .. } => query_url_terms(value),
|
||||||
|
_ => None,
|
||||||
|
};
|
||||||
|
let parts_pos = value_pos + 1;
|
||||||
let _ = write!(query, "@@ ({method}('{config}', ${value_pos})");
|
let _ = write!(query, "@@ ({method}('{config}', ${value_pos})");
|
||||||
|
if parts.is_some() {
|
||||||
|
let _ = write!(query, " || {method}('{config}', ${parts_pos})");
|
||||||
|
}
|
||||||
for fallback in [PG_FALLBACK_LANG, PG_UNSTEMMED_LANG] {
|
for fallback in [PG_FALLBACK_LANG, PG_UNSTEMMED_LANG] {
|
||||||
if fallback != config && self.ts_configs.contains(fallback) {
|
if fallback != config && self.ts_configs.contains(fallback) {
|
||||||
let _ =
|
let _ =
|
||||||
write!(query, " || {method}('{fallback}', ${value_pos})");
|
write!(query, " || {method}('{fallback}', ${value_pos})");
|
||||||
|
if parts.is_some() {
|
||||||
|
let _ = write!(
|
||||||
|
query,
|
||||||
|
" || {method}('{fallback}', ${parts_pos})"
|
||||||
|
);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
query.push(')');
|
query.push(')');
|
||||||
|
values.push(SqlParam::Ref(value));
|
||||||
|
if let Some(parts) = parts {
|
||||||
|
values.push(SqlParam::Owned(parts));
|
||||||
|
}
|
||||||
|
continue;
|
||||||
}
|
}
|
||||||
values.push(SqlParam::Ref(value));
|
values.push(SqlParam::Ref(value));
|
||||||
} else if let SearchValue::KeyValues(kv) = value {
|
} else if let SearchValue::KeyValues(kv) = value {
|
||||||
@@ -391,6 +418,75 @@ pub(crate) fn keyword_terms(value: &str) -> String {
|
|||||||
terms
|
terms
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// inbuxa: in language text (subject, body, attachments) PostgreSQL's parser
|
||||||
|
// keeps a URL, a host name, a path or a file name as tokens of its own:
|
||||||
|
// "https://x.example/shipping-support/" gives a url, a host and a url_path,
|
||||||
|
// "invoice-2024.pdf" a file, so a body search for "shipping" or "invoice"
|
||||||
|
// missed messages where the word appears only there, while the built-in index
|
||||||
|
// splits them into words. The text is indexed as it was, followed by the word
|
||||||
|
// parts of each such token (SpaceTokenizer, as keyword_terms() splits), so
|
||||||
|
// they go through the same configuration and stemming as the words around
|
||||||
|
// them. On sample mail the text vector grows by about 15% for a newsletter
|
||||||
|
// full of tracking links and 30% for a short order notice with three links.
|
||||||
|
// Plain words, and words that only carry punctuation ("end.", "(see"),
|
||||||
|
// add nothing; hyphenated words are already split by the parser. Returns None
|
||||||
|
// when there is nothing to add, so most text is indexed exactly as before.
|
||||||
|
/// Characters that join the parts of a URL, host, path, address or file name.
|
||||||
|
const URL_SEPARATORS: [char; 13] = [
|
||||||
|
'/', '.', '@', ':', '?', '=', '&', '#', '_', '%', '+', '~', '\\',
|
||||||
|
];
|
||||||
|
|
||||||
|
pub(crate) fn url_terms(value: &str) -> Option<String> {
|
||||||
|
let mut terms = String::new();
|
||||||
|
// Each word is added once: a phrase search still finds the first URL it
|
||||||
|
// is in, and a newsletter's hundred tracking links don't add a hundred
|
||||||
|
// positions for "utm" and "campaign"
|
||||||
|
let mut seen = std::collections::HashSet::new();
|
||||||
|
for token in value.split(|c: char| {
|
||||||
|
c.is_whitespace() || matches!(c, '<' | '>' | '"' | '(' | ')' | '[' | ']' | '{' | '}')
|
||||||
|
}) {
|
||||||
|
let token = token.trim_matches(|c: char| !c.is_alphanumeric());
|
||||||
|
if token.contains(URL_SEPARATORS) {
|
||||||
|
for word in SpaceTokenizer::new(token, MAX_TOKEN_LENGTH) {
|
||||||
|
if !seen.insert(word.clone()) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if terms.is_empty() {
|
||||||
|
terms.reserve(value.len() + 64);
|
||||||
|
terms.push_str(value);
|
||||||
|
terms.push('\n');
|
||||||
|
} else {
|
||||||
|
terms.push(' ');
|
||||||
|
}
|
||||||
|
terms.push_str(&word);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
(!terms.is_empty()).then_some(terms)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The query side of url_terms(): each query word that is a URL, host, file
|
||||||
|
/// name or hyphenated word replaced by its word parts, or None when there is
|
||||||
|
/// none. It is searched in addition to the query as written, so documents
|
||||||
|
/// indexed before url_terms() still match as they did.
|
||||||
|
pub(crate) fn query_url_terms(value: &str) -> Option<String> {
|
||||||
|
let mut terms = String::with_capacity(value.len());
|
||||||
|
let mut changed = false;
|
||||||
|
for token in value.split_whitespace() {
|
||||||
|
let word = token.trim_matches(|c: char| !c.is_alphanumeric());
|
||||||
|
if !terms.is_empty() {
|
||||||
|
terms.push(' ');
|
||||||
|
}
|
||||||
|
if word.contains(URL_SEPARATORS) || word.contains('-') {
|
||||||
|
changed = true;
|
||||||
|
terms.push_str(&keyword_terms(word));
|
||||||
|
} else {
|
||||||
|
terms.push_str(token);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
changed.then_some(terms)
|
||||||
|
}
|
||||||
|
|
||||||
pub(super) enum SqlParam<'x> {
|
pub(super) enum SqlParam<'x> {
|
||||||
Ref(&'x (dyn ToSql + Sync)),
|
Ref(&'x (dyn ToSql + Sync)),
|
||||||
Owned(String),
|
Owned(String),
|
||||||
|
|||||||
@@ -133,6 +133,11 @@ pub async fn test(test: &TestServer) {
|
|||||||
println!("Running address search tests...");
|
println!("Running address search tests...");
|
||||||
test_address_search(store.clone()).await;
|
test_address_search(store.clone()).await;
|
||||||
|
|
||||||
|
// inbuxa: words inside URLs, host names and file names in body text
|
||||||
|
// are found on every backend
|
||||||
|
println!("Running URL word search tests...");
|
||||||
|
test_url_word_search(store.clone()).await;
|
||||||
|
|
||||||
// Large document insert test
|
// Large document insert test
|
||||||
println!("Running large document insert tests...");
|
println!("Running large document insert tests...");
|
||||||
let mut large_text = String::with_capacity(20 * 1024 * 1024);
|
let mut large_text = String::with_capacity(20 * 1024 * 1024);
|
||||||
@@ -1129,3 +1134,79 @@ async fn test_address_search(store: SearchStore) {
|
|||||||
.await
|
.await
|
||||||
.unwrap();
|
.unwrap();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
async fn test_url_word_search(store: SearchStore) {
|
||||||
|
const ACCOUNT_ID: u32 = 8;
|
||||||
|
let bodies = [
|
||||||
|
"Track your parcel here: https://x.example/shipping-support/ and reply.",
|
||||||
|
"Reset it at https://mail.example.com/login/?password=reset&user=jane now.",
|
||||||
|
"Attached is invoice-2024.pdf for your records.",
|
||||||
|
"Shipping was fast, thanks again.",
|
||||||
|
"Nothing to see at www.example.org/about-us, really.",
|
||||||
|
];
|
||||||
|
|
||||||
|
let mut documents = Vec::new();
|
||||||
|
let mut mask = RoaringBitmap::new();
|
||||||
|
for (document_id, body) in bodies.iter().enumerate() {
|
||||||
|
let mut document = IndexDocument::new(SearchIndex::Email)
|
||||||
|
.with_account_id(ACCOUNT_ID)
|
||||||
|
.with_document_id(document_id as u32);
|
||||||
|
document.index_text(EmailSearchField::Body, body, Language::English);
|
||||||
|
document.index_unsigned(EmailSearchField::ReceivedAt, document_id as u64);
|
||||||
|
documents.push(document);
|
||||||
|
mask.insert(document_id as u32);
|
||||||
|
}
|
||||||
|
store.index(documents).await.unwrap();
|
||||||
|
if let SearchStore::ElasticSearch(store) = &store {
|
||||||
|
store.refresh_index(SearchIndex::Email).await.unwrap();
|
||||||
|
}
|
||||||
|
|
||||||
|
for (text, expected) in [
|
||||||
|
// only inside a URL path, a query string or a file name
|
||||||
|
("shipping", vec![0u32, 3]),
|
||||||
|
("support", vec![0]),
|
||||||
|
("password", vec![1]),
|
||||||
|
("login", vec![1]),
|
||||||
|
("jane", vec![1]),
|
||||||
|
("invoice", vec![2]),
|
||||||
|
("pdf", vec![2]),
|
||||||
|
("2024", vec![2]),
|
||||||
|
// host names
|
||||||
|
("example", vec![0, 1, 4]),
|
||||||
|
("mail", vec![1]),
|
||||||
|
// written as they appear
|
||||||
|
("https://x.example/shipping-support/", vec![0]),
|
||||||
|
("shipping-support", vec![0]),
|
||||||
|
("invoice-2024.pdf", vec![2]),
|
||||||
|
("mail.example.com", vec![1]),
|
||||||
|
// plain words are unaffected
|
||||||
|
("parcel", vec![0]),
|
||||||
|
("records", vec![2]),
|
||||||
|
("thanks", vec![3]),
|
||||||
|
// no match
|
||||||
|
("billing", vec![]),
|
||||||
|
("example.net", vec![]),
|
||||||
|
] {
|
||||||
|
let ids = store
|
||||||
|
.query_account(
|
||||||
|
SearchQuery::new(SearchIndex::Email)
|
||||||
|
.with_filters(vec![
|
||||||
|
SearchFilter::eq(SearchField::AccountId, ACCOUNT_ID),
|
||||||
|
SearchFilter::has_english_text(EmailSearchField::Body, text),
|
||||||
|
])
|
||||||
|
.with_comparator(SearchComparator::ascending(EmailSearchField::ReceivedAt))
|
||||||
|
.with_mask(mask.clone()),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
assert_eq!(ids, expected, "Body {text:?}");
|
||||||
|
}
|
||||||
|
|
||||||
|
store
|
||||||
|
.unindex(
|
||||||
|
SearchQuery::new(SearchIndex::Email)
|
||||||
|
.with_filter(SearchFilter::eq(SearchField::AccountId, ACCOUNT_ID)),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user