Skip to content
This repository was archived by the owner on May 13, 2026. It is now read-only.
Merged
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
129 changes: 119 additions & 10 deletions src/db_operations/native_index/anonymity.rs
Original file line number Diff line number Diff line change
Expand Up @@ -24,7 +24,7 @@ pub enum FragmentDecision {
}

/// Minimum Shannon entropy (bits per character) for a fragment to be publishable.
const MIN_ENTROPY_BITS: f64 = 2.0;
const MIN_ENTROPY_BITS: f64 = 1.5;
/// Minimum word count for a fragment to be publishable.
const MIN_WORD_COUNT: usize = 3;

Expand Down Expand Up @@ -120,7 +120,7 @@ pub fn default_privacy_class(field_name: &str) -> FieldPrivacyClass {
/// Check whether text contains named entities (PII patterns).
/// Returns true if any PII pattern is detected.
pub fn contains_named_entities(text: &str) -> bool {
has_email(text) || has_phone(text) || has_url(text) || has_id_pattern(text)
has_email(text) || has_phone(text) || has_url(text) || has_id_pattern(text) || has_address(text)
}

fn has_email(text: &str) -> bool {
Expand All @@ -139,7 +139,13 @@ fn has_phone(text: &str) -> bool {
// Check for phone-like patterns: sequences of digits with separators
let mut consecutive_phone_chars = 0;
for ch in text.chars() {
if ch.is_ascii_digit() || ch == '-' || ch == '(' || ch == ')' || ch == ' ' || ch == '+'
if ch.is_ascii_digit()
|| ch == '-'
|| ch == '('
|| ch == ')'
|| ch == ' '
|| ch == '+'
|| ch == '.'
{
consecutive_phone_chars += 1;
} else {
Expand Down Expand Up @@ -174,17 +180,86 @@ fn has_id_pattern(text: &str) -> bool {
false
}

/// Calculate Shannon entropy (bits per character) of text.
fn has_address(text: &str) -> bool {
// Look for patterns like "123 Main St" or "456 Oak Avenue"
let street_suffixes = [
" st",
" st.",
" street",
" ave",
" ave.",
" avenue",
" blvd",
" blvd.",
" boulevard",
" dr",
" dr.",
" drive",
" rd",
" rd.",
" road",
" ln",
" ln.",
" lane",
" ct",
" ct.",
" court",
" pl",
" pl.",
" place",
" way",
" cir",
" circle",
" pkwy",
" parkway",
];
let lower = text.to_lowercase();
// Check every occurrence of each street suffix for a preceding digit
for suffix in &street_suffixes {
let mut search_from = 0;
loop {
// Ensure search_from is on a char boundary
if search_from >= lower.len() || !lower.is_char_boundary(search_from) {
break;
}
let Some(rel) = lower[search_from..].find(suffix) else {
break;
};
let pos = search_from + rel;
// Walk backwards up to ~30 chars, staying on a char boundary
let start = lower[..pos]
.char_indices()
.rev()
.nth(30)
.map_or(0, |(i, _)| i);
let preceding = &lower[start..pos];
if preceding.chars().any(|c| c.is_ascii_digit()) {
return true;
}
let next = pos + suffix.len();
// Advance to the next char boundary after the suffix
search_from = lower[next..]
.char_indices()
.next()
.map_or(lower.len(), |(i, _)| next + i);
}
}
false
}

/// Calculate Shannon entropy (bits per token) of text.
pub fn token_entropy(text: &str) -> f64 {
if text.is_empty() {
let tokens: Vec<&str> = text.split_whitespace().collect();
if tokens.is_empty() {
return 0.0;
}

let mut freq = std::collections::HashMap::new();
let total = text.len() as f64;
let total = tokens.len() as f64;

for byte in text.bytes() {
*freq.entry(byte).or_insert(0u64) += 1;
for token in &tokens {
let lower = token.to_lowercase();
*freq.entry(lower).or_insert(0u64) += 1;
}

freq.values()
Expand Down Expand Up @@ -336,8 +411,8 @@ mod tests {

#[test]
fn test_entropy_too_low() {
// Very short, repetitive text has low entropy
let entropy = token_entropy("aaa");
// Repetitive tokens have low entropy
let entropy = token_entropy("the the the");
assert!(entropy < MIN_ENTROPY_BITS, "entropy was {}", entropy);
}

Expand Down Expand Up @@ -422,4 +497,38 @@ mod tests {
FragmentDecision::Reject("field name suggests PII")
);
}

#[test]
fn test_entropy_token_level_unicode() {
// Unicode text should get same entropy as ASCII with same token diversity
let ascii_entropy = token_entropy("hello world foo bar baz");
let unicode_entropy = token_entropy("café résumé naïve über straße");
// Both have 5 unique tokens, so entropy should be similar
assert!(
(ascii_entropy - unicode_entropy).abs() < 0.1,
"ascii={}, unicode={} — should be similar for same token count",
ascii_entropy,
unicode_entropy
);
}

#[test]
fn test_ner_detects_address() {
assert!(contains_named_entities("lives at 123 Main St in town"));
assert!(contains_named_entities("office is 456 Oak Avenue"));
assert!(contains_named_entities("send to 789 Elm Blvd."));
assert!(!contains_named_entities("no address information here"));
// Second occurrence has the digit (first "Main St" has no number)
assert!(contains_named_entities(
"Main St is nice but 123 Elm St is better"
));
// Unicode preceding the suffix must not panic
assert!(contains_named_entities("café résumé 42 Oak Dr in town"));
assert!(!contains_named_entities("café résumé naïve über straße"));
}

#[test]
fn test_ner_detects_phone_with_dots() {
assert!(contains_named_entities("call 555.123.4567 for info"));
}
}
Loading