From 27183385d6ebd2584a881b0c7791924cc5805fa9 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Wed, 30 Sep 2026 20:05:56 +0300 Subject: [PATCH 1/3] refactor(safety): delegate store::safety to the shared tinymemory-safety crate The scrubbers moved to tinymemory-safety (tinymemory w4-memory-rest), which the OpenHuman host and tinymemory-core share. store::safety keeps its public paths and pins the engine policy (corroborated bare-card gate, unchanged behaviour). Co-authored-by: Medulla --- Cargo.lock | 11 + Cargo.toml | 8 + src/memory/store/safety/mod.rs | 333 +--------- src/memory/store/safety/pii.rs | 571 ----------------- src/memory/store/safety/pii/checks.rs | 332 ---------- src/memory/store/safety/pii/checks_tests.rs | 167 ----- src/memory/store/safety/pii/normalize.rs | 79 --- src/memory/store/safety/pii/prefilter.rs | 238 ------- src/memory/store/safety/pii_tests.rs | 661 -------------------- src/memory/store/safety/safety_tests.rs | 123 +--- 10 files changed, 79 insertions(+), 2444 deletions(-) delete mode 100644 src/memory/store/safety/pii.rs delete mode 100644 src/memory/store/safety/pii/checks.rs delete mode 100644 src/memory/store/safety/pii/checks_tests.rs delete mode 100644 src/memory/store/safety/pii/normalize.rs delete mode 100644 src/memory/store/safety/pii/prefilter.rs delete mode 100644 src/memory/store/safety/pii_tests.rs diff --git a/Cargo.lock b/Cargo.lock index d0f73a35..859ec142 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1705,6 +1705,7 @@ dependencies = [ "tinycortex-api", "tinyinference-embeddings", "tinyinference-llm", + "tinymemory-safety", "tokio", "toml", "tracing", @@ -1782,6 +1783,16 @@ dependencies = [ "uuid", ] +[[package]] +name = "tinymemory-safety" +version = "0.1.0" +source = "git+https://github.com/tinyhumansai/tinymemory?rev=b49650e1eac6d55a41f641f4ec8a683a25df9518#b49650e1eac6d55a41f641f4ec8a683a25df9518" +dependencies = [ + "log", + "regex", + "serde_json", +] + [[package]] name = "tinystr" version = "0.8.3" diff --git a/Cargo.toml b/Cargo.toml index e64479b9..e1c16d8b 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -103,6 +103,14 @@ people = ["tokio"] contacts = ["people", "dep:objc2", "dep:objc2-foundation", "dep:objc2-contacts", "dep:block2"] [dependencies] +# The shared secret and PII scrubber (`memory::store::safety` is a thin wrapper +# that pins this engine's policy). By git rev for the same reason +# `tinymemory-api` is — neither crate is published — and a host that vendors +# tinymemory patches it to its checkout: +# +# [patch."https://github.com/tinyhumansai/tinymemory"] +# tinymemory-safety = { path = "crates/tinymemory-safety" } +tinymemory-safety = { git = "https://github.com/tinyhumansai/tinymemory", rev = "b49650e1eac6d55a41f641f4ec8a683a25df9518" } anyhow = "1" log = "0.4" futures = "0.3" diff --git a/src/memory/store/safety/mod.rs b/src/memory/store/safety/mod.rs index b1ea1b35..836a18b2 100644 --- a/src/memory/store/safety/mod.rs +++ b/src/memory/store/safety/mod.rs @@ -1,324 +1,53 @@ //! Secret-detection and redaction helpers for memory writes. //! -//! Ported from OpenHuman's `memory_store::safety`. Conservative by design — it -//! prefers false positives over leaking credentials into long-lived stores. +//! The scrubbers themselves — credential patterns, the sensitive-key +//! classifier, the JSON depth cap and the checksum-gated multilingual +//! national-ID PII module — live in the shared `tinymemory-safety` crate, which +//! the OpenHuman host and `tinymemory-core` use as well (it used to be copied +//! three times). This module keeps the engine's public `safety::*` paths and +//! pins the engine's one policy choice: a *bare* (separator-less) Luhn-valid +//! digit run is only redacted as a credit card when corroborated by a real +//! network IIN or a nearby card keyword, so 13-digit epoch-millisecond +//! timestamps in stored JSON envelopes are not corrupted (opencompany#1201). //! -//! The exhaustive multilingual national-ID PII module (`safety::pii`, ~1k lines -//! of checksum logic) is ported from OpenHuman and runs as part of -//! `sanitize_text`. The write-rejection boundary stays stricter than content -//! scrubbing: formatted national IDs are rejected, while phone/email-like text is -//! scrubbed from content without rejecting every write that mentions them. +//! The write-rejection boundary ([`has_likely_pii`]) stays stricter than +//! content scrubbing: formatted national IDs are rejected, while phone/email-like +//! text is scrubbed from content without rejecting every write that mentions +//! them. -use std::sync::LazyLock; - -use regex::Regex; use serde_json::Value; -/// Exhaustive checksum-gated multilingual national-ID PII module (ported from -/// OpenHuman). Content scrubbing runs from `sanitize_text`; the boundary -/// check is re-exported as [`has_likely_pii`]. -pub mod pii; - -pub use pii::{has_likely_email, has_likely_pii}; - -const REDACTED_SECRET: &str = "[REDACTED_SECRET]"; -const REDACTED_PRIVATE_KEY: &str = "[REDACTED_PRIVATE_KEY]"; -const MAX_JSON_SANITIZE_DEPTH: usize = 128; - -/// Tally of what a sanitization pass changed. -#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] -pub struct SanitizationReport { - /// Count of secret/token pattern matches rewritten in string text by the - /// text-pattern redaction pass. - pub text_redactions: usize, - /// Count of JSON object entries dropped wholesale because their key was - /// classified as sensitive by the key classifier. - pub key_redactions: usize, - /// Count of full private-key blocks replaced; these are - /// the most severe hits since the entire block is removed. - pub blocked_secret_hits: usize, - /// Count of nodes collapsed because JSON nesting reached - /// the JSON traversal depth cap; the subtree is replaced rather than walked. - pub depth_redactions: usize, - /// Count of personal-identifier matches replaced by the - /// lightweight PII screen. - pub pii_redactions: usize, -} - -impl SanitizationReport { - /// True when any field recorded a redaction. - pub fn changed(&self) -> bool { - self.text_redactions > 0 - || self.key_redactions > 0 - || self.blocked_secret_hits > 0 - || self.depth_redactions > 0 - || self.pii_redactions > 0 - } - - /// Sum two reports field-wise. - pub fn merge(self, rhs: Self) -> Self { - Self { - text_redactions: self.text_redactions + rhs.text_redactions, - key_redactions: self.key_redactions + rhs.key_redactions, - blocked_secret_hits: self.blocked_secret_hits + rhs.blocked_secret_hits, - depth_redactions: self.depth_redactions + rhs.depth_redactions, - pii_redactions: self.pii_redactions + rhs.pii_redactions, - } - } -} - -/// A sanitized value plus the [`SanitizationReport`] describing the changes. -#[derive(Debug, Clone)] -pub struct Sanitized { - /// The cleaned value with secrets and PII removed. - pub value: T, - /// Tally of what the sanitization pass changed to produce `value`. - pub report: SanitizationReport, -} - -static BLOCK_PATTERNS: LazyLock> = LazyLock::new(|| { - vec![ - Regex::new( - r"(?is)-----BEGIN(?: [A-Z]+)? PRIVATE KEY-----.*?-----END(?: [A-Z]+)? PRIVATE KEY-----", - ) - .expect("valid private key block"), - Regex::new(r"(?is)-----BEGIN OPENSSH PRIVATE KEY-----.*?-----END OPENSSH PRIVATE KEY-----") - .expect("valid openssh private key block"), - Regex::new( - r"(?is)-----BEGIN PGP PRIVATE KEY BLOCK-----.*?-----END PGP PRIVATE KEY BLOCK-----", - ) - .expect("valid pgp private key block"), - ] -}); +pub use tinymemory_safety::{ + has_likely_email, has_likely_pii, has_likely_secret, BareCardGate, Policy, SanitizationReport, + Sanitized, +}; -static REDACTION_PATTERNS: LazyLock> = LazyLock::new(|| { - vec![ - ( - Regex::new(r"(?i)(bearer\s+)[A-Za-z0-9._~+/=-]{8,}").expect("valid bearer redaction"), - "${1}[REDACTED]", - ), - ( - Regex::new(r#"(?i)(api[_-]?key\s*[=:\s]\s*["']?)[^\s"']+"#) - .expect("valid api key redaction"), - "${1}[REDACTED]", - ), - ( - Regex::new( - r#"(?i)\b(token|access[_-]?token|refresh[_-]?token|client[_-]?secret|password|secret)\b\s*[=:\s]\s*["']?[^\s"'&]+"#, - ) - .expect("valid token redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\bsk-[A-Za-z0-9]{20,}\b").expect("valid openai key redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\bgh[pousr]_[A-Za-z0-9_]{20,}\b").expect("valid github token redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\bAKIA[0-9A-Z]{16}\b").expect("valid aws key redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\bASIA[0-9A-Z]{16}\b").expect("valid aws sts key redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\beyJ[A-Za-z0-9_-]{8,}\.[A-Za-z0-9._-]{8,}\.[A-Za-z0-9._-]{8,}\b") - .expect("valid jwt redaction"), - "[REDACTED]", - ), - ( - Regex::new( - r#"(?i)\b(access_token|refresh_token|id_token|authorization_code|code_verifier|code_challenge)\b\s*[=:\s]\s*["']?[^\s"'&]+"#, - ) - .expect("valid oauth token redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\bAIza[0-9A-Za-z\-_]{35}\b").expect("valid google api key redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\bsk-ant-[A-Za-z0-9\-_]{16,}\b").expect("valid anthropic key redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\bsk-(?:proj|org)-[A-Za-z0-9\-_]{12,}\b") - .expect("valid openai scoped key redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\b(?:sk|rk)_(?:live|test)_[A-Za-z0-9]{16,}\b") - .expect("valid stripe key redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\bxox(?:a|b|p|s|r)-[A-Za-z0-9-]{10,}\b") - .expect("valid slack token redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\bgithub_pat_[A-Za-z0-9_]{20,}\b").expect("valid github pat redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\bglpat-[A-Za-z0-9\-_]{16,}\b").expect("valid gitlab pat redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\bnpm_[A-Za-z0-9]{20,}\b").expect("valid npm token redaction"), - "[REDACTED]", - ), - ( - Regex::new(r"\bSG\.[A-Za-z0-9_\-]{16,}\.[A-Za-z0-9_\-]{16,}\b") - .expect("valid sendgrid key redaction"), - "[REDACTED]", - ), - ] -}); - -/// True when `value` looks like it contains a credential. -pub fn has_likely_secret(value: &str) -> bool { - BLOCK_PATTERNS.iter().any(|p| p.is_match(value)) - || REDACTION_PATTERNS.iter().any(|(p, _)| p.is_match(value)) -} +/// The scrubbing policy every engine write uses. +const ENGINE_POLICY: Policy = Policy::corroborated(); /// Scrub secrets and PII from free text, returning the cleaned text plus a /// [`SanitizationReport`]. pub fn sanitize_text(value: &str) -> Sanitized { - let mut out = value.to_string(); - let mut report = SanitizationReport::default(); - - for pattern in BLOCK_PATTERNS.iter() { - let hits = pattern.find_iter(&out).count(); - if hits > 0 { - report.blocked_secret_hits += hits; - out = pattern.replace_all(&out, REDACTED_PRIVATE_KEY).into_owned(); - } - } - - for (pattern, replacement) in REDACTION_PATTERNS.iter() { - let hits = pattern.find_iter(&out).count(); - if hits > 0 { - report.text_redactions += hits; - out = pattern.replace_all(&out, *replacement).into_owned(); - } - } - - // Full multilingual national-ID PII scrub (checksum-gated, normalization - // pre-pass) — runs after secret redaction so every call site that scrubs - // secrets also scrubs PII. - let pii = pii::redact_pii(&out); - report = report.merge(pii.report); - out = pii.value; - - Sanitized { value: out, report } + tinymemory_safety::sanitize_text_with(value, ENGINE_POLICY) } /// Recursively scrub a JSON value: sensitive keys are replaced wholesale and -/// every string value runs through `sanitize_text`. +/// every string value runs through [`sanitize_text`]. pub fn sanitize_json(value: &Value) -> Sanitized { - sanitize_json_inner(value, 0) + tinymemory_safety::sanitize_json_with(value, ENGINE_POLICY) } -/// Recursive worker behind [`sanitize_json`]. -/// -/// `depth` counts nesting from the call in `sanitize_json` (which starts at -/// `0`); once it reaches [`MAX_JSON_SANITIZE_DEPTH`] the whole subtree at that -/// point is replaced by a single redaction marker rather than walked further, -/// bounding recursion against pathologically deep or adversarial JSON. -fn sanitize_json_inner(value: &Value, depth: usize) -> Sanitized { - if depth >= MAX_JSON_SANITIZE_DEPTH { - return Sanitized { - value: Value::String(REDACTED_SECRET.to_string()), - report: SanitizationReport { - depth_redactions: 1, - ..SanitizationReport::default() - }, - }; - } - - match value { - Value::Object(map) => { - let mut out = serde_json::Map::new(); - let mut report = SanitizationReport::default(); - for (key, value) in map { - if is_sensitive_key(key) { - report.key_redactions += 1; - out.insert(key.clone(), Value::String(REDACTED_SECRET.to_string())); - continue; - } - let sanitized = sanitize_json_inner(value, depth + 1); - report = report.merge(sanitized.report); - out.insert(key.clone(), sanitized.value); - } - Sanitized { - value: Value::Object(out), - report, - } - } - Value::Array(items) => { - let mut out = Vec::with_capacity(items.len()); - let mut report = SanitizationReport::default(); - for item in items { - let sanitized = sanitize_json_inner(item, depth + 1); - report = report.merge(sanitized.report); - out.push(sanitized.value); - } - Sanitized { - value: Value::Array(out), - report, - } - } - Value::String(value) => { - let sanitized = sanitize_text(value); - Sanitized { - value: Value::String(sanitized.value), - report: sanitized.report, - } - } - _ => Sanitized { - value: value.clone(), - report: SanitizationReport::default(), - }, - } -} +/// Personal-PII detection and redaction (national IDs, financial identifiers, +/// international phone), on-device, regex + checksum only. +pub mod pii { + use super::{Sanitized, ENGINE_POLICY}; -/// True when a JSON object key's name itself suggests it holds a secret -/// (`api_key`, `token`, `password`, …), independent of the value's contents. -/// -/// Matching keys are redacted wholesale in [`sanitize_json_inner`] — the -/// value is replaced rather than scanned, since a key named e.g. `password` -/// is assumed sensitive even if its value doesn't match any -/// [`REDACTION_PATTERNS`] regex. Matching is on the key with all -/// non-alphanumeric characters stripped and lowercased, so `API-Key`, -/// `api_key`, and `apiKey` are all treated identically. -fn is_sensitive_key(key: &str) -> bool { - let normalized: String = key - .chars() - .filter(|c| c.is_ascii_alphanumeric()) - .map(|c| c.to_ascii_lowercase()) - .collect(); + pub use tinymemory_safety::pii::{has_likely_email, has_likely_pii}; - matches!( - normalized.as_str(), - "apikey" - | "token" - | "accesstoken" - | "refreshtoken" - | "authorization" - | "password" - | "secret" - | "clientsecret" - ) || normalized.ends_with("token") - || normalized.ends_with("apikey") - || normalized.ends_with("clientsecret") - || normalized.contains("password") - || normalized.contains("secret") - || normalized.ends_with("key") + /// Redact format-based multilingual PII from `text` under the engine policy. + pub fn redact_pii(text: &str) -> Sanitized { + tinymemory_safety::pii::redact_pii_with(text, ENGINE_POLICY) + } } #[cfg(test)] diff --git a/src/memory/store/safety/pii.rs b/src/memory/store/safety/pii.rs deleted file mode 100644 index d8fbe8ce..00000000 --- a/src/memory/store/safety/pii.rs +++ /dev/null @@ -1,571 +0,0 @@ -//! Multilingual personal-PII redaction (national IDs, financial identifiers, -//! international phone) — on-device, regex + checksum only, zero network. -//! -//! ## Design — security first -//! -//! 1. **Checksum gating where possible.** CPF, CNPJ, CUIT, credit-card (Luhn), -//! IBAN (mod-97), Aadhaar (Verhoeff), Spanish DNI/NIE (check letter), and -//! US SSN reserved-range filters all reject look-alikes that aren't real -//! identifiers. The false-positive rate from format alone is too high; the -//! checksums bring it back to acceptable. -//! -//! 2. **Bypass-resistant.** Inputs are normalized before -//! matching, which: -//! - strips zero-width characters (U+200B/200C/200D/FEFF/2060/180E), -//! - folds fullwidth digits (`0-9` → `0-9`) and fullwidth `.-/:` -//! to their ASCII counterparts, -//! - folds Arabic-Indic and Eastern Arabic-Indic digits to ASCII. -//! Match offsets are mapped back to the original text so we only redact -//! the bytes that actually carry PII; surrounding text is untouched. -//! -//! 3. **Overlap-safe.** Patterns are run in priority order; later matches -//! that overlap an earlier redaction are dropped, so a credit-card span -//! can't also be partially matched as a phone number. -//! -//! 4. **Out of scope.** Contextual PII (`"call me at the usual number"`), -//! compound PII (`name + employer + city`), arbitrary names, and freeform -//! dates-of-birth all require NER/LLM and are NOT addressed here. This -//! module is honest about its scope. - -use regex::Regex; -use std::sync::LazyLock; - -use super::{SanitizationReport, Sanitized}; - -mod checks; -use checks::*; - -// ---------- Replacement tokens ---------- - -const PII_RFC: &str = "[REDACTED_PII_RFC]"; -const PII_CPF: &str = "[REDACTED_PII_CPF]"; -const PII_CNPJ: &str = "[REDACTED_PII_CNPJ]"; -const PII_CUIT: &str = "[REDACTED_PII_CUIT]"; -const PII_MYNUM: &str = "[REDACTED_PII_MYNUMBER]"; -const PII_PHONE: &str = "[REDACTED_PII_PHONE]"; -const PII_SSN: &str = "[REDACTED_PII_SSN]"; -const PII_CC: &str = "[REDACTED_PII_CREDIT_CARD]"; -const PII_IBAN: &str = "[REDACTED_PII_IBAN]"; -const PII_AADHAAR: &str = "[REDACTED_PII_AADHAAR]"; -const PII_PAN_IN: &str = "[REDACTED_PII_PAN_IN]"; -const PII_NINO: &str = "[REDACTED_PII_NINO]"; -const PII_DNI: &str = "[REDACTED_PII_DNI]"; -const PII_RRN: &str = "[REDACTED_PII_RRN]"; - -// ---------- Patterns ---------- - -// Brazilian CPF, formatted: NNN.NNN.NNN-NN -static CPF_FMT_RE: LazyLock = - LazyLock::new(|| Regex::new(r"\b\d{3}\.\d{3}\.\d{3}-\d{2}\b").expect("cpf fmt")); -// Brazilian CPF, bare: 11 consecutive digits. Checksum-gated; ~1% raw FP. -static CPF_BARE_RE: LazyLock = - LazyLock::new(|| Regex::new(r"\b\d{11}\b").expect("cpf bare")); - -// Brazilian CNPJ, formatted: NN.NNN.NNN/NNNN-NN -static CNPJ_FMT_RE: LazyLock = - LazyLock::new(|| Regex::new(r"\b\d{2}\.\d{3}\.\d{3}/\d{4}-\d{2}\b").expect("cnpj fmt")); -// Brazilian CNPJ, bare: 14 consecutive digits. -static CNPJ_BARE_RE: LazyLock = - LazyLock::new(|| Regex::new(r"\b\d{14}\b").expect("cnpj bare")); - -// Argentine CUIT/CUIL: NN-NNNNNNNN-N (formatted only — bare 11-digit with -// single check digit has ~9% FP on random IDs, too noisy without context). -static CUIT_RE: LazyLock = - LazyLock::new(|| Regex::new(r"\b\d{2}-\d{8}-\d\b").expect("cuit")); - -// Mexican RFC: 3-4 letters (incl. Ñ &) + 6 digits + 3 alphanumeric homoclave. -static RFC_RE: LazyLock = - LazyLock::new(|| Regex::new(r"(?i)\b[A-ZÑ&]{3,4}\d{6}[A-Z0-9]{3}\b").expect("rfc")); - -// Japan My Number (12 digits) gated by a Japanese or English keyword within -// ~30 chars. Bare 12-digit runs without keyword are too noisy. -static MYNUM_RE: LazyLock = LazyLock::new(|| { - Regex::new(r"(?:マイナンバー|個人番号|My\s?Number)[\s:はがを、.\-]{0,12}(\d{12})\b") - .expect("my number") -}); - -// E.164 phone: + followed by 7-15 digits, no separators. -static PHONE_E164_RE: LazyLock = - LazyLock::new(|| Regex::new(r"\+\d{7,15}\b").expect("e164")); - -// NANP (US/Canada) formatted phone. Area code must start 2-9; first digit of -// central-office code also 2-9 (real NANP rule). -static PHONE_NANP_RE: LazyLock = LazyLock::new(|| { - Regex::new(r"\b(?:\+?1[\s.\-]?)?\(?([2-9]\d{2})\)?[\s.\-]?([2-9]\d{2})[\s.\-]?(\d{4})\b") - .expect("nanp phone") -}); - -// US SSN: NNN-NN-NNNN. Range filter applied below. -static SSN_RE: LazyLock = - LazyLock::new(|| Regex::new(r"\b\d{3}-\d{2}-\d{4}\b").expect("ssn")); - -// Credit card: 13-19 digits with optional spaces/dashes every 4. Every match -// is Luhn-gated; a match with no separators at all additionally needs -// corroboration — a real network IIN at an issued length, or a card keyword -// nearby — because Luhn alone passes ~10% of arbitrary digit runs, and bare -// 13-digit epoch-millisecond timestamps were being redacted out of stored -// JSON envelopes at exactly that rate (opencompany#1201). Same split as -// Aadhaar below: formatted keeps the checksum-only gate, bare needs more. -static CC_RE: LazyLock = - LazyLock::new(|| Regex::new(r"\b(?:\d[\s\-]?){13,19}\b").expect("credit card")); - -// Card keyword corroborating a bare digit run. Three tiers, matched -// case-insensitively: -// -// * Standalone words, bounded by `[\W_]` rather than `\b` — the regex crate -// counts `_` as a word character, so `\bcard\b` never fires inside -// `card_number`, which is among the most common serialized key shapes a -// stored payload carries. The explicit class keeps `pan` from firing -// inside `japan` while still matching `card_number=` and `cc=`. -// * Compound identifiers matched as substrings, because camelCase provides -// no boundary of any kind: `cardNumber`, `creditCard`, `ccNum`, `cardNo`, -// `panNumber` all lowercase into these. -// * Native-script terms, per this module's multilingual mandate (the Aadhaar -// and My Number patterns already carry theirs). CJK terms match as -// substrings because CJK sentences provide no `[\W_]` boundaries around a -// word — the case this buys is a keyword embedded mid-sentence -// (`…信用卡账单 `). It does not buy `卡号` with the digits -// directly attached: there `CC_RE`'s own leading `\b` already fails -// (CJK is `\w`), so the run is never a candidate in the first place. -static CC_KEYWORD_RE: LazyLock = LazyLock::new(|| { - Regex::new( - r"(?i)(?:^|[\W_])(?:card|credit|debit|visa|mastercard|amex|american\s?express|discover|jcb|diners|unionpay|hipercard|rupay|cvv|cvc|cc|pan|tarjeta|cart[aã]o|carte|karte|карта|карты|карту|картой|карте|кредитка)(?:[\W_]|$)|(?i:cardnumber|creditcard|ccnum|cardno|pannumber|カード|信用卡|卡号|银行卡|카드)", - ) - .expect("cc keyword") -}); - -// IBAN: 2 letter country code + 2 check digits + 11-30 alphanumeric. -// Allow optional spaces every 4 chars (common human format). -static IBAN_RE: LazyLock = - LazyLock::new(|| Regex::new(r"\b[A-Z]{2}\d{2}(?:[\s]?[A-Z0-9]){11,30}\b").expect("iban")); - -// India Aadhaar: 4-4-4 digit groups (space or hyphen) OR contiguous 12 digits -// gated by keyword. Verhoeff-checksum-gated when grouped, keyword-gated when -// bare (Verhoeff alone has ~10% raw FP rate on random 12-digit runs). -static AADHAAR_FMT_RE: LazyLock = - LazyLock::new(|| Regex::new(r"\b\d{4}[\s\-]\d{4}[\s\-]\d{4}\b").expect("aadhaar formatted")); -static AADHAAR_KW_RE: LazyLock = LazyLock::new(|| { - Regex::new(r"(?i)(?:aadhaar|aadhar|आधार|uidai|uid)[\s:#\-no.]{0,10}(\d{12})\b") - .expect("aadhaar keyword") -}); - -// India PAN: 5 letters, 4 digits, 1 letter. Very high signal — no checksum. -static PAN_IN_RE: LazyLock = - LazyLock::new(|| Regex::new(r"(?i)\b[A-Z]{5}\d{4}[A-Z]\b").expect("pan-in")); - -// UK NINO: 2 letters + 6 digits + suffix A/B/C/D. -static NINO_RE: LazyLock = - LazyLock::new(|| Regex::new(r"(?i)\b[A-Z]{2}\d{6}[A-D]\b").expect("nino")); - -// Spain DNI: 8 digits + check letter. NIE: starts X/Y/Z, then 7 digits + letter. -static DNI_RE: LazyLock = LazyLock::new(|| Regex::new(r"(?i)\b\d{8}[A-Z]\b").expect("dni")); -static NIE_RE: LazyLock = - LazyLock::new(|| Regex::new(r"(?i)\b[XYZ]\d{7}[A-Z]\b").expect("nie")); - -// South Korea RRN: NNNNNN-CXXXXXX where C is gender/century digit (1-4). -static RRN_RE: LazyLock = - LazyLock::new(|| Regex::new(r"\b\d{6}-[1-4]\d{6}\b").expect("rrn")); -static EMAIL_RE: LazyLock = - LazyLock::new(|| Regex::new(r"(?i)\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}\b").expect("email")); - -// ---------- Byte-oriented candidate pre-filter ---------- -// -// The single cheap byte pass that replaces the always-resident combined -// `RegexSet`. Lives in its own module — see `prefilter.rs` for the full rationale. -mod prefilter; -use prefilter::{scan_candidates, Candidates}; - -// ---------- Public API ---------- - -/// Redact format-based multilingual PII from `text`. -/// -/// Runs a Unicode normalization pre-pass to defeat fullwidth-digit and -/// zero-width-char bypasses. Match indices from the normalized form are -/// translated back to original byte offsets so only the PII bytes are -/// replaced — surrounding text (including any preserved fullwidth glyphs) -/// is untouched. -pub fn redact_pii(text: &str) -> Sanitized { - let mut report = SanitizationReport::default(); - - // Fast path: cheap byte pre-filter on the raw text. Fullwidth / Arabic-Indic - // digits and folded punctuation only surface after normalization, so a clean - // raw scan still re-checks the normalized view before declaring the text PII- - // free (mirrors the old two-phase SCREEN check). - let raw_cand = scan_candidates(text); - if !raw_cand.any() { - let nview = NormalizedView::build(text); - let ncand = scan_candidates(&nview.normalized); - if !ncand.any() { - log::trace!( - "[pii] redact_pii: no candidate before or after normalization (len={})", - text.len() - ); - return Sanitized { - value: text.to_string(), - report, - }; - } - log::debug!("[pii] redact_pii: candidate surfaced only after normalization"); - return splice_redactions( - text, - &nview, - collect_redactions(&nview.normalized, &ncand), - &mut report, - ); - } - - let nview = NormalizedView::build(text); - // Gate on candidates from the NORMALIZED text — the precise regexes run - // against it, so normalization-induced classes (folded digits) are included. - let ncand = scan_candidates(&nview.normalized); - let redactions = collect_redactions(&nview.normalized, &ncand); - splice_redactions(text, &nview, redactions, &mut report) -} - -/// True if `value` looks like it carries any PII. Used to *reject* -/// namespace/key inputs at boundary checks (analogous to -/// [`super::has_likely_secret`]). -/// -/// Uses the **strict** match set — only formatted / keyword-gated patterns. -/// Bare-numeric patterns whose only signal is a digit run (credit card via -/// Luhn, bare CPF, bare CNPJ) or a phone-shaped digit run (NANP without -/// separators, E.164 leading `+`) are excluded here because their false- -/// positive rate against scanner-built namespace/key identifiers (WhatsApp -/// JIDs like `12025551234-1543890267@g.us`, telegram numeric peer IDs, -/// millisecond timestamps, padded counters) is too high to use as a hard -/// rejection signal. Content scrubbing via [`redact_pii`] still applies -/// those patterns — a content false positive replaces bytes inside a string -/// rather than rejecting the whole write, which is cheaper but *not* free: -/// a redaction landing inside structured content corrupts it for whatever -/// wrote it (opencompany#1201 — timestamps in stored JSON envelopes), which -/// is why the credit-card pattern's bare form now demands corroboration -/// beyond its checksum. -pub fn has_likely_pii(value: &str) -> bool { - let nview = NormalizedView::build(value); - let cand = scan_candidates(&nview.normalized); - if !cand.any() { - return false; - } - !collect_strict_redactions(&nview.normalized, &cand).is_empty() -} - -/// True when `value` contains an ordinary email address. Kept separate from -/// [`has_likely_pii`] because scanner-built identifiers may legitimately -/// contain email-like `@` segments. -pub fn has_likely_email(value: &str) -> bool { - // Cheap gate: every email requires an `@`. Skip compiling the regex when - // the byte is absent (the common namespace/key case). - if !value.as_bytes().contains(&b'@') { - return false; - } - EMAIL_RE.is_match(value) -} - -// ---------- Match collection ---------- - -#[derive(Debug)] -struct Hit { - start: usize, // byte offset in NORMALIZED text - end: usize, - token: &'static str, -} - -fn collect_redactions(norm: &str, cand: &Candidates) -> Vec { - collect_redactions_inner(norm, cand, true) -} - -/// Variant of [`collect_redactions`] that omits bare-numeric patterns -/// whose only signal is a digit-run shape: credit card via Luhn, bare -/// CPF, bare CNPJ, NANP phones (separators optional, so any 10-11 digit -/// run starting `[2-9]`/`1[2-9]` matches), and E.164 phones (literal `+` -/// the only signal). Used for boundary checks like [`has_likely_pii`] -/// where rejection on such a hit alone would have too many false -/// positives on scanner-built identifiers (WhatsApp group JIDs -/// `-@g.us`, timestamps, padded counters). -fn collect_strict_redactions(norm: &str, cand: &Candidates) -> Vec { - collect_redactions_inner(norm, cand, false) -} - -/// Run only the precise regexes whose class was flagged by [`scan_candidates`]. -/// Priority order (and therefore overlap-resolution) is byte-identical to the -/// unconditional version; the `if cand.*` guards only decide whether each class -/// runs, so a flagged class produces exactly the hits it always did. -fn collect_redactions_inner(norm: &str, cand: &Candidates, include_bare_numeric: bool) -> Vec { - let mut hits: Vec = Vec::new(); - - // Priority order: most specific / highest-confidence first. - if cand.cpf_fmt { - push_checksum(&mut hits, norm, &CPF_FMT_RE, PII_CPF, |s| { - valid_cpf(digits(s).as_slice()) - }); - } - if cand.cnpj_fmt { - push_checksum(&mut hits, norm, &CNPJ_FMT_RE, PII_CNPJ, |s| { - valid_cnpj(digits(s).as_slice()) - }); - } - if cand.cuit { - push_checksum(&mut hits, norm, &CUIT_RE, PII_CUIT, |s| { - valid_cuit(digits(s).as_slice()) - }); - } - - // IBAN before credit card: CC can match an IBAN tail of all digits. - if cand.iban { - push_checksum(&mut hits, norm, &IBAN_RE, PII_IBAN, valid_iban); - } - - if include_bare_numeric { - // Credit card before bare CPF/CNPJ to avoid catching a 13-19 digit run as CPF/CNPJ. - if cand.cc { - push_credit_cards(&mut hits, norm); - } - if cand.cnpj_bare { - push_checksum(&mut hits, norm, &CNPJ_BARE_RE, PII_CNPJ, |s| { - valid_cnpj(digits(s).as_slice()) - }); - } - if cand.cpf_bare { - push_checksum(&mut hits, norm, &CPF_BARE_RE, PII_CPF, |s| { - valid_cpf(digits(s).as_slice()) - }); - } - } - - if cand.aadhaar_fmt { - push_checksum(&mut hits, norm, &AADHAAR_FMT_RE, PII_AADHAAR, |s| { - valid_verhoeff(digits(s).as_slice()) - }); - } - // Keyword-gated Aadhaar redacts only the captured 12-digit group. - if cand.aadhaar_kw { - push_captured(&mut hits, norm, &AADHAAR_KW_RE, PII_AADHAAR, |digits_str| { - valid_verhoeff(digits(digits_str).as_slice()) - }); - } - - if cand.dni { - push_checksum(&mut hits, norm, &DNI_RE, PII_DNI, valid_dni_es); - } - if cand.nie { - push_checksum(&mut hits, norm, &NIE_RE, PII_DNI, valid_nie_es); - } - if cand.nino { - push_checksum(&mut hits, norm, &NINO_RE, PII_NINO, valid_nino); - } - if cand.ssn { - push_checksum(&mut hits, norm, &SSN_RE, PII_SSN, valid_ssn); - } - if cand.rrn { - push_simple(&mut hits, norm, &RRN_RE, PII_RRN); - } - if cand.rfc { - push_simple(&mut hits, norm, &RFC_RE, PII_RFC); - } - if cand.pan_in { - push_simple(&mut hits, norm, &PAN_IN_RE, PII_PAN_IN); - } - - if include_bare_numeric { - // Phones: E.164 first (more specific), then NANP. Both are bare-numeric - // shapes — NANP allows optional separators (`\b\d{10,11}\b` matches as - // `XXX-XXX-XXXX`), and E.164 keys on a literal `+` with no further gate. - // Strict callers (boundary checks like `has_likely_pii`) exclude these - // so scanner-built namespace/key values (WhatsApp JIDs - // `-@g.us`, telegram numeric peer IDs) don't get rejected. - if cand.phone_e164 { - push_simple(&mut hits, norm, &PHONE_E164_RE, PII_PHONE); - } - if cand.phone_nanp { - push_simple(&mut hits, norm, &PHONE_NANP_RE, PII_PHONE); - } - } - - // My Number — captured digit group only, keyword remains visible. - if cand.mynumber { - push_captured(&mut hits, norm, &MYNUM_RE, PII_MYNUM, |_| true); - } - - dedupe_overlaps(&mut hits); - log::debug!( - "[pii] collect_redactions strict={} hits={}", - !include_bare_numeric, - hits.len() - ); - hits -} - -fn push_simple(hits: &mut Vec, norm: &str, re: &Regex, token: &'static str) { - for m in re.find_iter(norm) { - hits.push(Hit { - start: m.start(), - end: m.end(), - token, - }); - } -} - -fn push_checksum( - hits: &mut Vec, - norm: &str, - re: &Regex, - token: &'static str, - ok: impl Fn(&str) -> bool, -) { - for m in re.find_iter(norm) { - if ok(m.as_str()) { - hits.push(Hit { - start: m.start(), - end: m.end(), - token, - }); - } - } -} - -fn push_captured( - hits: &mut Vec, - norm: &str, - re: &Regex, - token: &'static str, - ok: impl Fn(&str) -> bool, -) { - for caps in re.captures_iter(norm) { - let Some(group) = caps.get(1) else { continue }; - if ok(group.as_str()) { - hits.push(Hit { - start: group.start(), - end: group.end(), - token, - }); - } - } -} - -/// Credit-card collection. Every match must pass Luhn; a bare match (no -/// separators) additionally needs structural or contextual corroboration. -/// -/// Luhn alone passes ~10% of arbitrary digit runs, and 13-19 contiguous -/// digits is a common machine-identifier shape — 13-digit epoch-millisecond -/// timestamps above all, which were being redacted out of stored JSON -/// envelopes at that rate and corrupting them (opencompany#1201). The strict -/// boundary set ([`collect_strict_redactions`]) already excludes credit card -/// for exactly this reason; this brings the content path to the same -/// judgement without giving up real cards: a separated run keeps the -/// Luhn-only gate it always had, and a bare run still redacts when its -/// prefix is a real network IIN at an issued length -/// ([`plausible_card_number`]) or a card keyword sits within -/// [`CC_KEYWORD_WINDOW`] bytes. -fn push_credit_cards(hits: &mut Vec, norm: &str) { - for m in CC_RE.find_iter(norm) { - let s = m.as_str(); - if !valid_luhn(s) { - continue; - } - // Judge bare-vs-separated on the interior of the digit run: the - // regex's per-digit `[\s\-]?` can capture one trailing separator - // (`"…773 "`), which is not the human 4-4-4-4 grouping this - // distinction is after. - let interior = s.trim_matches(|c: char| !c.is_ascii_digit()); - let bare = interior.bytes().all(|b| b.is_ascii_digit()); - let corroborated = - !bare || plausible_card_number(&digits(s)) || cc_keyword_near(norm, m.start(), m.end()); - if corroborated { - hits.push(Hit { - start: m.start(), - end: m.end(), - token: PII_CC, - }); - } - } -} - -/// Bytes of context searched either side of a bare digit run for a card -/// keyword. 64 bytes rather than 32 because the window is counted in bytes -/// while text is not: 32 bytes is only ~10 CJK characters or 8 emoji, so a -/// run of either could evict an English keyword that a reader would call -/// adjacent. 64 comfortably spans `{"payment_method":{"card":{"number":…` -/// and a `カード`-prefixed line alike. -const CC_KEYWORD_WINDOW: usize = 64; - -/// True when [`CC_KEYWORD_RE`] matches within the window around -/// `start..end`, widened outward to char boundaries so the slice cannot -/// split a multi-byte character. -fn cc_keyword_near(norm: &str, start: usize, end: usize) -> bool { - let mut lo = start.saturating_sub(CC_KEYWORD_WINDOW); - while lo > 0 && !norm.is_char_boundary(lo) { - lo -= 1; - } - let mut hi = (end + CC_KEYWORD_WINDOW).min(norm.len()); - while hi < norm.len() && !norm.is_char_boundary(hi) { - hi += 1; - } - CC_KEYWORD_RE.is_match(&norm[lo..hi]) -} - -// Sort by start asc, length desc. Then walk in order, dropping any hit whose -// range overlaps a kept hit. Result: earlier + longer wins; no double-redact. -fn dedupe_overlaps(hits: &mut Vec) { - hits.sort_by(|a, b| { - a.start - .cmp(&b.start) - .then((b.end - b.start).cmp(&(a.end - a.start))) - }); - let mut kept: Vec = Vec::with_capacity(hits.len()); - for h in hits.drain(..) { - let overlaps = kept.last().is_some_and(|k| h.start < k.end); - if !overlaps { - kept.push(h); - } - } - *hits = kept; -} - -// Splice redactions (whose indices reference NORMALIZED text) back into the -// ORIGINAL text via NormalizedView's byte-offset mapping. This preserves -// non-PII original bytes verbatim (including fullwidth glyphs the user -// intentionally typed) while still scrubbing detected PII. -fn splice_redactions( - original: &str, - nview: &NormalizedView, - hits: Vec, - report: &mut SanitizationReport, -) -> Sanitized { - if hits.is_empty() { - return Sanitized { - value: original.to_string(), - report: *report, - }; - } - let mut out = String::with_capacity(original.len()); - let mut cursor = 0; - for h in &hits { - let start_orig = nview.norm_to_orig(h.start); - let end_orig = nview.norm_to_orig(h.end); - if start_orig < cursor || start_orig > original.len() || end_orig > original.len() { - continue; - } - out.push_str(&original[cursor..start_orig]); - out.push_str(h.token); - cursor = end_orig; - } - out.push_str(&original[cursor..]); - report.pii_redactions += hits.len(); - Sanitized { - value: out, - report: *report, - } -} - -// ---------- Unicode normalization for matching ---------- - -// Fullwidth / zero-width normalization used before matching. Lives in its own -// module — see `normalize.rs`. -mod normalize; -use normalize::NormalizedView; - -// ---------- Checksum helpers ---------- - -#[cfg(test)] -#[path = "pii_tests.rs"] -mod tests; diff --git a/src/memory/store/safety/pii/checks.rs b/src/memory/store/safety/pii/checks.rs deleted file mode 100644 index e72513d7..00000000 --- a/src/memory/store/safety/pii/checks.rs +++ /dev/null @@ -1,332 +0,0 @@ -//! Checksum and structural validators for PII candidates. - -pub(super) fn digits(s: &str) -> Vec { - s.chars() - .filter(|c| c.is_ascii_digit()) - .map(|c| c.to_digit(10).expect("ascii digit")) - .collect() -} - -pub(super) fn valid_cpf(d: &[u32]) -> bool { - if d.len() != 11 || d.iter().all(|x| *x == d[0]) { - return false; - } - let s1: u32 = (0..9).map(|i| d[i] * (10 - i as u32)).sum(); - let dv1 = (s1 * 10) % 11 % 10; - if dv1 != d[9] { - return false; - } - let s2: u32 = (0..10).map(|i| d[i] * (11 - i as u32)).sum(); - let dv2 = (s2 * 10) % 11 % 10; - dv2 == d[10] -} - -pub(super) fn valid_cnpj(d: &[u32]) -> bool { - if d.len() != 14 || d.iter().all(|x| *x == d[0]) { - return false; - } - let w1: [u32; 12] = [5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2]; - let s1: u32 = (0..12).map(|i| d[i] * w1[i]).sum(); - let r1 = s1 % 11; - let dv1 = if r1 < 2 { 0 } else { 11 - r1 }; - if dv1 != d[12] { - return false; - } - let w2: [u32; 13] = [6, 5, 4, 3, 2, 9, 8, 7, 6, 5, 4, 3, 2]; - let s2: u32 = (0..13).map(|i| d[i] * w2[i]).sum(); - let r2 = s2 % 11; - let dv2 = if r2 < 2 { 0 } else { 11 - r2 }; - dv2 == d[13] -} - -pub(super) fn valid_cuit(d: &[u32]) -> bool { - if d.len() != 11 { - return false; - } - let w: [u32; 10] = [5, 4, 3, 2, 7, 6, 5, 4, 3, 2]; - let s: u32 = (0..10).map(|i| d[i] * w[i]).sum(); - let r = s % 11; - let dv = match r { - 0 => 0, - 1 => return false, - _ => 11 - r, - }; - dv == d[10] -} - -// Luhn — used for credit-card validation. -pub(super) fn valid_luhn(s: &str) -> bool { - let d = digits(s); - if d.len() < 13 || d.len() > 19 { - return false; - } - let mut sum = 0u32; - let mut alt = false; - for x in d.iter().rev() { - let v = if alt { - let doubled = x * 2; - if doubled > 9 { - doubled - 9 - } else { - doubled - } - } else { - *x - }; - sum += v; - alt = !alt; - } - sum.is_multiple_of(10) -} - -/// True when a digit string has a plausible payment-card shape: a known -/// major-network IIN prefix at a length that network actually issues. -/// -/// The structural gate behind bare (separator-less) credit-card redaction. -/// Luhn alone passes ~10% of arbitrary digit runs — the same raw -/// false-positive rate that put bare Aadhaar behind a keyword — and 13-19 -/// contiguous digits is a common machine-identifier shape: 13-digit -/// epoch-millisecond timestamps (`17…`/`18…` for decades either side of now) -/// sit squarely in the window and were being redacted out of stored JSON at -/// that rate (opencompany#1201). No card network issues from a `17`/`18` -/// prefix, so requiring a real IIN removes that entire class while keeping -/// every number a major network could actually have issued. -/// -/// The table lists each supported network's published IIN ranges at the -/// lengths that network issues: Visa, Mastercard (incl. the 2-series), Amex, -/// Discover, JCB, Diners Club, UnionPay, Maestro, Mir, RuPay, and the -/// Brazilian networks Elo and Hipercard (in scope by this module's own -/// design: it already carries bare and formatted CPF/CNPJ). It is a -/// *corroboration* tier, not an acquirer's validator: an exhaustive BIN -/// registry is a licensed, continuously updated database, and a range missing -/// here is not silently dropped from redaction — a bare PAN on an unlisted -/// network still redacts whenever a card keyword appears within the keyword -/// window (the third gate in `push_credit_cards`). -/// -/// One dated caveat, so nobody inherits a stronger claim than the code makes: -/// "timestamps can never corroborate" is prefix-and-length dependent, not -/// absolute. 13-digit epoch-milliseconds stay out of every range until the -/// year 2096 (`4…`, Visa's 13-digit arm); but 16-digit epoch-MICROsecond -/// stamps enter Mir's `2200-2204` window in late 2039 and Mastercard's -/// 2-series `2221-2720` from ~2040 to ~2056. If this code outlives that, -/// those stamps redact at Luhn's ~10% again and this gate needs a rethink. -pub(super) fn plausible_card_number(d: &[u32]) -> bool { - let len = d.len(); - if !(13..=19).contains(&len) { - return false; - } - // len >= 13 makes the first four digits always present. - let p2 = d[0] * 10 + d[1]; - let p3 = p2 * 10 + d[2]; - let p4 = p3 * 10 + d[3]; - match d[0] { - // Visa: 16 standard, 13 legacy, 19 extended. (Also where Elo's - // 4-prefixed ranges land, at the same 16.) - 4 => matches!(len, 13 | 16 | 19), - 5 => { - // Mastercard 51-55 (16 only). - ((51..=55).contains(&p2) && len == 16) - // Maestro 5018/5020/5038/5893 and 56-58. Maestro issues - // 12-19; the floor here is 13 because CC_RE requires 13 - // digits, so the explicit bound below is the whole window - // this function can see — kept explicit so the "at an - // issued length" promise stays visibly true. - || ((matches!(p4, 5018 | 5020 | 5038 | 5893) || (56..=58).contains(&p2)) - && (13..=19).contains(&len)) - // Elo 5041/5066/5067 (16), RuPay 508 (16). - || ((matches!(p4, 5041 | 5066 | 5067) || p3 == 508) && len == 16) - } - 2 => { - // Mastercard 2-series 2221-2720 (16 only). - // Mastercard 2221-2720 and Mir 2200-2204 (16 only). - ((2221..=2720).contains(&p4) || (2200..=2204).contains(&p4)) && len == 16 - } - 3 => { - // Amex 34/37 (15 only). - (matches!(p2, 34 | 37) && len == 15) - // JCB 3528-3589 (16-19). - || ((3528..=3589).contains(&p4) && (16..=19).contains(&len)) - // Diners Club 36 / 300-305 / 3095 / 38-39, one scheme, one - // length rule: 14 (Diners International / Carte Blanche - // classic — 30569309025904, the canonical test PAN, is 14) - // through 19. - || ((p2 == 36 - || (300..=305).contains(&p3) - || p4 == 3095 - || matches!(p2, 38 | 39)) - && (14..=19).contains(&len)) - } - 6 => { - // Discover 6011 / 644-649 / 65 (16 or 19). - ((p4 == 6011 || (644..=649).contains(&p3) || p2 == 65) && matches!(len, 16 | 19)) - // UnionPay 62 (16-19), which also covers the - // Discover-processed 622126-622925 range. - || (p2 == 62 && (16..=19).contains(&len)) - // Maestro 6304/6759/6761-6763, bounded as the 5-prefix - // Maestro arm above. - || (matches!(p4, 6304 | 6759 | 6761 | 6762 | 6763) && (13..=19).contains(&len)) - // RuPay 60 (16) beyond the 65 range shared with Discover; - // Elo 6277/6362/6363 and Hipercard 6062 (16). - || ((p2 == 60 || matches!(p4, 6277 | 6362 | 6363 | 6062)) && len == 16) - } - // RuPay 81/82 (16). - 8 => matches!(p2, 81 | 82) && len == 16, - _ => false, - } -} - -// IBAN mod-97. Steps: strip spaces, move first 4 chars to end, expand letters -// (A=10..Z=35), divide as a big-integer mod 97, require remainder == 1. -pub(super) fn valid_iban(s: &str) -> bool { - let cleaned: String = s.chars().filter(|c| !c.is_whitespace()).collect(); - if cleaned.len() < 15 || cleaned.len() > 34 { - return false; - } - if !cleaned.chars().take(2).all(|c| c.is_ascii_alphabetic()) { - return false; - } - if !cleaned[2..4].chars().all(|c| c.is_ascii_digit()) { - return false; - } - let rotated: String = cleaned[4..].chars().chain(cleaned[..4].chars()).collect(); - let mut remainder: u64 = 0; - for c in rotated.chars() { - let chunk = if let Some(d) = c.to_digit(10) { - d as u64 - } else if c.is_ascii_alphabetic() { - (c.to_ascii_uppercase() as u64) - ('A' as u64) + 10 - } else { - return false; - }; - // Expand into the running remainder digit-by-digit so we never need - // u128. Each letter contributes 2 decimal digits. - if chunk >= 10 { - remainder = (remainder * 100 + chunk) % 97; - } else { - remainder = (remainder * 10 + chunk) % 97; - } - } - remainder == 1 -} - -// Verhoeff — used for Aadhaar. -const VERHOEFF_D: [[u8; 10]; 10] = [ - [0, 1, 2, 3, 4, 5, 6, 7, 8, 9], - [1, 2, 3, 4, 0, 6, 7, 8, 9, 5], - [2, 3, 4, 0, 1, 7, 8, 9, 5, 6], - [3, 4, 0, 1, 2, 8, 9, 5, 6, 7], - [4, 0, 1, 2, 3, 9, 5, 6, 7, 8], - [5, 9, 8, 7, 6, 0, 4, 3, 2, 1], - [6, 5, 9, 8, 7, 1, 0, 4, 3, 2], - [7, 6, 5, 9, 8, 2, 1, 0, 4, 3], - [8, 7, 6, 5, 9, 3, 2, 1, 0, 4], - [9, 8, 7, 6, 5, 4, 3, 2, 1, 0], -]; -const VERHOEFF_P: [[u8; 10]; 8] = [ - [0, 1, 2, 3, 4, 5, 6, 7, 8, 9], - [1, 5, 7, 6, 2, 8, 3, 0, 9, 4], - [5, 8, 0, 3, 7, 9, 6, 1, 4, 2], - [8, 9, 1, 6, 0, 4, 3, 5, 2, 7], - [9, 4, 5, 3, 1, 2, 6, 8, 7, 0], - [4, 2, 8, 6, 5, 7, 3, 9, 0, 1], - [2, 7, 9, 3, 8, 0, 6, 4, 1, 5], - [7, 0, 4, 6, 9, 1, 3, 2, 5, 8], -]; - -pub(super) fn valid_verhoeff(d: &[u32]) -> bool { - if d.len() != 12 { - return false; - } - // Aadhaar can't start with 0 or 1. - if d[0] < 2 { - return false; - } - let mut c: u8 = 0; - for (i, digit) in d.iter().rev().enumerate() { - c = VERHOEFF_D[c as usize][VERHOEFF_P[i % 8][*digit as usize] as usize]; - } - c == 0 -} - -// US SSN reserved/invalid ranges per SSA. -pub(super) fn valid_ssn(s: &str) -> bool { - let d = digits(s); - if d.len() != 9 { - return false; - } - let area = d[0] * 100 + d[1] * 10 + d[2]; - let group = d[3] * 10 + d[4]; - let serial = d[5] * 1000 + d[6] * 100 + d[7] * 10 + d[8]; - if area == 0 || area == 666 || area >= 900 { - return false; - } - if group == 0 || serial == 0 { - return false; - } - true -} - -// Spain DNI check letter — 8 digits mod 23 indexes into a fixed letter table. -const DNI_LETTERS: &[u8; 23] = b"TRWAGMYFPDXBNJZSQVHLCKE"; - -pub(super) fn valid_dni_es(s: &str) -> bool { - let upper = s.to_ascii_uppercase(); - let bytes = upper.as_bytes(); - if bytes.len() != 9 { - return false; - } - let num_str = &upper[..8]; - let letter = bytes[8]; - let Ok(num) = num_str.parse::() else { - return false; - }; - DNI_LETTERS[(num % 23) as usize] == letter -} - -pub(super) fn valid_nie_es(s: &str) -> bool { - let upper = s.to_ascii_uppercase(); - let bytes = upper.as_bytes(); - if bytes.len() != 9 { - return false; - } - let prefix = match bytes[0] { - b'X' => 0u32, - b'Y' => 1, - b'Z' => 2, - _ => return false, - }; - let Ok(rest) = std::str::from_utf8(&bytes[1..8]) else { - return false; - }; - let Ok(num) = rest.parse::() else { - return false; - }; - let composed = prefix * 10_000_000 + num; - DNI_LETTERS[(composed % 23) as usize] == bytes[8] -} - -// UK NINO reserved-prefix blacklist. -pub(super) fn valid_nino(s: &str) -> bool { - let upper = s.to_ascii_uppercase(); - let bytes = upper.as_bytes(); - if bytes.len() != 9 { - return false; - } - // First char cannot be D F I Q U V; second cannot be D F I O Q U V. - let bad_first = b"DFIQUV"; - let bad_second = b"DFIOQUV"; - if bad_first.contains(&bytes[0]) || bad_second.contains(&bytes[1]) { - return false; - } - // Reserved two-letter prefixes. - let reserved = ["BG", "GB", "KN", "NK", "NT", "TN", "ZZ"]; - let prefix = &upper[..2]; - if reserved.contains(&prefix) { - return false; - } - true -} - -#[cfg(test)] -#[path = "checks_tests.rs"] -mod tests; diff --git a/src/memory/store/safety/pii/checks_tests.rs b/src/memory/store/safety/pii/checks_tests.rs deleted file mode 100644 index e05291e9..00000000 --- a/src/memory/store/safety/pii/checks_tests.rs +++ /dev/null @@ -1,167 +0,0 @@ -use super::*; - -#[test] -fn tax_ids_enforce_lengths_checksums_and_repetition_rules() { - assert!(valid_cpf(&digits("529.982.247-25"))); - assert!(!valid_cpf(&digits("111.111.111-11"))); - assert!(!valid_cpf(&digits("5299822472"))); - assert!(valid_cnpj(&digits("11.222.333/0001-81"))); - assert!(!valid_cnpj(&digits("11.222.333/0001-82"))); - assert!(!valid_cnpj(&digits("00000000000000"))); - assert!(valid_cuit(&digits("20-12345678-6"))); - assert!(!valid_cuit(&digits("20-12345678-7"))); - assert!(!valid_cuit(&digits("2012345678"))); -} - -#[test] -fn payment_checksums_reject_bad_bounds_and_checksums() { - assert!(valid_luhn("4111 1111 1111 1111")); - assert!(!valid_luhn("4111 1111 1111 1112")); - assert!(!valid_luhn("7992739871")); - assert!(valid_iban("GB82 WEST 1234 5698 7654 32")); - assert!(!valid_iban("GB82 WEST 1234 5698 7654 33")); - assert!(!valid_iban("GB00")); -} - -#[test] -fn identity_validators_cover_checksums_reserved_values_and_prefixes() { - assert!(valid_verhoeff(&digits("234567890124"))); - assert!(!valid_verhoeff(&digits("134567890124"))); - assert!(!valid_verhoeff(&digits("234567890125"))); - assert!(valid_ssn("123-45-6789")); - assert!(!valid_ssn("666-45-6789")); - assert!(!valid_ssn("123-00-6789")); - assert!(!valid_ssn("123-45-0000")); - assert!(valid_dni_es("12345678Z")); - assert!(!valid_dni_es("12345678A")); - assert!(valid_nie_es("X1234567L")); - assert!(!valid_nie_es("A1234567L")); - assert!(valid_nino("AA123456A")); - assert!(!valid_nino("BG123456A")); - assert!(!valid_nino("DA123456A")); - assert!(!valid_nino("AA12345A")); -} - -#[test] -fn plausible_card_number_requires_a_real_iin_at_an_issued_length() { - // No network's prefix: epoch-millisecond timestamps and other machine ids. - assert!(!plausible_card_number(&digits("1787178633773"))); // 13-digit epoch ms - assert!(!plausible_card_number(&digits("1700000000000"))); - assert!(!plausible_card_number(&digits("9111111111111119"))); - assert!(!plausible_card_number(&digits("2000000000000000"))); // year-2033 epoch-µs shape - assert!(!plausible_card_number(&digits("1900000000000"))); // 13-digit, no IIN starts 1 - - // Out of the card length window entirely. - assert!(!plausible_card_number(&digits("411111111111"))); // 12 - assert!(!plausible_card_number(&digits("41111111111111111111"))); // 20 -} - -// The accept direction, per network, at its boundary lengths — with the -// reject cases one step past each length and each range edge. Luhn is -// irrelevant here: the function judges shape only, and the redaction path -// tests Luhn separately. -#[test] -fn plausible_card_number_accepts_each_network_at_its_boundary_lengths() { - // Visa 13/16/19; nothing between or past. - assert!(plausible_card_number(&digits("4222222222222"))); // 13 - assert!(plausible_card_number(&digits("4111111111111111"))); // 16 - assert!(plausible_card_number(&digits("4111111111111111111"))); // 19 - assert!(!plausible_card_number(&digits("41111111111111"))); // 14 - assert!(!plausible_card_number(&digits("411111111111111"))); // 15 - assert!(!plausible_card_number(&digits("41111111111111111"))); // 17 - - // Mastercard 51-55 and 2221-2720, 16 only; edges out both sides. - assert!(plausible_card_number(&digits("5100000000000000"))); - assert!(plausible_card_number(&digits("5500005555555559"))); - assert!(plausible_card_number(&digits("2221000000000009"))); - assert!(plausible_card_number(&digits("2720000000000000"))); - assert!(!plausible_card_number(&digits("550000555555555"))); // 15 - assert!(!plausible_card_number(&digits("55000055555555590"))); // 17 - assert!(!plausible_card_number(&digits("5000000000000000"))); // 50: not MC - assert!(!plausible_card_number(&digits("2220000000000000"))); // below 2221 - assert!(!plausible_card_number(&digits("2721000000000000"))); // above 2720 - - // Mir 2200-2204, 16 only. - assert!(plausible_card_number(&digits("2200000000000004"))); - assert!(plausible_card_number(&digits("2204000000000000"))); - assert!(!plausible_card_number(&digits("2205000000000009"))); // past range - assert!(!plausible_card_number(&digits("220000000000000"))); // 15 - assert!(!plausible_card_number(&digits("22000000000000004"))); // 17 - - // Amex 34/37, 15 only. - assert!(plausible_card_number(&digits("378282246310005"))); - assert!(plausible_card_number(&digits("340000000000009"))); - assert!(!plausible_card_number(&digits("37828224631000"))); // 14 - assert!(!plausible_card_number(&digits("3782822463100051"))); // 16 - assert!(!plausible_card_number(&digits("350000000000000"))); // 35 alone - - // JCB 3528-3589, 16-19. - assert!(plausible_card_number(&digits("3530111333300000"))); - assert!(plausible_card_number(&digits("3589000000000000000"))); // 19 - assert!(!plausible_card_number(&digits("3527000000000000"))); - assert!(!plausible_card_number(&digits("3590000000000000"))); - assert!(!plausible_card_number(&digits("353011133330000"))); // 15 - - // Diners Club — one scheme, one rule: 36, 300-305, 3095, 38, 39 all at - // 14-19. 30569309025904 is the canonical Diners test PAN; regression - // for the review finding that 14-digit Diners had been split away. - assert!(plausible_card_number(&digits("30569309025904"))); // 300-305 @ 14 - assert!(plausible_card_number(&digits("38520000023237"))); // 38 @ 14 - assert!(plausible_card_number(&digits("36700102000000"))); // 36 @ 14 - assert!(plausible_card_number(&digits("30950000000000"))); // 3095 @ 14 - assert!(plausible_card_number(&digits("39000000000005"))); // 39 @ 14 - assert!(plausible_card_number(&digits("3050000000000000002"))); // 305 @ 19 - assert!(!plausible_card_number(&digits("3060000000000000"))); // 306 - assert!(!plausible_card_number(&digits("3700010200000"))); // 37 @ 13: not Diners - - // Discover 6011 / 644-649 / 65 at 16 or 19. - assert!(plausible_card_number(&digits("6011111111111117"))); - assert!(plausible_card_number(&digits("6440000000000000"))); - assert!(plausible_card_number(&digits("6500000000000000000"))); // 19 - assert!(!plausible_card_number(&digits("60111111111111170"))); // 17 - assert!(!plausible_card_number(&digits("6430000000000000"))); // 643 - - // UnionPay 62 at 16-19. - assert!(plausible_card_number(&digits("6200000000000005"))); - assert!(plausible_card_number(&digits("6200000000000000005"))); // 19 - assert!(!plausible_card_number(&digits("620000000000000"))); // 15 - - // Maestro (5018/5020/5038/5893, 56-58, 6304/6759/6761-6763) at 13-19. - assert!(plausible_card_number(&digits("5018000000000"))); // 13 - assert!(plausible_card_number(&digits("5600000000002"))); // 56 @ 13 - assert!(plausible_card_number(&digits("5800000000000000008"))); // 58 @ 19 - assert!(plausible_card_number(&digits("6759000000000000000"))); // 19 - assert!(plausible_card_number(&digits("6763000000000000"))); - assert!(!plausible_card_number(&digits("5019000000000000"))); // 5019 - assert!(!plausible_card_number(&digits("5900000000000000"))); // 59 - assert!(!plausible_card_number(&digits("6760000000000000"))); // 6760 - - // RuPay 60 / 508 / 81 / 82 at 16. - assert!(plausible_card_number(&digits("6069850000000000"))); - assert!(plausible_card_number(&digits("5080000000000002"))); - assert!(plausible_card_number(&digits("8100000000000000"))); - assert!(plausible_card_number(&digits("8200000000000000"))); - assert!(!plausible_card_number(&digits("8300000000000000"))); // 83 - assert!(!plausible_card_number(&digits("810000000000000"))); // 15 - assert!(!plausible_card_number(&digits("5090000000000000"))); // 509 - - // Elo 5041/5066/5067/6277/6362/6363 and Hipercard 6062, 16. - assert!(plausible_card_number(&digits("5067310000000010"))); - assert!(plausible_card_number(&digits("5041000000000000"))); - assert!(plausible_card_number(&digits("6362000000000009"))); - assert!(plausible_card_number(&digits("6277000000000000"))); - // Hipercard 6062 sits inside RuPay's blanket 60 range as well — listed - // in the table in its own right, but note 60xx@16 is corroborable for - // any xx, which is what the published RuPay range says. - assert!(plausible_card_number(&digits("6062821234567890"))); // Hipercard - assert!(!plausible_card_number(&digits("5065000000000001"))); // 5065 - assert!(!plausible_card_number(&digits("6364000000000007"))); // 6364 - assert!(!plausible_card_number(&digits("506731000000001"))); // 15 - - // The documented expiry (see the function doc): 16-digit - // epoch-microsecond stamps enter Mir/Mastercard-2-series territory - // around 2039/2040. Asserted as truth, not as an endorsement — when - // this line starts mattering, the gate needs a rethink. - assert!(plausible_card_number(&digits("2221787178633773"))); // µs in ~2040 - assert!(!plausible_card_number(&digits("1787178633773000"))); // µs today -} diff --git a/src/memory/store/safety/pii/normalize.rs b/src/memory/store/safety/pii/normalize.rs deleted file mode 100644 index 36ee8a54..00000000 --- a/src/memory/store/safety/pii/normalize.rs +++ /dev/null @@ -1,79 +0,0 @@ -//! Unicode normalization for PII matching. -//! -//! A pre-pass that defeats fullwidth-digit and zero-width-char bypasses while -//! keeping a byte map back to the original string, so matches found on the -//! normalized view can be spliced onto the exact original bytes. - -pub(super) struct NormalizedView { - pub(super) normalized: String, - // For each byte offset i in `normalized`, `byte_map[i]` is the byte offset - // in the original string where the corresponding char *starts*. - // The last entry maps the normalized length to the original length, so - // `norm_to_orig(normalized.len())` is well-defined. - byte_map: Vec, -} - -impl NormalizedView { - pub(super) fn build(original: &str) -> Self { - let mut normalized = String::with_capacity(original.len()); - let mut byte_map: Vec = Vec::with_capacity(original.len() + 1); - for (idx, ch) in original.char_indices() { - if is_zero_width(ch) { - continue; - } - let mapped = fold_char(ch); - let start = normalized.len(); - normalized.push(mapped); - // One byte_map entry per byte of the normalized char. - let added = normalized.len() - start; - for _ in 0..added { - byte_map.push(idx); - } - } - byte_map.push(original.len()); - Self { - normalized, - byte_map, - } - } - - pub(super) fn norm_to_orig(&self, norm_byte: usize) -> usize { - if norm_byte >= self.byte_map.len() { - return *self.byte_map.last().unwrap_or(&0); - } - self.byte_map[norm_byte] - } -} - -fn is_zero_width(c: char) -> bool { - matches!( - c, - '\u{200B}' - | '\u{200C}' - | '\u{200D}' - | '\u{200E}' - | '\u{200F}' - | '\u{2060}' - | '\u{180E}' - | '\u{FEFF}' - ) -} - -fn fold_char(c: char) -> char { - match c { - // Fullwidth digits 0-9 - '\u{FF10}'..='\u{FF19}' => char::from_u32(c as u32 - 0xFF10 + 0x30).unwrap_or(c), - // Arabic-Indic digits ٠-٩ - '\u{0660}'..='\u{0669}' => char::from_u32(c as u32 - 0x0660 + 0x30).unwrap_or(c), - // Eastern Arabic-Indic digits ۰-۹ - '\u{06F0}'..='\u{06F9}' => char::from_u32(c as u32 - 0x06F0 + 0x30).unwrap_or(c), - // Common fullwidth punctuation we care about for PII formats - '\u{FF0D}' => '-', - '\u{FF0E}' => '.', - '\u{FF0F}' => '/', - '\u{FF1A}' => ':', - '\u{2010}'..='\u{2015}' => '-', // various unicode hyphens/dashes - '\u{2212}' => '-', // minus sign - other => other, - } -} diff --git a/src/memory/store/safety/pii/prefilter.rs b/src/memory/store/safety/pii/prefilter.rs deleted file mode 100644 index abace534..00000000 --- a/src/memory/store/safety/pii/prefilter.rs +++ /dev/null @@ -1,238 +0,0 @@ -//! Byte-oriented candidate pre-filter for the PII redactor. -//! -//! Replaces the always-resident combined `RegexSet` (one shared NFA plus a -//! per-thread lazy-DFA cache in *every* process/thread) with a single cheap pass -//! over the raw bytes. The scan derives per-class candidate flags from a handful -//! of structural signals — digit-run lengths, punctuation presence, uppercase / -//! alpha presence, `+`, and case-insensitive keyword probes (including the -//! non-Latin Aadhaar `आधार` and My-Number `マイナンバー` / `個人番号` keywords). -//! Each flag then decides whether that class's precise validation regex is worth -//! compiling and running; the precise `Regex`es stay `LazyLock`, so a class that -//! never sees a candidate is never compiled at all. At 100–1000 concurrent -//! agents that turns "combined NFA + N thread-local DFA caches resident forever" -//! into "only the regexes a workload actually needs, compiled on first hit". -//! -//! Correctness: every flag is a NECESSARY CONDITION of the class's *precise* -//! regex, so a flag can only over-fire (harmless — the precise regex then simply -//! fails to match), never under-fire on real PII. Consequently, whenever a -//! precise pattern would have matched under the old code path, its flag is set -//! and it still runs — output is unchanged. The union of the flags is a superset -//! of the old `SCREEN` set (pinned by `prefilter_is_superset_of_legacy_screen`). -//! The NANP phone class gates on the *screen*-entry necessary condition — an -//! internal `digit sep digit` separator OR a `\d{11,}` run (the old SCREEN reached -//! `PHONE_NANP_RE` through both) — faithfully preserving the documented "a bare -//! 10-digit NANP run is never reached" behavior while still redacting a bare -//! `1`+10-digit country-code number — see -//! `redact_pii_does_not_reach_bare_10_digit_nanp_today`. - -/// Per-class candidate flags produced by [`scan_candidates`]. A set flag means -/// "run this class's precise regex"; an unset flag means the class cannot -/// possibly match, so its regex is skipped (and never compiled). -#[derive(Default, Clone, Copy)] -pub(super) struct Candidates { - pub(super) cpf_fmt: bool, - pub(super) cnpj_fmt: bool, - pub(super) cuit: bool, - pub(super) iban: bool, - pub(super) cc: bool, - pub(super) cnpj_bare: bool, - pub(super) cpf_bare: bool, - pub(super) aadhaar_fmt: bool, - pub(super) aadhaar_kw: bool, - pub(super) dni: bool, - pub(super) nie: bool, - pub(super) nino: bool, - pub(super) ssn: bool, - pub(super) rrn: bool, - pub(super) rfc: bool, - pub(super) pan_in: bool, - pub(super) phone_e164: bool, - pub(super) phone_nanp: bool, - pub(super) mynumber: bool, -} - -impl Candidates { - /// True if any class is a candidate — i.e. the text is worth a precise pass. - pub(super) fn any(&self) -> bool { - self.cpf_fmt - || self.cnpj_fmt - || self.cuit - || self.iban - || self.cc - || self.cnpj_bare - || self.cpf_bare - || self.aadhaar_fmt - || self.aadhaar_kw - || self.dni - || self.nie - || self.nino - || self.ssn - || self.rrn - || self.rfc - || self.pan_in - || self.phone_e164 - || self.phone_nanp - || self.mynumber - } -} - -/// Case-insensitive (ASCII-only case folding) substring test over raw bytes. -/// Non-ASCII bytes compare exactly, so this also serves as an exact matcher for -/// the multibyte Devanagari / Japanese keyword needles. -fn contains_ci(hay: &[u8], needle: &[u8]) -> bool { - if needle.is_empty() { - return true; - } - if hay.len() < needle.len() { - return false; - } - hay.windows(needle.len()) - .any(|w| w.iter().zip(needle).all(|(a, b)| a.eq_ignore_ascii_case(b))) -} - -/// Aadhaar keyword needles — ASCII forms plus Devanagari `आधार`. -const AADHAAR_KEYWORDS: &[&[u8]] = &[b"aadhaar", b"aadhar", b"uidai", b"uid", "आधार".as_bytes()]; -/// My-Number Japanese keyword needles. The English `My\s?Number` variant is -/// handled separately (see `scan_candidates`) so any `\s` separator between the -/// two words is recognised, not just a literal space. -const MYNUMBER_JP_KEYWORDS: &[&[u8]] = &["マイナンバー".as_bytes(), "個人番号".as_bytes()]; - -/// Single linear pass over the bytes deriving every per-class candidate flag. -/// -/// Only ASCII structural bytes carry signal here; multibyte UTF-8 lead / -/// continuation bytes are all `>= 0x80`, so scanning `as_bytes()` for ASCII -/// digits/punctuation/letters is boundary-safe. Keyword probes run over the -/// same byte slice so the non-Latin needles match verbatim. -pub(super) fn scan_candidates(text: &str) -> Candidates { - let bytes = text.as_bytes(); - - let mut total_digits: usize = 0; - let mut max_digit_run: usize = 0; - let mut cur_run: usize = 0; - let mut has_dot = false; - let mut has_dash = false; - let mut has_slash = false; - // Any ASCII whitespace separator (space, tab, newline, CR, form feed, - // vertical tab). The precise Aadhaar pattern separates its groups with - // `[\s-]`, which matches the whole `\s` class — so gating on space/tab - // alone would under-fire on newline-separated Aadhaar (a real PII drop). - let mut has_ws = false; - let mut has_upper = false; - let mut has_alpha = false; - let mut has_xyz = false; - let mut has_plus = false; - // NANP-style "separated group" signal: some `[digit or ')'] [sep] [digit]` - // window exists (sep ∈ space/tab/./-). This is the necessary condition of - // the old SCREEN NANP entry, which required internal separators — keeping - // bare separator-less 10-digit runs out of the phone path. - let mut nanp_sep = false; - - for (i, &b) in bytes.iter().enumerate() { - if b.is_ascii_digit() { - total_digits += 1; - cur_run += 1; - if cur_run > max_digit_run { - max_digit_run = cur_run; - } - } else { - cur_run = 0; - match b { - b'.' => has_dot = true, - b'-' => has_dash = true, - b'/' => has_slash = true, - b' ' | b'\t' | b'\n' | b'\r' | 0x0b | 0x0c => has_ws = true, - b'+' => has_plus = true, - b'A'..=b'Z' => { - has_upper = true; - has_alpha = true; - if matches!(b, b'X' | b'Y' | b'Z') { - has_xyz = true; - } - } - b'a'..=b'z' => { - has_alpha = true; - if matches!(b, b'x' | b'y' | b'z') { - has_xyz = true; - } - } - _ => {} - } - } - - if matches!(b, b' ' | b'\t' | b'.' | b'-') && i > 0 && i + 1 < bytes.len() { - let prev = bytes[i - 1]; - let next = bytes[i + 1]; - if (prev.is_ascii_digit() || prev == b')') && next.is_ascii_digit() { - nanp_sep = true; - } - } - } - - let has_digit = total_digits > 0; - let aadhaar_kw = AADHAAR_KEYWORDS.iter().any(|kw| contains_ci(bytes, kw)); - // English `My\s?Number` accepts any single `\s` between the words, so a tab- - // or newline-separated keyword (`My\tNumber`) must still flag. Requiring both - // `my` and `number` substrings is a necessary condition of the precise regex - // and covers every whitespace variant; it may over-fire (harmless — the - // precise `MYNUM_RE` re-checks the separator and the trailing 12 digits). - let mynumber = MYNUMBER_JP_KEYWORDS.iter().any(|kw| contains_ci(bytes, kw)) - || (contains_ci(bytes, b"my") && contains_ci(bytes, b"number")); - - let cand = Candidates { - // Formatted CPF `\d{3}\.\d{3}\.\d{3}-\d{2}` — needs digits, `.`, `-`. - cpf_fmt: has_digit && has_dot && has_dash, - // Formatted CNPJ `\d{2}\.\d{3}\.\d{3}/\d{4}-\d{2}` — adds `/`. - cnpj_fmt: has_digit && has_dot && has_slash && has_dash, - // CUIT `\d{2}-\d{8}-\d` — needs digits and `-`. - cuit: has_digit && has_dash, - // IBAN `[A-Z]{2}\d{2}…` — case-sensitive uppercase letters and digits. - iban: has_upper && has_digit, - // Credit card `(?:\d[\s\-]?){13,19}` — at least 13 digits total. - cc: total_digits >= 13, - // Bare CNPJ `\d{14}` — a 14-long digit run. - cnpj_bare: max_digit_run >= 14, - // Bare CPF `\d{11}` — an 11-long digit run. - cpf_bare: max_digit_run >= 11, - // Formatted Aadhaar `\d{4}[\s-]\d{4}[\s-]\d{4}` — 12 digits + a `\s`/dash - // separator (any ASCII whitespace, matching the precise `[\s-]` class). - aadhaar_fmt: total_digits >= 12 && (has_ws || has_dash), - // Keyword-gated Aadhaar — keyword suffices (precise regex checks digits). - aadhaar_kw, - // Spain DNI `\d{8}[A-Z]` — 8-run plus a letter. - dni: max_digit_run >= 8 && has_alpha, - // Spain NIE `[XYZ]\d{7}[A-Z]` — X/Y/Z, 7-run, letter. - nie: has_xyz && max_digit_run >= 7 && has_alpha, - // UK NINO `[A-Z]{2}\d{6}[A-D]` — letters and a 6-run. - nino: max_digit_run >= 6 && has_alpha, - // US SSN `\d{3}-\d{2}-\d{4}` — digits and `-`. - ssn: has_digit && has_dash, - // Korea RRN `\d{6}-[1-4]\d{6}` — a 6-run and `-`. - rrn: max_digit_run >= 6 && has_dash, - // Mexico RFC `[A-ZÑ&]{3,4}\d{6}[A-Z0-9]{3}` — a 6-run (leading class may - // be all non-ASCII `Ñ`, so gate on the digit run alone, not on letters). - rfc: max_digit_run >= 6, - // India PAN `[A-Z]{5}\d{4}[A-Z]` — letters and a 4-run. - pan_in: max_digit_run >= 4 && has_alpha, - // E.164 `\+\d{7,15}` — a `+` and a 7+ digit run. - phone_e164: has_plus && max_digit_run >= 7, - // NANP — screen-entry necessary condition. The old SCREEN reached - // `PHONE_NANP_RE` via either the separated-group pattern OR the long - // `\d{11,}` run (which covers a bare `1`+10-digit country-code number - // like `12025551234`). A bare 10-digit run still stays out of the phone - // path (no internal separator, run length 10 < 11). - phone_nanp: nanp_sep || max_digit_run >= 11, - // My Number — keyword suffices (precise regex checks the 12 digits). - mynumber, - }; - - log::trace!( - "[pii] scan_candidates bytes={} digits={} max_run={} nanp_sep={} any={}", - bytes.len(), - total_digits, - max_digit_run, - nanp_sep, - cand.any() - ); - - cand -} diff --git a/src/memory/store/safety/pii_tests.rs b/src/memory/store/safety/pii_tests.rs deleted file mode 100644 index e6b4c48d..00000000 --- a/src/memory/store/safety/pii_tests.rs +++ /dev/null @@ -1,661 +0,0 @@ -use super::*; - -fn redacts(input: &str, token: &str) { - let out = redact_pii(input); - assert!( - out.value.contains(token), - "expected {token} in output. input={input:?} output={out:?}" - ); -} - -fn unchanged(input: &str) { - let out = redact_pii(input); - assert_eq!( - out.value, input, - "expected no change; report={:?}", - out.report - ); - assert_eq!(out.report.pii_redactions, 0); -} - -// --- CPF --- -#[test] -fn cpf_formatted_valid_redacted() { - redacts("CPF: 111.444.777-35.", PII_CPF); -} -#[test] -fn cpf_formatted_invalid_kept() { - unchanged("CPF 111.444.777-99 nope"); -} -#[test] -fn cpf_all_same_digits_rejected() { - unchanged("Test 111.111.111-11"); -} -#[test] -fn cpf_bare_valid_redacted() { - redacts("Sem mascara 11144477735 ok", PII_CPF); -} - -// --- CNPJ --- -#[test] -fn cnpj_formatted_valid_redacted() { - redacts("CNPJ 11.222.333/0001-81", PII_CNPJ); -} -#[test] -fn cnpj_bare_valid_redacted() { - redacts("contract 11222333000181 yes", PII_CNPJ); -} - -// --- CUIT --- -#[test] -fn cuit_valid_redacted() { - redacts("CUIT 20-11111111-2", PII_CUIT); -} -#[test] -fn cuit_invalid_kept() { - unchanged("noise 20-12345678-0 noise"); -} - -// --- RFC --- -#[test] -fn rfc_redacted() { - redacts("Mi RFC VECJ880326XK4 .", PII_RFC); -} -#[test] -fn rfc_lowercase_redacted() { - redacts("rfc vecj880326xk4", PII_RFC); -} - -// --- My Number --- -#[test] -fn my_number_redacted_with_keyword() { - redacts("マイナンバー: 123456789012", PII_MYNUM); -} -#[test] -fn bare_12_digits_without_keyword_kept() { - unchanged("Order 123456789012 shipped today."); -} -#[test] -fn my_number_keyword_tab_separator_redacted() { - // `My\s?Number` accepts any single `\s`; the byte prefilter must recognise a - // tab-separated keyword, not just a literal space. - redacts("My\tNumber 123456789012", PII_MYNUM); -} -#[test] -fn my_number_keyword_newline_separator_redacted() { - redacts("My\nNumber 123456789012", PII_MYNUM); -} - -// --- E.164 + NANP phone --- -#[test] -fn e164_redacted() { - redacts("phone +15551234567", PII_PHONE); -} -#[test] -fn nanp_formatted_redacted() { - redacts("call 415-555-0123 thanks", PII_PHONE); -} -#[test] -fn nanp_with_country_code_redacted() { - redacts("+1 (212) 555-7890", PII_PHONE); -} -#[test] -fn nanp_invalid_area_code_kept() { - unchanged("score 115-555-0123 ish"); -} -#[test] -fn nanp_bare_country_code_redacted() { - // Separator-less `1`+10-digit NANP: the old SCREEN reached PHONE_NANP_RE via - // the `\d{11,}` run; the prefilter must keep gating this through the phone - // class (the bare-CPF checksum rejects it, so nothing else redacts it). - redacts("12025551234", PII_PHONE); -} - -// --- SSN --- -#[test] -fn ssn_valid_redacted() { - redacts("ssn 123-45-6789", PII_SSN); -} -#[test] -fn ssn_reserved_area_kept() { - unchanged("test 666-12-3456"); -} -#[test] -fn ssn_zero_serial_kept() { - unchanged("test 123-45-0000"); -} - -// --- Credit card / Luhn --- -#[test] -fn credit_card_visa_redacted() { - // Visa test number with valid Luhn. - redacts("card 4111 1111 1111 1111 thanks", PII_CC); -} -#[test] -fn credit_card_amex_redacted() { - redacts("card 378282246310005 used", PII_CC); -} -#[test] -fn credit_card_invalid_luhn_kept() { - unchanged("invoice 4111 1111 1111 1112"); -} -#[test] -fn credit_card_bare_visa_redacted_without_keyword() { - // A real network IIN (Visa `4`) at an issued length: bare runs with an - // issued card shape still redact with no keyword anywhere near. - redacts("4111111111111111", PII_CC); -} -#[test] -fn credit_card_bare_amex_redacted_without_keyword() { - redacts("378282246310005", PII_CC); -} -#[test] -fn credit_card_keyword_corroborates_a_bare_non_iin_run() { - // Luhn-valid, but `17` is no network's IIN — the keyword is what makes - // this a card mention rather than a machine identifier. - redacts("card 1787178633773", PII_CC); -} -#[test] -fn bare_luhn_valid_timestamp_kept() { - // A 13-digit epoch-millisecond timestamp that happens to pass Luhn - // (~10% of them do). No IIN, no keyword: not a card. - unchanged("run 1787178633773 finished"); -} -#[test] -fn credit_card_keyword_matches_serialized_key_shapes() { - // The keyword net is what the IIN table's incompleteness leans on, so it - // has to fire for the key shapes serialized payloads actually use. The - // digit run is Luhn-valid with no network's IIN (`17…`), so only the - // keyword tier can be doing the work in each of these. - for text in [ - r#"{"card_number":"1787178633773"}"#, - r#"{"cardNumber":"1787178633773"}"#, - r#"{"credit_card":"1787178633773"}"#, - r#"{"creditCard":"1787178633773"}"#, - r#"{"ccNumber":"1787178633773"}"#, - r#"{"card_no":"1787178633773"}"#, - "CARD_NUMBER=1787178633773", - "cc=1787178633773", - ] { - let out = redact_pii(text); - assert!( - out.value.contains(PII_CC), - "expected the keyword tier to corroborate: {text:?} -> {out:?}" - ); - } -} - -#[test] -fn credit_card_keyword_speaks_more_than_english() { - // Multilingual mandate: native card words corroborate too, and the - // window is wide enough that non-ASCII text does not evict them. - for text in [ - "カード 1787178633773", - "信用卡 1787178633773", - "카드 1787178633773", - "карта 1787178633773", - "tarjeta 1787178633773", - "cartão 1787178633773", - "card 😀😀😀😀😀😀😀😀 1787178633773", - ] { - let out = redact_pii(text); - assert!( - out.value.contains(PII_CC), - "expected corroboration: {text:?} -> {out:?}" - ); - } -} - -#[test] -fn credit_card_keyword_still_respects_word_boundaries() { - // `pan` must not fire inside an unrelated word: Luhn-valid non-IIN run - // next to `japan` stays untouched. - unchanged("japan 1787178633773 spans"); -} - -#[test] -fn bare_brazilian_network_pans_redact_without_keyword() { - // Elo and Hipercard are in the IIN table (the module targets Brazilian - // PII by design — it carries CPF/CNPJ), so their bare PANs corroborate - // structurally, keyword or not. - redacts("5067310000000010", PII_CC); - redacts("6062821234567890", PII_CC); -} - -#[test] -fn bare_diners_14_digit_pan_redacts() { - // Regression for the review finding: the canonical 14-digit Diners test - // PAN redacted before the corroboration gate and must keep redacting. - redacts("30569309025904", PII_CC); -} - -#[test] -fn json_envelope_luhn_valid_timestamp_kept() { - // opencompany#1201: the exact corruption — a serialized record whose - // `at_millis` passed Luhn was redacted into unparseable JSON, and the - // read side then dropped the whole record as undecodable. - unchanged( - r#"{"v":1,"record":{"cycle_id":"c1","summary":"summary 1","at_millis":1787178633773}}"#, - ); -} - -// --- IBAN --- -#[test] -fn iban_de_redacted() { - // Known test IBAN with valid mod-97. - redacts("IBAN DE89370400440532013000 ok", PII_IBAN); -} -#[test] -fn iban_invalid_kept() { - unchanged("noise DE89370400440532013001 noise"); -} - -// --- Aadhaar --- -#[test] -fn aadhaar_formatted_verhoeff_valid_redacted() { - // 234123412346 is a known Verhoeff-valid Aadhaar test number. - redacts("Aadhaar 2341 2341 2346", PII_AADHAAR); -} -#[test] -fn aadhaar_keyword_bare_redacted() { - redacts("Aadhaar: 234123412346", PII_AADHAAR); -} -#[test] -fn aadhaar_invalid_verhoeff_kept() { - unchanged("Random 2341 2341 2345 nope"); -} -#[test] -fn aadhaar_formatted_newline_separator_redacted() { - // AADHAAR_FMT_RE separates groups with `[\s-]`; a newline-separated Aadhaar - // (no keyword, no dash) must still flag the formatted class in the prefilter. - redacts("2341\n2341\n2346", PII_AADHAAR); -} - -// --- PAN-IN --- -#[test] -fn pan_in_redacted() { - redacts("PAN: ABCDE1234F", PII_PAN_IN); -} - -// --- NINO --- -#[test] -fn nino_redacted() { - redacts("NI no AB123456C", PII_NINO); -} -#[test] -fn nino_reserved_prefix_kept() { - unchanged("BG123456A"); -} - -// --- DNI / NIE --- -#[test] -fn dni_es_redacted() { - redacts("DNI 12345678Z", PII_DNI); -} -#[test] -fn dni_es_bad_letter_kept() { - unchanged("ID 12345678A code"); -} -#[test] -fn nie_es_redacted() { - redacts("NIE X1234567L", PII_DNI); -} - -// --- RRN Korea --- -#[test] -fn rrn_kr_redacted() { - redacts("주민번호 900101-1234567", PII_RRN); -} -#[test] -fn rrn_kr_bad_gender_digit_kept() { - unchanged("ref 900101-5234567 nope"); -} - -// --- Bypass resistance --- -#[test] -fn fullwidth_digits_cannot_bypass_cpf() { - // 111.444.777-35 with fullwidth digits and punctuation. - let input = "CPF: 111.444.777-35 done"; - let out = redact_pii(input); - assert!(out.value.contains(PII_CPF), "got {out:?}"); -} - -#[test] -fn zero_width_chars_cannot_bypass_ssn() { - // U+200B inserted between digits. - let input = "ssn 1\u{200B}23-4\u{200B}5-6789 done"; - let out = redact_pii(input); - assert!(out.value.contains(PII_SSN), "got {out:?}"); -} - -#[test] -fn arabic_indic_digits_normalize_for_phone() { - let input = "phone +١٥٥٥١٢٣٤٥٦٧"; - let out = redact_pii(input); - assert!(out.value.contains(PII_PHONE), "got {out:?}"); -} - -// --- Aggressive mix end-to-end --- -#[test] -fn aggressive_mixed_document() { - let input = "\ -Cliente RFC VECJ880326XK4. \ -Empresa CNPJ 11.222.333/0001-81. \ -Argentino CUIT 20-11111111-2. \ -Brasileiro CPF 111.444.777-35. \ -マイナンバー: 123456789012. \ -SSN 123-45-6789. \ -Card 4111 1111 1111 1111. \ -IBAN DE89370400440532013000. \ -PAN ABCDE1234F. \ -NI AB123456C. \ -DNI 12345678Z. \ -RRN 900101-1234567. \ -Phone +15551234567."; - let out = redact_pii(input); - for token in [ - PII_RFC, PII_CNPJ, PII_CUIT, PII_CPF, PII_MYNUM, PII_SSN, PII_CC, PII_IBAN, PII_PAN_IN, - PII_NINO, PII_DNI, PII_RRN, PII_PHONE, - ] { - assert!( - out.value.contains(token), - "missing {token} in: {}", - out.value - ); - } - assert!(out.report.pii_redactions >= 13); -} - -// --- has_likely_pii --- -#[test] -fn has_likely_pii_detects_cpf() { - assert!(has_likely_pii("user/111.444.777-35")); -} - -#[test] -fn has_likely_email_detects_email_without_changing_boundary_pii() { - assert!(has_likely_email("user/alice@example.com")); - assert!(!has_likely_pii("user/alice@example.com")); -} -#[test] -fn has_likely_pii_quiet_on_normal_text() { - assert!(!has_likely_pii("memory/global/preferences")); -} - -/// Regression: zero-padded millisecond-timestamp keys must NOT be -/// flagged as PII even when the digit run happens to satisfy Luhn. -/// `redact_pii` content scrubbing may still flag the same string — -/// `has_likely_pii` (used for boundary rejection of internal keys) -/// must stay strict to formatted/keyword PII only. -#[test] -fn has_likely_pii_ignores_bare_luhn_timestamp_keys() { - // 18-digit padded timestamps where the digit total mod 10 == 0 - // (the Luhn-passing case that previously rejected autocomplete - // KV writes and screen-intelligence document writes). - for key in [ - "accepted:000001747729035001", - "completion:000001747729035011", - "screen_intelligence_vision-1747729035001-VSCode", - ] { - assert!( - !has_likely_pii(key), - "internal key {key:?} must not be rejected as PII" - ); - } -} - -/// Strict boundary check should still reject formatted PII even though -/// it skips bare-numeric checksum patterns. -#[test] -fn has_likely_pii_still_blocks_formatted_secrets() { - assert!(has_likely_pii("ssn-123-45-6789")); - assert!(has_likely_pii("cliente-RFC-VECJ880326XK4")); - assert!(has_likely_pii("cuit-20-11111111-2")); -} - -/// Regression for Sentry TAURI-RUST-54T / GH #2848: scanner-built -/// `namespace` and `key` values containing bare-numeric phone-shaped -/// digit runs (WhatsApp group JID `-@g.us`, WhatsApp -/// broadcast `@broadcast`, US-prefixed WhatsApp 1:1 JID, -/// telegram numeric peer ID) must NOT be rejected by the boundary -/// PII check. NANP matches `\d{10,11}` with optional separators — -/// strict mode must skip it. Content scrubbing via `redact_pii` -/// continues to redact these substrings (see -/// `redact_pii_still_blurs_bare_phone_in_content` below). -#[test] -fn has_likely_pii_ignores_scanner_bare_phone_keys() { - for key in [ - // WhatsApp group JID — chat_id = "-@g.us" - "12025551234-1543890267@g.us:2026-05-30", - // WhatsApp broadcast list - "12025551234@broadcast:2026-05-30", - // WhatsApp 1:1 JID, country-coded US number (`1` + 10 digits) - "12025551234@c.us:2026-05-30", - // Same shape carried in the namespace - "whatsapp-web:12025551234@c.us", - "whatsapp-web:12025551234-1543890267@g.us", - // Telegram numeric peer_id key - "4123456789:2026-05-30", - ] { - assert!( - !has_likely_pii(key), - "scanner-built key {key:?} must not be rejected as PII" - ); - } -} - -/// Same regression but for the E.164 (`+`-prefixed) shape — iMessage -/// posts `key = format!("{chat_id}:{day}")` where `chat_id` can be -/// `+12025551234`. Strict mode must skip; content redaction stays. -#[test] -fn has_likely_pii_ignores_bare_e164_phone_keys() { - for key in [ - "+12025551234:2026-05-30", - "imessage:+12025551234", - "imessage:+12025551234:2026-05-30", - ] { - assert!( - !has_likely_pii(key), - "E.164-shaped key {key:?} must not be rejected as PII" - ); - } -} - -/// `redact_pii` (content scrubbing path — NOT the boundary check) -/// must still redact formatted NANP and E.164 phone numbers found -/// inside document bodies. False positives in the content path only -/// blur substring bytes; they do not reject the write — which is the -/// asymmetry this PR preserves vs. the boundary check. -/// -/// Note: bare 10-digit NANP runs (`2025551234` with no separators) -/// are NOT reached by `redact_pii` at all — the SCREEN fast-path -/// requires either `\d{11,}`, a separator, or `+`, so a bare 10-digit -/// run short-circuits as "no candidate". That pre-existed this PR; a -/// pinning sentinel for it lives below. -#[test] -fn redact_pii_still_blurs_formatted_and_e164_phone_in_content() { - let out = redact_pii("call me at 202-555-1234 or +12025551234"); - let n_phone = out.value.matches(PII_PHONE).count(); - assert!( - n_phone >= 2, - "redact_pii must still blur both formatted NANP and E.164 phones in content, \ - got {n_phone} PII_PHONE token(s) in: {}", - out.value - ); - assert!(out.report.pii_redactions >= 2); -} - -/// Sentinel pinning a pre-existing SCREEN limitation: a bare 10-digit -/// NANP run (`2025551234` with no separators) is short-circuited by -/// the `SCREEN` fast-path because no `SCREEN` regex matches a 10-digit -/// bare run (`\d{11,}` is the closest, but it needs 11+). This is the -/// status quo on `main` — this PR does not change it. The test exists -/// so any future widening of `SCREEN` (e.g. to catch bare NANP) trips -/// here as a deliberate review checkpoint, NOT a regression. -#[test] -fn redact_pii_does_not_reach_bare_10_digit_nanp_today() { - let out = redact_pii("call me at 2025551234 thanks"); - assert!( - !out.value.contains(PII_PHONE), - "SCREEN fast-path historically skips bare 10-digit NANP — \ - if this test fails, SCREEN was widened; revisit the boundary-check \ - behavior in `has_likely_pii` before adjusting. Got: {}", - out.value - ); -} - -#[test] -fn empty_text_is_noop() { - unchanged(""); -} - -// --- Byte prefilter: per-class positives (incl. non-Latin) --- - -/// Devanagari Aadhaar keyword must still route into the keyword-gated Aadhaar -/// path (the `आधार` needle lives in `AADHAAR_KEYWORDS`). -#[test] -fn aadhaar_devanagari_keyword_redacted() { - redacts("आधार 234123412346", PII_AADHAAR); -} - -/// Japanese My-Number keyword (kanji form) routes into the My-Number path. -#[test] -fn my_number_kanji_keyword_redacted() { - redacts("個人番号 123456789012", PII_MYNUM); -} - -/// `scan_candidates` flags the right class for representative per-class inputs. -#[test] -fn scan_flags_expected_classes() { - assert!(scan_candidates("111.444.777-35").cpf_fmt); - assert!(scan_candidates("11.222.333/0001-81").cnpj_fmt); - assert!(scan_candidates("20-11111111-2").cuit); - assert!(scan_candidates("DE89370400440532013000").iban); - assert!(scan_candidates("4111111111111111").cc); - assert!(scan_candidates("11222333000181").cnpj_bare); - assert!(scan_candidates("11144477735").cpf_bare); - assert!(scan_candidates("2341 2341 2346").aadhaar_fmt); - assert!(scan_candidates("aadhaar 234123412346").aadhaar_kw); - assert!(scan_candidates("आधार 234123412346").aadhaar_kw); - assert!(scan_candidates("12345678Z").dni); - assert!(scan_candidates("X1234567L").nie); - assert!(scan_candidates("AB123456C").nino); - assert!(scan_candidates("123-45-6789").ssn); - assert!(scan_candidates("900101-1234567").rrn); - assert!(scan_candidates("VECJ880326XK4").rfc); - assert!(scan_candidates("ABCDE1234F").pan_in); - assert!(scan_candidates("+15551234567").phone_e164); - assert!(scan_candidates("415-555-0123").phone_nanp); - assert!(scan_candidates("マイナンバー 123456789012").mynumber); - assert!(scan_candidates("My Number 123456789012").mynumber); -} - -/// Clean, PII-free text flags no class at all — the whole precise pass is -/// skipped and every precise regex stays uncompiled. -#[test] -fn scan_clean_text_flags_nothing() { - for clean in [ - "", - "just some ordinary words here", - "memory/global/preferences", - "the quick brown fox", - "https://example.com/path?q=1", - "snake_case_identifier_v2", - ] { - let cand = scan_candidates(clean); - assert!(!cand.any(), "clean text flagged a class: {clean:?}"); - } -} - -/// A bare separator-less 10-digit run must NOT flag the NANP phone class — this -/// is what preserves the documented "bare 10-digit NANP is never reached" -/// behavior even though the precise NANP regex would otherwise match it. -#[test] -fn scan_bare_10_digit_run_does_not_flag_nanp() { - assert!(!scan_candidates("call me at 2025551234 thanks").phone_nanp); -} - -/// Parity oracle: the new byte prefilter must be a SUPERSET of the legacy -/// `SCREEN` regex set. For every corpus input, if the old combined set would -/// have matched the normalized text, the new per-class scan must flag at least -/// one class — otherwise a real PII candidate would be silently dropped. -#[test] -fn prefilter_is_superset_of_legacy_screen() { - use regex::RegexSet; - - // Byte-for-byte the pattern list this PR removed from `pii.rs`. - let legacy_screen = RegexSet::new([ - r"\d{11,}", - r"\d{3}\.\d{3}\.\d{3}-\d{2}", - r"\d{2}\.\d{3}\.\d{3}/\d{4}-\d{2}", - r"\d{2}-\d{8}-\d", - r"(?i)[A-Z]{3,4}\d{6}", - r"(?:マイナンバー|個人番号|My\s?Number)", - r"\+\d{7}", - r"\(?[2-9]\d{2}\)?[\s.\-]\d{3}[\s.\-]\d{4}", - r"\d{3}-\d{2}-\d{4}", - r"\b[A-Z]{2}\d{2}[A-Z0-9]", - r"\d{4}[\s\-]\d{4}[\s\-]\d{4}", - r"(?i)aadhaar|aadhar|आधार|uidai", - r"(?i)[A-Z]{5}\d{4}[A-Z]", - r"(?i)[A-Z]{2}\d{6}[A-D]", - r"\b\d{8}[A-Z]\b", - r"(?i)[XYZ]\d{7}[A-Z]", - r"\d{6}-[1-4]\d{6}", - ]) - .expect("legacy screen"); - - let corpus = [ - // Real PII, one per class. - "CPF: 111.444.777-35.", - "Sem mascara 11144477735 ok", - "CNPJ 11.222.333/0001-81", - "contract 11222333000181 yes", - "CUIT 20-11111111-2", - "Mi RFC VECJ880326XK4 .", - "マイナンバー: 123456789012", - "個人番号 123456789012", - "My Number 123456789012", - // Whitespace-separator variants the precise regexes accept via `\s`. - "My\tNumber 123456789012", - "My\nNumber 123456789012", - "2341\n2341\n2346", - "12025551234", - "phone +15551234567", - "call 415-555-0123 thanks", - "+1 (212) 555-7890", - "ssn 123-45-6789", - "card 4111 1111 1111 1111 thanks", - "card 378282246310005 used", - "IBAN DE89370400440532013000 ok", - "Aadhaar 2341 2341 2346", - "Aadhaar: 234123412346", - "आधार 234123412346", - "uidai 234123412346", - "PAN: ABCDE1234F", - "NI no AB123456C", - "DNI 12345678Z", - "NIE X1234567L", - "주민번호 900101-1234567", - // Scanner-built / borderline identifiers. - "12025551234-1543890267@g.us:2026-05-30", - "+12025551234:2026-05-30", - "accepted:000001747729035001", - "screen_intelligence_vision-1747729035001-VSCode", - "Order 123456789012 shipped today.", - // Clean text (screen won't match; nothing to assert but exercises path). - "memory/global/preferences", - "the quick brown fox jumps", - "just some ordinary words here", - ]; - - for input in corpus { - let nview = NormalizedView::build(input); - if legacy_screen.is_match(&nview.normalized) { - assert!( - scan_candidates(&nview.normalized).any(), - "legacy SCREEN matched but new prefilter flagged nothing: {input:?}" - ); - } - } -} diff --git a/src/memory/store/safety/safety_tests.rs b/src/memory/store/safety/safety_tests.rs index 8f2c5935..01856e43 100644 --- a/src/memory/store/safety/safety_tests.rs +++ b/src/memory/store/safety/safety_tests.rs @@ -1,106 +1,41 @@ +//! The scrubbers' own tests live in `tinymemory-safety`. These pin what this +//! module adds: the engine policy and the public `safety::*` paths. + use super::*; use serde_json::json; -#[test] -fn sanitize_text_redacts_bearer_and_openai_key() { - let input = "Authorization: Bearer abcdefghijklmnop and sk-1234567890123456789012345"; - let sanitized = sanitize_text(input); - assert!(sanitized.value.contains("Bearer [REDACTED]")); - assert!(!sanitized.value.contains("sk-1234567890123456789012345")); - assert!(sanitized.report.text_redactions >= 2); -} - -#[test] -fn sanitize_text_blocks_private_key_blocks() { - let input = "-----BEGIN PRIVATE KEY-----\nabc\n-----END PRIVATE KEY-----"; - let sanitized = sanitize_text(input); - assert!(sanitized.value.contains(REDACTED_PRIVATE_KEY)); - assert!(sanitized.report.blocked_secret_hits >= 1); -} - -#[test] -fn sanitize_json_redacts_sensitive_keys_and_nested_strings() { - let input = json!({ - "token": "abc123", - "nested": { - "notes": "Bearer supersecretvalue", - "ok": "hello" - }, - "arr": ["sk-1234567890123456789012345", "safe"] - }); - - let sanitized = sanitize_json(&input); - assert_eq!(sanitized.value["token"], json!(REDACTED_SECRET)); - assert_eq!(sanitized.value["nested"]["ok"], json!("hello")); - assert!(sanitized.value["nested"]["notes"] - .as_str() - .unwrap_or_default() - .contains("[REDACTED]")); - assert!(sanitized.report.key_redactions >= 1); - assert!(sanitized.report.text_redactions >= 2); -} - -#[test] -fn sanitize_json_redacts_common_sensitive_key_variants() { - let input = json!({ - "db_password": "p@ss", - "secret_key": "abc123", - "api_secret": "def456", - "monkey": "banana" - }); - - let sanitized = sanitize_json(&input); - assert_eq!(sanitized.value["db_password"], json!(REDACTED_SECRET)); - assert_eq!(sanitized.value["secret_key"], json!(REDACTED_SECRET)); - assert_eq!(sanitized.value["api_secret"], json!(REDACTED_SECRET)); - assert_eq!(sanitized.value["monkey"], json!(REDACTED_SECRET)); - assert!(sanitized.report.key_redactions >= 4); -} - -#[test] -fn has_likely_secret_detects_common_patterns() { - assert!(has_likely_secret("api_key=abc123")); - assert!(has_likely_secret("Bearer abcdefghijklmnopqrstuvwxyz")); - assert!(has_likely_secret("xoxb-1234567890-abcdef-ghijklmnop")); - assert!(has_likely_secret("glpat-aaaaaaaaaaaaaaaaaaaa")); - assert!(has_likely_secret("SG.aaaaaaaaaaaaaaaa.bbbbbbbbbbbbbbbb")); - assert!(!has_likely_secret("I prefer rust")); -} +/// A Luhn-valid 13-digit epoch-millisecond timestamp: no network issues a `17` +/// prefix, so the engine must leave it alone. +const TIMESTAMP: &str = "1700000000004"; #[test] -fn has_likely_pii_strict_boundary_flags_formatted_national_ids() { - // The write-rejection boundary is the *strict* set: formatted national IDs - // only. Bare-numeric / phone-shaped runs and email are excluded (too many - // false positives against scanner-built identifiers); they are still - // scrubbed by content redaction. Exhaustive coverage lives in `pii`'s tests. - assert!(has_likely_pii("ssn 123-45-6789")); - assert!(has_likely_pii("CPF 111.444.777-35")); - assert!(has_likely_pii("cliente RFC VECJ880326XK4")); - assert!(!has_likely_pii("call +15551234567")); // phone: content-scrub only - assert!(!has_likely_pii("contact alice@example.com")); // email: out of scope - assert!(!has_likely_pii("just a normal note")); +fn engine_policy_leaves_bare_timestamps_alone() { + let envelope = format!("{{\"ts\": {TIMESTAMP}}}"); + assert_eq!(pii::redact_pii(&envelope).value, envelope); + assert_eq!(sanitize_text(&envelope).value, envelope); + let value = json!({ "ts": TIMESTAMP }); + assert_eq!(sanitize_json(&value).value, value); } #[test] -fn sanitize_text_scrubs_pii_after_secrets() { - let input = "Token sk-abcdefghijklmnopqrstuvwxyz; CPF 111.444.777-35; phone +15551234567"; - let sanitized = sanitize_text(input); - assert!(!sanitized.value.contains("sk-abcdefghijklmnopqrstuvwxyz")); - assert!(!sanitized.value.contains("111.444.777-35")); - assert!(!sanitized.value.contains("+15551234567")); - assert!(sanitized.report.pii_redactions >= 2); +fn engine_policy_still_redacts_real_cards_and_secrets() { + assert!(pii::redact_pii("card 4111111111111111") + .value + .contains("[REDACTED_PII_CREDIT_CARD]")); + assert!(pii::redact_pii("4111111111111111") + .value + .contains("[REDACTED_PII_CREDIT_CARD]")); + let key = format!("sk-{}", "1234567890123456789012345"); + let out = sanitize_text(&format!("key {key}")); + assert!(!out.value.contains(&key)); + assert!(out.report.text_redactions >= 1); } #[test] -fn sanitize_json_redacts_values_beyond_max_depth() { - let mut nested = json!("leaf"); - for _ in 0..(MAX_JSON_SANITIZE_DEPTH + 2) { - nested = json!({ "nested": nested }); - } - let sanitized = sanitize_json(&nested); - assert!(sanitized.report.depth_redactions >= 1); - assert!(sanitized - .value - .to_string() - .contains(&format!("\"{REDACTED_SECRET}\""))); +fn boundary_predicates_are_reexported() { + assert!(has_likely_pii("ssn-123-45-6789")); + assert!(pii::has_likely_pii("ssn-123-45-6789")); + assert!(has_likely_email("user/alice@example.com")); + assert!(!has_likely_pii("user/alice@example.com")); + assert!(has_likely_secret("Bearer abcdefghijklmnop")); } From 629371ee54f1becaee365fa45afbb5ce90e7c7ff Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Thu, 1 Oct 2026 00:24:38 +0300 Subject: [PATCH 2/3] fix(safety): update doc reference to use full module path The documentation comment for the write-rejection boundary now uses the full module path `pii::has_likely_pii` instead of the bare function name, making the cross-reference unambiguous when the module is not imported at the call site. Auto-committed-on: dragonfly --- src/memory/store/safety/mod.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/memory/store/safety/mod.rs b/src/memory/store/safety/mod.rs index 836a18b2..f0982c6a 100644 --- a/src/memory/store/safety/mod.rs +++ b/src/memory/store/safety/mod.rs @@ -10,7 +10,7 @@ //! network IIN or a nearby card keyword, so 13-digit epoch-millisecond //! timestamps in stored JSON envelopes are not corrupted (opencompany#1201). //! -//! The write-rejection boundary ([`has_likely_pii`]) stays stricter than +//! The write-rejection boundary ([`pii::has_likely_pii`]) stays stricter than //! content scrubbing: formatted national IDs are rejected, while phone/email-like //! text is scrubbed from content without rejecting every write that mentions //! them. From 6958497ed3132a7d0255debcb57d7a9b8b678303 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Thu, 1 Oct 2026 00:24:46 +0300 Subject: [PATCH 3/3] fix(docs): update cross-reference to use full path for has_likely_pii The doc comment for the safety module now links to `has_likely_pii` using its full module path instead of a relative reference, ensuring the link resolves correctly in generated documentation. Auto-committed-on: dragonfly --- src/memory/store/safety/mod.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/memory/store/safety/mod.rs b/src/memory/store/safety/mod.rs index f0982c6a..a0c9e816 100644 --- a/src/memory/store/safety/mod.rs +++ b/src/memory/store/safety/mod.rs @@ -10,7 +10,7 @@ //! network IIN or a nearby card keyword, so 13-digit epoch-millisecond //! timestamps in stored JSON envelopes are not corrupted (opencompany#1201). //! -//! The write-rejection boundary ([`pii::has_likely_pii`]) stays stricter than +//! The write-rejection boundary ([`has_likely_pii`](tinymemory_safety::has_likely_pii)) stays stricter than //! content scrubbing: formatted national IDs are rejected, while phone/email-like //! text is scrubbed from content without rejecting every write that mentions //! them.