From ca684f4c8e12c2a014d494bdd36e1d981058b902 Mon Sep 17 00:00:00 2001 From: AyushAggarwal1 Date: Sat, 26 Sep 2026 18:40:07 +0530 Subject: [PATCH 1/3] regional regex changes --- scripts/benchmark_public_datasets.py | 687 +++++++++++++++++++++ src/engine/layers.py | 95 ++- src/engine/recognizers/__init__.py | 18 + src/engine/recognizers/asia_pacific.py | 5 +- src/engine/recognizers/identifiers.py | 45 +- src/engine/recognizers/in_sg_au_kr_th.py | 2 + src/engine/recognizers/us_ca.py | 13 +- src/engine/recognizers/za_ng_ph_generic.py | 4 +- src/engine/rules.py | 50 +- tests/fixtures/accuracy_cases.json | 201 ++++++ tests/sample_dataset_builder.py | 2 +- tests/test_detector_names.py | 2 +- tests/test_policy_confidence.py | 2 +- tests/test_public_benchmark.py | 64 ++ tests/test_recognizers_de_se_fi_pl.py | 2 +- tests/test_recognizers_gb_es_it_tr.py | 14 +- tests/test_recognizers_in_sg_au_kr_th.py | 2 +- tests/test_recognizers_us_ca.py | 53 +- tests/test_recognizers_world.py | 6 +- tests/test_recognizers_za_ng_ph_generic.py | 4 +- 20 files changed, 1206 insertions(+), 65 deletions(-) create mode 100644 scripts/benchmark_public_datasets.py create mode 100644 tests/test_public_benchmark.py diff --git a/scripts/benchmark_public_datasets.py b/scripts/benchmark_public_datasets.py new file mode 100644 index 0000000..4fcfe16 --- /dev/null +++ b/scripts/benchmark_public_datasets.py @@ -0,0 +1,687 @@ +"""Benchmark the detection engine on public PII datasets. + +Datasets, cached under output/benchmarks/datasets/ (downloaded from Hugging Face, see README there): + + privy beki/privy (MIT). Synthetic API payloads as JSON, SQL, XML and HTML protocol traces, + 26 Presidio-style labels. File: test-large.json inside privy-dataset.zip (streamed, reservoir-sampled). + nemotron nvidia/Nemotron-PII (CC BY 4.0). Persona-grounded documents, US and international locales, + 55 labels. File: nemotron-pii-test.parquet (random sample). + gretel gretelai/synthetic_pii_finance_multilingual (Apache 2.0). Full-length financial documents, + 29 labels. File: gretel-english-test.parquet (whole English test split). + +Run: + + .venv/bin/python -m scripts.benchmark_public_datasets --dataset all --limit 5000 --seed 7 + .venv/bin/python -m scripts.benchmark_public_datasets --dataset privy --privy-mode json + +Scoring. Every record is one TextBlob through UnitClassifier with aggregation off and the reporting +floor at `possible`; metrics are computed afterwards at each tier (possible / likely / very_likely). +A gold span counts as detected when an engine hit of an accepted detector overlaps it (lenient span +matching). Dataset labels are classed as target (scored for recall), ambiguous (an accepted hit is fine, +a miss is not counted) or unscored (the engine has no detector for them: dates, companies, demographics). +Engine hits that overlap no gold span are false positives ("no_gold"); hits on a scored span with a +detector of another type are "wrong_type"; hits on unscored spans are left out of precision. +Gold values that fail their own checksum (Luhn, IBAN mod-97, ABA routing, SSN structure) are reported +separately: LLM-generated datasets contain many invalid numbers, and `recall_on_valid` is the fair column. + +Privy has a second mode (`--privy-mode json`): JSON payloads are fed as documents through +document_record, so field names drive the verdict as they do for a database or a JSON file, and gold +spans are matched by value instead of offset. + +Reports hold labels, detector names and counts only, never values. +""" +import argparse +import ast +import json +import os +import random +import re +import sys +import time +import zipfile +from collections import Counter, defaultdict +from pathlib import Path +from typing import Any, Dict, Iterator, List, Optional, Set, Tuple + +ROOT = Path(__file__).resolve().parents[1] +DATA_DIR = ROOT / "output" / "benchmarks" / "datasets" +REPORT_DIR = ROOT / "output" / "benchmarks" +MAPPING_FILES = ("findings-mapping.json", "findings-mapping-v2.json", "findings-mapping-v1.json") + +TIERS = ("possible", "likely", "very_likely") +RANK = {tier: index for index, tier in enumerate(TIERS)} + +# Detectors that never count as false positives: document-level keyword verdicts and disabled types. +IGNORED_DETECTORS = {"Healthcare Data Detection", "URL", "UUID"} + + +# ---------------------------------------------------------------------------- detector groups +def _load_categories() -> Dict[str, str]: + for name in MAPPING_FILES: + path = ROOT / "fixtures" / name + if path.exists(): + data = json.loads(path.read_text(encoding="utf-8")) + data = data[0] if isinstance(data, list) else data + return {detector: entry.get("category", "") for detector, entry in data.items()} + return {} + + +CATEGORY = _load_categories() +SECRET_DETECTORS = {d for d, c in CATEGORY.items() if c in ("Credentials and Secrets", "Entropy-Based Secret Detection")} +REGIONAL_DETECTORS = {d for d, c in CATEGORY.items() if c == "Regional Compliance"} + +GROUPS: Dict[str, Set[str]] = { + "name": {"PII.PersonName"}, + "email": {"Email"}, + "phone": {"Phone Number", "ZA_MOBILE_NUMBER", "ZA_TELEPHONE_NUMBER", "PH_MOBILE_NUMBER"}, + "card": {"Credit Card"}, + "iban": {"IBAN"}, + "bank": {"Bank Account", "IBAN", "ABA_ROUTING_NUMBER", "GB_SORT_CODE", "AU_BSB"}, + "routing": {"ABA_ROUTING_NUMBER", "Bank Account"}, + "swift": {"SWIFT/BIC"}, + "ssn": {"US SSN"}, + "itin": {"US_ITIN", "US SSN"}, + "passport": {"US_PASSPORT", "UK_PASSPORT", "PASSPORT_MRZ", "IN PASSPORT", "DE_PASSPORT", "ES_PASSPORT", + "IT_PASSPORT", "FR_PASSPORT", "JP_PASSPORT", "KR_PASSPORT", "PH_PASSPORT", "ZA_PASSPORT"}, + "driver": {"US_DRIVER_LICENSE", "UK_DRIVING_LICENCE", "IT_DRIVER_LICENSE", "KR_DRIVER_LICENSE", + "ZA_DRIVER_LICENSE", "DE_FUEHRERSCHEIN"}, + "ip": {"PII.IPAddress"}, + "mac": {"MAC_ADDRESS"}, + "imei": {"IMEI"}, + "coordinate": {"GEO_COORDINATES"}, + "address": {"Address"}, + "dob": {"Date of Birth"}, + "secret": set(SECRET_DETECTORS) or {"Password Pattern", "API Key", "Bearer Token", "High Entropy Secret"}, + "mrn": {"MEDICAL_RECORD_NUMBER"}, + "health_member": {"US_HEALTH_INSURANCE_MEMBER_ID", "US_MBI"}, + "vin": {"VIN"}, + "plate": {"UK_VEHICLE_REGISTRATION", "TR_LICENSE_PLATE", "DE_KFZ", "ZA_LICENSE_PLATE", + "IN_VEHICLE_REGISTRATION", "NG_VEHICLE_REGISTRATION"}, + "user": {"PII.UserIdentifier"}, + "postcode": {"UK_POSTCODE", "CA_POSTAL_CODE", "DE_PLZ", "Address"}, + "regional": set(REGIONAL_DETECTORS), + "device": {"IMEI", "ICCID", "MAC_ADDRESS"}, + "licence": {"MEDICAL_LICENSE", "US_NPI"} | {"US_DRIVER_LICENSE", "UK_DRIVING_LICENCE"}, + "url": {"URL"}, + "uuid": {"UUID"}, +} +GROUPS["financial"] = GROUPS["bank"] | GROUPS["swift"] | GROUPS["card"] | GROUPS["iban"] +GROUPS.update({ + "in_pan": {"IN PAN"}, "in_aadhaar": {"IN Aadhaar"}, "in_upi": {"IN_UPI_ID"}, "in_ifsc": {"IN_IFSC"}, + "in_voter": {"IN VOTER ID"}, "in_gst": {"IN GST"}, "in_plate": {"IN_VEHICLE_REGISTRATION"}, + "medical_id": {"MEDICAL_RECORD_NUMBER", "US_HEALTH_INSURANCE_MEMBER_ID", "US_MBI"} | set(REGIONAL_DETECTORS), +}) + +# Which group's values carry a checksum the benchmark can verify. +VALIDATED_GROUPS = {"card": "card", "iban": "iban", "routing": "routing", "ssn": "ssn", "imei": "imei"} + +# ---------------------------------------------------------------------------- dataset label specs +# label -> (class, group). class: target (scored) | ambiguous (accepted hit fine, miss not counted). +# Labels missing from the spec are unscored. Privy's "O" spans are negatives and are dropped from gold. +PRIVY_SPEC = { + "PERSON": ("target", "name"), "EMAIL_ADDRESS": ("target", "email"), "PHONE_NUMBER": ("target", "phone"), + "CREDIT_CARD": ("target", "card"), "IBAN_CODE": ("target", "iban"), "US_BANK_NUMBER": ("target", "bank"), + "US_SSN": ("target", "ssn"), "US_ITIN": ("target", "itin"), "US_PASSPORT": ("target", "passport"), + "US_DRIVER_LICENSE": ("target", "driver"), "IP_ADDRESS": ("target", "ip"), "MAC_ADDRESS": ("target", "mac"), + "IMEI": ("target", "imei"), "COORDINATE": ("target", "coordinate"), "PASSWORD": ("target", "secret"), + "FINANCIAL": ("ambiguous", "financial"), "LOCATION": ("ambiguous", "address"), "DATE_TIME": ("ambiguous", "dob"), + "URL": ("ambiguous", "url"), "US_LICENSE_PLATE": ("ambiguous", "plate"), +} +NEMOTRON_SPEC = { + "first_name": ("target", "name"), "last_name": ("target", "name"), "email": ("target", "email"), + "phone_number": ("target", "phone"), "fax_number": ("target", "phone"), "street_address": ("target", "address"), + "date_of_birth": ("target", "dob"), "credit_debit_card": ("target", "card"), "ssn": ("target", "ssn"), + "bank_routing_number": ("target", "routing"), "account_number": ("target", "bank"), "swift_bic": ("target", "swift"), + "ipv4": ("target", "ip"), "ipv6": ("target", "ip"), "mac_address": ("target", "mac"), + "coordinate": ("target", "coordinate"), "password": ("target", "secret"), "api_key": ("target", "secret"), + "http_cookie": ("target", "secret"), "medical_record_number": ("target", "mrn"), + "health_plan_beneficiary_number": ("target", "health_member"), "tax_id": ("target", "regional"), + "vehicle_identifier": ("ambiguous", "vin"), "license_plate": ("ambiguous", "plate"), + "certificate_license_number": ("ambiguous", "licence"), "user_name": ("ambiguous", "user"), + "pin": ("ambiguous", "secret"), "device_identifier": ("ambiguous", "device"), "unique_id": ("ambiguous", "uuid"), + "url": ("ambiguous", "url"), "postcode": ("ambiguous", "postcode"), "cvv": ("ambiguous", "secret"), +} +GRETEL_SPEC = { + "name": ("target", "name"), "first_name": ("target", "name"), "last_name": ("target", "name"), + "email": ("target", "email"), "phone_number": ("target", "phone"), "street_address": ("target", "address"), + "date_of_birth": ("target", "dob"), "credit_card_number": ("target", "card"), "ssn": ("target", "ssn"), + "bank_routing_number": ("target", "routing"), "bban": ("target", "bank"), "iban": ("target", "iban"), + "swift_bic_code": ("target", "swift"), "ipv4": ("target", "ip"), "ipv6": ("target", "ip"), + "local_latlng": ("target", "coordinate"), "passport_number": ("target", "passport"), + "driver_license_number": ("target", "driver"), "password": ("target", "secret"), "api_key": ("target", "secret"), + "account_pin": ("ambiguous", "secret"), "user_name": ("ambiguous", "user"), + "credit_card_security_code": ("ambiguous", "secret"), +} +GRETEL_GENERAL_SPEC = { + **NEMOTRON_SPEC, + "name": ("target", "name"), "address": ("target", "address"), "credit_card_number": ("target", "card"), + "national_id": ("target", "regional"), "unique_identifier": ("ambiguous", "uuid"), +} +YLEMIS_SPEC = { + "person": ("target", "name"), "indian_phone": ("target", "phone"), "address": ("target", "address"), + "pan": ("target", "in_pan"), "aadhaar": ("target", "in_aadhaar"), "email": ("target", "email"), + "upi_id": ("target", "in_upi"), "ifsc": ("target", "in_ifsc"), "voter_id": ("target", "in_voter"), + "gstin": ("target", "in_gst"), "indian_passport": ("target", "passport"), + "vehicle_registration": ("target", "in_plate"), "account_id": ("ambiguous", "bank"), # policy_1467545_783-style ids, not account numbers + "medical_id": ("ambiguous", "medical_id"), "location": ("ambiguous", "address"), +} +SPECS = {"privy": PRIVY_SPEC, "nemotron": NEMOTRON_SPEC, "gretel": GRETEL_SPEC, "gretel-general": GRETEL_GENERAL_SPEC, "ylemis": YLEMIS_SPEC} +LICENCES = { + "privy": "beki/privy, MIT", "nemotron": "nvidia/Nemotron-PII, CC BY 4.0", + "gretel": "gretelai/synthetic_pii_finance_multilingual, Apache 2.0", + "gretel-general": "gretelai/gretel-pii-masking-en-v1, Apache 2.0", + "ylemis": "Pranshurs/ylemis-india-pii-benchmark, CC0 1.0", +} +# Ylemis slices whose text is Latin-script; the Indic-script slices are out of scope for an English engine +YLEMIS_LANGUAGES = ("en-IN", "hi-Latn", "code-mixed-IN", "mixed-script-IN", "ocr-en-IN", "llm-prompt-IN") + + +# ---------------------------------------------------------------------------- validators +def _digits(value: str) -> str: + return re.sub(r"\D", "", value or "") + + +def luhn_ok(digits: str) -> bool: + total, parity = 0, len(digits) % 2 + for index, char in enumerate(digits): + d = ord(char) - 48 + if index % 2 == parity: + d *= 2 + if d > 9: + d -= 9 + total += d + return total % 10 == 0 + + +def card_valid(value: str) -> bool: + d = _digits(value) + return 13 <= len(d) <= 19 and luhn_ok(d) + + +def imei_valid(value: str) -> bool: + d = _digits(value) + return len(d) == 15 and luhn_ok(d) + + +def iban_valid(value: str) -> bool: + s = re.sub(r"\s+", "", value or "").upper() + if not re.fullmatch(r"[A-Z]{2}\d{2}[A-Z0-9]{11,30}", s): + return False + rearranged = s[4:] + s[:4] + number = "".join(str(ord(c) - 55) if c.isalpha() else c for c in rearranged) + return int(number) % 97 == 1 + + +def routing_valid(value: str) -> bool: + d = _digits(value) + if len(d) != 9: + return False + n = [int(c) for c in d] + return (3 * (n[0] + n[3] + n[6]) + 7 * (n[1] + n[4] + n[7]) + (n[2] + n[5] + n[8])) % 10 == 0 + + +def ssn_valid(value: str) -> bool: + d = _digits(value) + if len(d) != 9: + return False + area, group, serial = int(d[:3]), int(d[3:5]), int(d[5:]) + return area not in (0, 666) and area < 900 and group != 0 and serial != 0 + + +VALIDATORS = {"card": card_valid, "iban": iban_valid, "routing": routing_valid, "ssn": ssn_valid, "imei": imei_valid} + + +# ---------------------------------------------------------------------------- loaders +class Record: + __slots__ = ("id", "text", "gold", "meta") + + def __init__(self, rid: str, text: str, gold: List[Tuple[str, int, int, str]], meta: Dict[str, Any]): + self.id, self.text, self.gold, self.meta = rid, text, gold, meta + + +def _parse_spans(raw: Any) -> List[Dict[str, Any]]: + if raw is None: + return [] + if not isinstance(raw, str): + return list(raw) + try: + return json.loads(raw) + except ValueError: + return ast.literal_eval(raw) + + +def _payload_format(text: str) -> str: + head = text.lstrip()[:8] + if head[:1] in "{[": + return "json" + if head[:1] == "<": + return "xml/html" + if head[:6].upper() in ("INSERT", "SELECT", "UPDATE"): + return "sql" + return "other" + + +def load_privy(limit: Optional[int], seed: int) -> List[Record]: + """Reservoir-samples test-large.json straight out of the zip; never extracts 700 MB to disk.""" + import ijson + + archive = DATA_DIR / "privy-dataset.zip" + rng = random.Random(seed) + sample: List[Record] = [] + seen = 0 + with zipfile.ZipFile(archive).open("test-large.json") as handle: + for item in ijson.items(handle, "item"): + text = item["full_text"] + gold = [ + (s["entity_type"], int(s["start_position"]), int(s["end_position"]), s.get("entity_value") or text[s["start_position"]:s["end_position"]]) + for s in item.get("spans", []) if s["entity_type"] != "O" + ] + record = Record(f"privy:{item.get('template_id')}:{seen}", text, gold, {"format": _payload_format(text)}) + seen += 1 + if limit is None: + sample.append(record) + elif len(sample) < limit: + sample.append(record) + else: + slot = rng.randrange(seen) + if slot < limit: + sample[slot] = record + return sample + + +def load_nemotron(limit: Optional[int], seed: int) -> List[Record]: + import pandas as pd + + frame = pd.read_parquet(DATA_DIR / "nemotron-pii-test.parquet", columns=["uid", "text", "spans", "locale", "document_format"]) + if limit is not None and limit < len(frame): + frame = frame.sample(n=limit, random_state=seed) + records = [] + for row in frame.itertuples(index=False): + gold = [(s["label"], int(s["start"]), int(s["end"]), row.text[int(s["start"]):int(s["end"])]) for s in _parse_spans(row.spans)] + records.append(Record(f"nemotron:{row.uid}", row.text, gold, {"locale": row.locale, "format": row.document_format})) + return records + + +def load_gretel(limit: Optional[int], seed: int) -> List[Record]: + import pandas as pd + + frame = pd.read_parquet(DATA_DIR / "gretel-english-test.parquet", columns=["generated_text", "pii_spans", "document_type"]) + if limit is not None and limit < len(frame): + frame = frame.sample(n=limit, random_state=seed) + records = [] + for position, row in enumerate(frame.itertuples(index=False)): + text = row.generated_text + gold = [(s["label"], int(s["start"]), int(s["end"]), text[int(s["start"]):int(s["end"])]) for s in _parse_spans(row.pii_spans)] + records.append(Record(f"gretel:{position}", text, gold, {"document_type": row.document_type})) + return records + + +def load_gretel_general(limit: Optional[int], seed: int) -> List[Record]: + """gretel-pii-masking-en-v1 lists entity values without offsets: every verbatim occurrence becomes a gold span.""" + import pandas as pd + + frame = pd.read_parquet(DATA_DIR / "gretel-general-test.parquet", columns=["uid", "text", "entities", "domain", "document_type"]) + if limit is not None and limit < len(frame): + frame = frame.sample(n=limit, random_state=seed) + records = [] + for row in frame.itertuples(index=False): + text = row.text + gold = [] + for entity in _parse_spans(row.entities): + value = entity.get("entity") or "" + if not value: + continue + position = text.find(value) + while position >= 0: + for label in entity.get("types", []): + gold.append((label, position, position + len(value), value)) + position = text.find(value, position + len(value)) + records.append(Record(f"gretel-general:{row.uid}", text, gold, {"domain": row.domain, "document_type": row.document_type})) + return records + + +def load_ylemis(limit: Optional[int], seed: int) -> List[Record]: + """Ylemis India-PII benchmark: Latin-script slices only, hard negatives and benign rows included.""" + rows = [] + with (DATA_DIR / "ylemis-india-benchmark.jsonl").open(encoding="utf-8") as handle: + for line in handle: + row = json.loads(line) + if row.get("language") in YLEMIS_LANGUAGES: + rows.append(row) + if limit is not None and limit < len(rows): + rows = random.Random(seed).sample(rows, limit) + records = [] + for row in rows: + text = row["text"] + gold = [(e["type"], int(e["start"]), int(e["end"]), text[int(e["start"]):int(e["end"])]) for e in row.get("entities", [])] + records.append(Record(f"ylemis:{row['id']}", text, gold, {"language": row.get("language"), "category": row.get("category")})) + return records + + +LOADERS = {"privy": load_privy, "nemotron": load_nemotron, "gretel": load_gretel, "gretel-general": load_gretel_general, "ylemis": load_ylemis} + + +# ---------------------------------------------------------------------------- engine +def build_engine(args): + if args.ner_model: + os.environ["NER_MODEL"] = args.ner_model + os.environ["NER_ENABLED"] = "false" if args.no_ner else "true" + sys.path.insert(0, str(ROOT)) + from src.engine.detector import DetectionEngine # noqa: WPS433 (after env is set) + + config = { + "aggregation_threshold": 0, + "min_confidence": "possible", + "enabled_regions": [r.strip().upper() for r in args.regions.split(",") if r.strip()], + "report_private_ips": True, # the datasets label every IP address, private ranges included + "ner": not args.no_ner, + } + return DetectionEngine(config), config + + +def scan_text(engine, config, record: Record) -> List[Dict[str, Any]]: + from src.pipeline import TextBlob, UnitClassifier + + classifier = UnitClassifier(engine, record.id, config=config) + classifier.feed(TextBlob(record.text, location="text", locate=lambda start, end: f"{start}:{end}")) + hits = [] + for finding in classifier.finish(): + match = re.fullmatch(r"(\d+):(\d+)", str(finding.get("location", ""))) + if not match: + continue # document-level verdicts carry no span + hits.append({ + "detector": finding["detector"], "start": int(match.group(1)), "end": int(match.group(2)), + "confidence": finding["confidence"], "value": str(finding.get("value", "")), + }) + return hits + + +def scan_json_document(engine, config, record: Record) -> Optional[List[Dict[str, Any]]]: + """Privy JSON mode: the payload as a document, so field names count. None when it is not a JSON object.""" + from src.pipeline import UnitClassifier, document_record + + try: + document = json.loads(record.text) + except ValueError: + return None + if not isinstance(document, dict): + return None + classifier = UnitClassifier(engine, record.id, config=config) + classifier.feed(document_record(document, lambda path: path)) + return [ + {"detector": f["detector"], "start": -1, "end": -1, "confidence": f["confidence"], "value": str(f.get("value", ""))} + for f in classifier.finish() + ] + + +# ---------------------------------------------------------------------------- scoring +def _overlaps(hit: Dict[str, Any], start: int, end: int) -> bool: + return hit["start"] < end and hit["end"] > start + + +def _value_match(hit_value: str, gold_value: str) -> bool: + a, b = hit_value.strip().strip("\"'"), gold_value.strip().strip("\"'") + return bool(a) and bool(b) and (a == b or a in b or b in a) + + +def score(records: List[Record], results: List[List[Dict[str, Any]]], spec: Dict[str, Tuple[str, str]], tier: str, by_value: bool) -> Dict[str, Any]: + per_label: Dict[str, Counter] = defaultdict(Counter) + per_group: Dict[str, Counter] = defaultdict(Counter) + fp_by_detector: Counter = Counter() + fp_kind: Counter = Counter() + unscored_labels: Counter = Counter() + accepted_hits = 0 + + for record, hits in zip(records, results): + live = [h for h in hits if RANK[h["confidence"]] >= RANK[tier] and h["detector"] not in IGNORED_DETECTORS] + matched_hits: Set[int] = set() + for label, start, end, value in record.gold: + klass, group = spec.get(label, ("unscored", None)) + if klass == "unscored": + unscored_labels[label] += 1 + continue + accepted = GROUPS[group] + found = None + for index, hit in enumerate(live): + near = _value_match(hit["value"], value) if by_value else _overlaps(hit, start, end) + if near and hit["detector"] in accepted: + found = index + break + if found is not None: + matched_hits.add(found) + if klass != "target": + continue + counters = per_label[label] + counters["gold"] += 1 + counters["tp" if found is not None else "fn"] += 1 + if group == "email" and re.search(r"@(?:[\w.-]*\.)?example\.(?:com|org|net)$", value.strip().lower()): + counters["demo_domain_gold"] += 1 # the engine skips example.* addresses on purpose + validator = VALIDATORS.get(VALIDATED_GROUPS.get(group, "")) + if validator is not None: + if validator(value): + counters["gold_valid"] += 1 + counters["tp_valid" if found is not None else "fn_valid"] += 1 + else: + counters["gold_invalid"] += 1 + per_group[group]["gold"] += 1 + per_group[group]["tp" if found is not None else "fn"] += 1 + + for index, hit in enumerate(live): + if index in matched_hits: + accepted_hits += 1 + continue + touching = [] + for label, start, end, value in record.gold: + near = _value_match(hit["value"], value) if by_value else _overlaps(hit, start, end) + if near: + touching.append(spec.get(label, ("unscored", None))[0]) + if not touching: + fp_kind["no_gold"] += 1 + fp_by_detector[hit["detector"]] += 1 + elif all(k == "unscored" for k in touching): + fp_kind["on_unscored_span"] += 1 # not counted against precision + else: + fp_kind["wrong_type"] += 1 + fp_by_detector[hit["detector"]] += 1 + + tp = sum(c["tp"] for c in per_label.values()) + fn = sum(c["fn"] for c in per_label.values()) + fp = fp_kind["no_gold"] + fp_kind["wrong_type"] + precision = accepted_hits / (accepted_hits + fp) if accepted_hits + fp else None + recall = tp / (tp + fn) if tp + fn else None + f1 = (2 * precision * recall / (precision + recall)) if precision and recall else None + + def label_row(counters: Counter) -> Dict[str, Any]: + row = {"gold": counters["gold"], "tp": counters["tp"], "fn": counters["fn"], + "recall": round(counters["tp"] / counters["gold"], 3) if counters["gold"] else None} + if counters["demo_domain_gold"]: + row["demo_domain_gold"] = counters["demo_domain_gold"] + if counters["gold_valid"] or counters["gold_invalid"]: + row["gold_valid"] = counters["gold_valid"] + row["gold_invalid"] = counters["gold_invalid"] + row["recall_on_valid"] = round(counters["tp_valid"] / counters["gold_valid"], 3) if counters["gold_valid"] else None + return row + + return { + "tier": tier, + "summary": { + "precision": round(precision, 3) if precision is not None else None, + "recall": round(recall, 3) if recall is not None else None, + "f1": round(f1, 3) if f1 is not None else None, + "accepted_hits": accepted_hits, "false_positives": fp, "gold_target_spans": tp + fn, + "fp_no_gold": fp_kind["no_gold"], "fp_wrong_type": fp_kind["wrong_type"], + "hits_on_unscored_spans": fp_kind["on_unscored_span"], + }, + "per_label": {label: label_row(c) for label, c in sorted(per_label.items())}, + "per_group": {group: {"gold": c["gold"], "recall": round(c["tp"] / c["gold"], 3) if c["gold"] else None} + for group, c in sorted(per_group.items())}, + "false_positives_by_detector": dict(fp_by_detector.most_common(20)), + "unscored_label_spans": dict(unscored_labels.most_common()), + } + + +# ---------------------------------------------------------------------------- driver +def run(dataset: str, args, engine, config) -> Dict[str, Any]: + spec = SPECS[dataset] + started = time.time() + records = LOADERS[dataset](args.limit, args.seed) + by_value = dataset == "privy" and args.privy_mode == "json" + if by_value: + records = [r for r in records if r.meta.get("format") == "json"] + print(f"[{dataset}] {len(records)} records loaded in {time.time() - started:.0f}s", file=sys.stderr, flush=True) + + results: List[List[Dict[str, Any]]] = [] + kept: List[Record] = [] + started = time.time() + for index, record in enumerate(records, 1): + hits = scan_json_document(engine, config, record) if by_value else scan_text(engine, config, record) + if hits is None: + continue + kept.append(record) + results.append(hits) + if index % 250 == 0: + print(f"[{dataset}] {index}/{len(records)} scanned, {time.time() - started:.0f}s", file=sys.stderr, flush=True) + seconds = time.time() - started + if args.save_hits: + suffix = "-json" if by_value else "" + with (args.output_dir / f"{dataset}{suffix}-hits.jsonl").open("w", encoding="utf-8") as handle: + for record, hits in zip(kept, results): + handle.write(json.dumps({ + "id": record.id, "meta": record.meta, + "gold": [(label, start, end, value if by_value else "") for label, start, end, value in record.gold], + "hits": [{k: h[k] for k in ("detector", "start", "end", "confidence", *(("value",) if by_value else ()))} for h in hits], + }) + "\n") + + if args.show_misses: + shown = 0 + for record, hits in zip(kept, results): + for label, start, end, value in record.gold: + klass, group = spec.get(label, ("unscored", None)) + if klass != "target": + continue + ok = any((_value_match(h["value"], value) if by_value else _overlaps(h, start, end)) and h["detector"] in GROUPS[group] for h in hits) + if not ok and shown < args.show_misses: + shown += 1 + context = record.text[max(0, start - 40):end + 40].replace("\n", " ") + print(f" MISS {label:<28} {value!r:<32} ...{context}...", file=sys.stderr) + + report = { + "dataset": dataset, "licence": LICENCES[dataset], "mode": "json-document" if by_value else "text", + "records": len(kept), "limit": args.limit, "seed": args.seed, "seconds": round(seconds, 1), + "config": {**config, "ner_model": os.environ.get("NER_MODEL") or "default"}, + "matching": "value" if by_value else "lenient span overlap", + "tiers": {tier: score(kept, results, spec, tier, by_value) for tier in TIERS}, + } + if dataset == "privy" and not by_value: + report["formats"] = dict(Counter(r.meta.get("format") for r in kept)) + return report + + +def markdown_summary(reports: List[Dict[str, Any]]) -> str: + lines = ["| Dataset | Mode | Records | Tier | Precision | Recall | F1 | Accepted hits | False positives | Gold spans |", "| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |"] + for report in reports: + for tier in TIERS: + s = report["tiers"][tier]["summary"] + lines.append(f"| {report['dataset']} | {report['mode']} | {report['records']} | {tier} | {s['precision']} | {s['recall']} | {s['f1']} | {s['accepted_hits']} | {s['false_positives']} | {s['gold_target_spans']} |") + lines.append("") + for report in reports: + likely = report["tiers"]["likely"] + lines.append(f"### {report['dataset']} ({report['mode']}), per label at `likely`") + lines.append("") + lines.append("| Label | Gold | Recall | Valid gold | Recall on valid |") + lines.append("| --- | --- | --- | --- | --- |") + for label, row in likely["per_label"].items(): + lines.append(f"| {label} | {row['gold']} | {row['recall']} | {row.get('gold_valid', '')} | {row.get('recall_on_valid', '')} |") + lines.append("") + lines.append("False positives by detector at `likely`: " + ", ".join(f"{d} {n}" for d, n in likely["false_positives_by_detector"].items())) + lines.append("") + return "\n".join(lines) + + +def rescore(path: Path, args) -> Dict[str, Any]: + """Re-score a saved hits file with the current label specs (no engine run).""" + name = path.name.replace("-hits.jsonl", "") + dataset, by_value = name.replace("-json", ""), name.endswith("-json") + records, results = [], [] + with path.open(encoding="utf-8") as handle: + for line in handle: + row = json.loads(line) + records.append(Record(row["id"], "", [tuple(g) for g in row["gold"]], row["meta"])) + results.append([{**h, "value": h.get("value", "")} for h in row["hits"]]) + return { + "dataset": dataset, "licence": LICENCES[dataset], "mode": "json-document" if by_value else "text", + "records": len(records), "rescored_from": str(path), "matching": "value" if by_value else "lenient span overlap", + "tiers": {tier: score(records, results, SPECS[dataset], tier, by_value) for tier in TIERS}, + } + + +def summarize(output_dir: Path) -> Path: + reports = [] + for path in sorted(output_dir.glob("*.json")): + if path.name.endswith(("-hits.jsonl", "-rescored.json")): + continue + data = json.loads(path.read_text(encoding="utf-8")) + if "tiers" in data: + reports.append(data) + lines = ["# Public-dataset benchmark results", "", + "Engine: src/engine + src/pipeline, aggregation off, NER model per report. Matching: lenient span overlap " + "(value match in JSON-document mode). Precision counts accepted hits against false positives of kind no_gold and " + "wrong_type; recall is over target labels only. `recall_on_valid` is recall on gold values that pass their own " + "checksum, the fair column for LLM-generated numbers.", ""] + lines += markdown_summary(reports).splitlines() + lines += ["", "## Runs", ""] + for report in reports: + lines.append(f"- {report['dataset']} ({report['mode']}): {report['records']} records, seed {report.get('seed')}, " + f"{report.get('seconds', '?')} s, NER {report.get('config', {}).get('ner_model', '?')}, licence {report['licence']}") + out = output_dir / "RESULTS.md" + out.write_text("\n".join(lines) + "\n", encoding="utf-8") + return out + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--dataset", default="all", choices=["all", "privy", "nemotron", "gretel", "gretel-general", "ylemis"]) + parser.add_argument("--privy-mode", default="text", choices=["text", "json"]) + parser.add_argument("--limit", type=int, default=5000, help="records per dataset (Gretel's English test split is smaller)") + parser.add_argument("--seed", type=int, default=7) + parser.add_argument("--regions", default="US,IN,GB", help="country packs, the product default") + parser.add_argument("--ner-model", default="", help="en_core_web_trf (default when installed) or en_core_web_sm") + parser.add_argument("--no-ner", action="store_true") + parser.add_argument("--show-misses", type=int, default=0, help="print this many missed target spans per dataset to stderr") + parser.add_argument("--output-dir", type=Path, default=REPORT_DIR) + parser.add_argument("--save-hits", action="store_true", default=True, help="write -hits.jsonl (spans and detectors, no values) for --rescore") + parser.add_argument("--rescore", type=Path, help="re-score a saved -hits.jsonl with the current label specs instead of running the engine") + parser.add_argument("--summarize", action="store_true", help="write RESULTS.md from every report in --output-dir and exit") + args = parser.parse_args() + if args.summarize: + print(f"results: {summarize(args.output_dir)}") + return + if args.rescore: + report = rescore(args.rescore, args) + for tier in TIERS: + print(f"[{report['dataset']}] {tier:<12} {json.dumps(report['tiers'][tier]['summary'])}", flush=True) + out = args.rescore.with_name(args.rescore.name.replace("-hits.jsonl", "-rescored.json")) + out.write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8") + print(f"report: {out}", flush=True) + return + + engine, config = build_engine(args) + datasets = ["privy", "nemotron", "gretel", "gretel-general", "ylemis"] if args.dataset == "all" else [args.dataset] + args.output_dir.mkdir(parents=True, exist_ok=True) + reports = [] + for dataset in datasets: + report = run(dataset, args, engine, config) + suffix = "-json" if dataset == "privy" and args.privy_mode == "json" else "" + path = args.output_dir / f"{dataset}{suffix}.json" + path.write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8") + reports.append(report) + for tier in TIERS: + print(f"[{dataset}{suffix}] {tier:<12} {json.dumps(report['tiers'][tier]['summary'])}", flush=True) + print(f"[{dataset}{suffix}] report: {path}", flush=True) + summary = args.output_dir / ("summary" + ("-privy-json" if args.privy_mode == "json" and args.dataset == "privy" else "") + ".md") + summary.write_text(markdown_summary(reports), encoding="utf-8") + print(f"summary: {summary}", flush=True) + + +if __name__ == "__main__": + main() diff --git a/src/engine/layers.py b/src/engine/layers.py index 2d8b389..2347a42 100644 --- a/src/engine/layers.py +++ b/src/engine/layers.py @@ -205,10 +205,29 @@ def validate_email(email: str) -> Optional[str]: LOOSE_PHONE_RE = re.compile(r"\+?\(?\d[\d\s().\-/]{5,22}\d") -DOB_REGEX = re.compile(r"\b(19|20)\d{2}[-/](0[1-9]|1[0-2])[-/](0[1-9]|[12]\d|3[01])\b") +_MONTHS = r"(?:jan|feb|mar|apr|may|jun|jul|aug|sep|sept|oct|nov|dec)[a-z]*\.?" +DOB_REGEX = re.compile( + r"\b(?:(?:19|20)\d{2}[-/.](?:0[1-9]|1[0-2])[-/.](?:0[1-9]|[12]\d|3[01])" # 1985-08-12 + r"|(?:0?[1-9]|[12]\d|3[01])[-/.](?:0?[1-9]|1[0-2])[-/.](?:19|20)\d{2}" # 12/08/1985, 8-12-1985 + r"|(?:0?[1-9]|[12]\d|3[01])(?:st|nd|rd|th)?\s+" + _MONTHS + r",?\s+(?:19|20)\d{2}" # 12 August 1985 + r"|" + _MONTHS + r"\s+(?:0?[1-9]|[12]\d|3[01])(?:st|nd|rd|th)?,?\s+(?:19|20)\d{2})\b", # August 12, 1985 + re.IGNORECASE, +) +_DOB_YEAR_RE = re.compile(r"(?:19|20)\d{2}") DOB_MIN_YEAR = 1930 DOB_MAX_YEAR = 2012 + +def _dob_year(value: str) -> Optional[int]: + """The four-digit year of a date match, or None for an implausible numeric day/month pair.""" + numeric = re.fullmatch(r"(\d{1,2})[-/.](\d{1,2})[-/.](\d{4})", value) + if numeric: + a, b = int(numeric.group(1)), int(numeric.group(2)) + if not (1 <= a <= 31 and 1 <= b <= 31 and min(a, b) <= 12): + return None + found = _DOB_YEAR_RE.search(value) + return int(found.group(0)) if found else None + # Street addresses: " [, more]" - the street # type must be a whole word, so 'broadcast', 'roadmap' and 'BlockRootUser' never trigger STREET_TYPES = ( @@ -276,9 +295,20 @@ def looks_like_person_name(value: str) -> bool: return False if v.lower() in {"null", "none", "test", "admin", "user", "unknown", "n/a", "na", "root", "guest", "anonymous", "default", "system", "name", "desc", "string", "value", "true", "false", "undefined"}: return False + if ORGANISATION_WORD_RE.search(v): # "Global Trust Bank", "Acme Holdings Ltd" are organisations, not people + return False return True +ORGANISATION_WORD_RE = re.compile( + r"(?:^|\s)(?:bank|trust|holdings?|group|ltd|limited|inc|llc|llp|plc|corp|corporation|company|co|gmbh|ag|sa|pvt|pty|" + r"enterprises?|industries|partners|associates|solutions|services|systems|technologies|labs|motors|farms?|foundation|" + r"institute|university|college|hospital|clinic|insurance|capital|investments?|ventures|fund|agency|council|ministry|" + r"department|authority|association|federation|union|church|school|store|market|hotel|airlines?|logistics|consulting)\.?$", + re.IGNORECASE, +) + + def _mask_name(value: str) -> str: return " ".join(t[0] + "*" * (len(t) - 1) if len(t) > 1 else t for t in value.split()) @@ -316,13 +346,17 @@ def scan_pii( findings.append(_finding("Phone Number", _CATEGORY_PII, "Medium", raw, score, match.start, match.end)) except Exception: pass - if phone_regions and (phone_context or near(text, 0, min(len(text), 200), CONTEXT_WORDS["Phone Number"])): + if phone_regions: for region in phone_regions: try: for match in phonenumbers.PhoneNumberMatcher(text, region.upper(), leniency=phonenumbers.Leniency.VALID): if (match.start, match.end) in seen_spans: continue seen_spans.add((match.start, match.end)) + # the keyword must sit next to the number, not somewhere in the first 200 characters; + # a bare national-format digit run is left to the checksum detectors (NHS, SSN, accounts) + if not (phone_context or near(text, match.start, match.end, CONTEXT_WORDS["Phone Number"])): + continue findings.append(_finding("Phone Number", _CATEGORY_PII, "Medium", match.raw_string, 0.85, match.start, match.end, region=region.upper())) except Exception: continue @@ -338,8 +372,8 @@ def scan_pii( # 3. Dates of birth: plausible year range + birth context or field for match in DOB_REGEX.finditer(text): - year = int(match.group(0)[:4]) - if not (DOB_MIN_YEAR <= year <= DOB_MAX_YEAR): + year = _dob_year(match.group(0)) + if year is None or not (DOB_MIN_YEAR <= year <= DOB_MAX_YEAR): continue score = 0.4 if field_hints("Date of Birth", field_name) or near(text, match.start(), match.end(), CONTEXT_WORDS["Date of Birth"]): @@ -430,6 +464,22 @@ def scan_addresses(text: str, field_name: Optional[str] = None) -> list: _SINGLE_NAME_FIELD_RE = re.compile(r"\b(?:first|last|given|family|middle|sur) ?name\b") +# a name label in prose or markdown, unquoted: "First Name: Facundo", "**Patient Name**: Ana Rojas", "Name: Scott A. Smith" +# "Bank Name", "Employer Name", "Campaign Name": a bare "name" label owned by a thing, not a person +NON_PERSON_NAME_QUALIFIERS = { + "company", "business", "employer", "farm", "bank", "account", "campaign", "product", "project", "brand", "store", + "vendor", "supplier", "organisation", "organization", "institution", "fund", "trust", "school", "hospital", "clinic", + "team", "group", "plan", "policy", "file", "user", "host", "domain", "server", "table", "column", "field", "display", + "screen", "model", "device", "app", "application", "service", "agency", "firm", "corporation", "entity", "branch", + "merchant", "issuer", "carrier", "insurer", "ship", "vessel", "property", "estate", "building", "street", "road", +} +NAME_KEYS_PROSE_RE = re.compile( + r"(?first name|last name|surname|given name|family name|middle name|full name|patient name|" + r"customer name|employee name|applicant name|account holder|cardholder name|contact name|name|borrower|co-borrower|guarantor|" + r"insured|consignee|shipper|buyer|seller|tenant|landlord|beneficiary|shareholder|signatory|attorney|client|patient|employee|" + r"customer|applicant|holder|spouse|dependent|witness)\**)\s*[:\-]\s*" + r"(?P(?:[A-Z]\.|[A-Z][A-Za-z'\u2019\-]+)(?:[ \t](?:[A-Z]\.|[A-Z][A-Za-z'\u2019\-]+)){0,3})(?![A-Za-z])", +) def scan_person_names(text: str, field_name: Optional[str] = None) -> list: @@ -445,6 +495,20 @@ def scan_person_names(text: str, field_name: Optional[str] = None) -> list: key_single = _SINGLE_NAME_FIELD_RE.search(match.group("key").lower().replace("_", " ")) is not None if looks_like_person_name(val) and (key_single or len(val.split()) >= 2): findings.append(_finding("PII.PersonName", _CATEGORY_PII, "Low", val, 0.85, match.start("val"), match.end("val"), masked=_mask_name(val), key=match.group("key"))) + taken = [(f["start"], f["end"]) for f in findings] + for match in NAME_KEYS_PROSE_RE.finditer(text): + val = match.group("val").strip().rstrip(".") + if any(s <= match.start("val") < e for s, e in taken): + continue + key = match.group("key").lower() + if key == "name": # the word in front decides: "Patient Name" is a person, "Bank Name" is not + before = text[max(0, match.start("key") - 40):match.start("key")] + qualifier = re.findall(r"[A-Za-z]+", before) + if qualifier and qualifier[-1].lower() in NON_PERSON_NAME_QUALIFIERS: + continue + key_single = key != "name" and _SINGLE_NAME_FIELD_RE.search(key) is not None + if looks_like_person_name(val) and (key_single or len(val.split()) >= 2): + findings.append(_finding("PII.PersonName", _CATEGORY_PII, "Low", val, 0.85, match.start("val"), match.end("val"), masked=_mask_name(val), key=match.group("key"))) return findings @@ -456,6 +520,12 @@ def scan_person_names(text: str, field_name: Optional[str] = None) -> list: r"(?i)(?]+\s*[\"']?([^\s\"',;]{1,128})", ) +# Passwords stated in prose: "the password is hunter2!", "temporary password: Xk9#pq2L", "your new passcode will be ..." +PASSWORD_PROSE_REGEX = re.compile( + r"(?i)(?[^\s\"'\u201d\u2019,;]{6,128})", +) API_KEY_REGEX = re.compile( r"(?i)(? list: if any(f["start"] <= match.start() and match.end() <= f["end"] for f in findings): continue # inside an IBAN score = 0.5 - if field_hints("SWIFT/BIC", field_name) or near(text, match.start(), match.end(), CONTEXT_WORDS["SWIFT/BIC"]): - score = 0.85 + explicit = field_hints("SWIFT/BIC", field_name) or near(text, match.start(), match.end(), SWIFT_EXPLICIT_WORDS, before=40, after=10) + generic = near(text, match.start(), match.end(), CONTEXT_WORDS["SWIFT/BIC"]) + if explicit or (generic and any(c.isdigit() for c in val)): + score = 0.85 # "TRANSFER" next to "bank" is a word; "DEUTDEFF" next to "SWIFT" is a code findings.append(_finding("SWIFT/BIC", _CATEGORY_FIN, "High", val, score, match.start(), match.end())) # 4. Bank account numbers: digit runs with bank context, not cloud account ids diff --git a/src/engine/recognizers/__init__.py b/src/engine/recognizers/__init__.py index a95427c..05be40f 100644 --- a/src/engine/recognizers/__init__.py +++ b/src/engine/recognizers/__init__.py @@ -26,6 +26,21 @@ ) _ALL: Optional[List[Rule]] = None + +# Rules whose upstream pattern score is low (0.01-0.35) but whose shape is specific enough that a keyword next +# to the value makes it `likely` rather than merely `possible`: passports and licences with fixed letter/digit +# layouts, postcodes with their own grammar, tax and healthcare administration ids with prefix rules, vehicle ids. +# Left out on purpose: bare digit runs that collide with phones, dates and counters (bank accounts, PLZ, TIN, +# sort codes, BSB) and ICD-10, whose letter + 2 digits shape is also a vitamin or a bus route. +LIKELY_FLOOR = 0.8 +KEYWORD_MAKES_LIKELY = frozenset({ + "AR_DNI", "CA_POSTAL_CODE", "UK_POSTCODE", + "DE_FUEHRERSCHEIN", "DE_KFZ", "IT_DRIVER_LICENSE", "IT_IDENTITY_CARD", + "ES_PASSPORT", "FR_PASSPORT", "UK_PASSPORT", "IT_PASSPORT", "JP_PASSPORT", "KR_PASSPORT", "PH_PASSPORT", + "US_ALIEN_REGISTRATION", "US_EIN", # US_PROVIDER_TAX_ID stays at 0.7: "employee tax id" is a deliberate non-match for the PHI class + "US_CLAIM_NUMBER", "US_PRIOR_AUTHORIZATION_NUMBER", "US_REFERRAL_NUMBER", + "VN_CCCD", "NDC_CODE", "VIN", +}) _BY_NAME: Dict[str, Rule] = {} @@ -42,6 +57,9 @@ def load_all() -> List[Rule]: if rule.name in by_name: raise ValueError(f"Duplicate upstream rule name: {rule.name}") by_name[rule.name] = rule + for rule in rules: + if rule.name in KEYWORD_MAKES_LIKELY: + rule.min_score_with_context = max(rule.min_score_with_context, LIKELY_FLOOR) _BY_NAME.update(by_name) _ALL = rules return _ALL diff --git a/src/engine/recognizers/asia_pacific.py b/src/engine/recognizers/asia_pacific.py index 080b284..4a67a13 100644 --- a/src/engine/recognizers/asia_pacific.py +++ b/src/engine/recognizers/asia_pacific.py @@ -157,10 +157,10 @@ def _validate_au_ihi(text: str) -> bool: return len(v) == 16 and v.startswith("800360") and _plain_luhn(v) -def _rule(name, region, description, patterns, context, validator=None, field_hint=None, examples=(), category=_REGIONAL, severity="Critical"): +def _rule(name, region, description, patterns, context, validator=None, field_hint=None, examples=(), category=_REGIONAL, severity="Critical", **kwargs): return Rule( name=name, category=category, severity=severity, region=region, description=description, - patterns=patterns, context=context, validator=validator, field_hint=field_hint, examples=list(examples), + patterns=patterns, context=context, validator=validator, field_hint=field_hint, examples=list(examples), **kwargs, ) @@ -253,6 +253,7 @@ def _rule(name, region, description, patterns, context, validator=None, field_hi [Pattern("IFSC", r"\b[A-Z]{4}0[A-Z0-9]{6}\b", 0.3)], ["ifsc", "ifsc code", "bank", "branch", "neft", "rtgs", "imps"], None, r"ifsc", ["HDFC0001234"], category=_FINANCIAL, severity="Low", # pragma: allowlist secret + min_score_with_context=0.8, # the 4-letter + 0 + 6 shape next to a bank word is a code, not a word ), _rule( "IN_UPI_ID", "IN", "Indian UPI virtual payment address: handle@bank-psp (okaxis, ybl, paytm, upi ...).", diff --git a/src/engine/recognizers/identifiers.py b/src/engine/recognizers/identifiers.py index 54e4df6..57c2045 100644 --- a/src/engine/recognizers/identifiers.py +++ b/src/engine/recognizers/identifiers.py @@ -5,6 +5,7 @@ detector policy - weak shapes only surface through a column name, a keyword or column density. """ +import re from typing import Optional from src.engine.rules import Pattern, Rule @@ -81,20 +82,27 @@ def _validate_passport_mrz(text: str) -> Optional[bool]: def _validate_geo(text: str) -> Optional[bool]: + """A lat,lon / lat lon / hemisphere pair within range, or a single in-range value; never authoritative.""" + cleaned = re.sub(r"[°NSEW]", "", text, flags=re.IGNORECASE) try: - lat, lon = (float(part.strip()) for part in text.split(",")) + parts = [float(part) for part in re.split(r"[,\s]+", cleaned.strip()) if part] except ValueError: return False - if not (-90.0 <= lat <= 90.0 and -180.0 <= lon <= 180.0): - return False - return None if (lat, lon) != (0.0, 0.0) else False + if len(parts) == 2: + lat, lon = parts + if not (-90.0 <= lat <= 90.0 and -180.0 <= lon <= 180.0): + return False + return None if (lat, lon) != (0.0, 0.0) else False + if len(parts) == 1: + return None if -180.0 <= parts[0] <= 180.0 and parts[0] != 0.0 else False + return False -def _rule(name, description, patterns, context, validator=None, field_hint=None, examples=(), category=_PII, severity="Low", weak_validation=False): +def _rule(name, description, patterns, context, validator=None, field_hint=None, examples=(), category=_PII, severity="Low", weak_validation=False, **kwargs): return Rule( name=name, category=category, severity=severity, region=None, description=description, patterns=patterns, context=context, validator=validator, field_hint=field_hint, examples=list(examples), - weak_validation=weak_validation, + weak_validation=weak_validation, **kwargs, ) @@ -103,7 +111,7 @@ def _rule(name, description, patterns, context, validator=None, field_hint=None, "IMEI", "Mobile equipment identity (IMEI): 15 digits with a Luhn check digit.", [Pattern("IMEI (formatted)", r"\b\d{2}[- ]\d{6}[- ]\d{6}[- ]\d\b", 0.3), Pattern("IMEI (weak)", r"\b\d{15}\b", 0.05)], ["imei", "device id", "handset", "equipment identity", "device identifier"], - _validate_imei, r"imei|device_?id|handset", ["35-845422-110932-2", "352318504122227"], # pragma: allowlist secret + _validate_imei, r"imei|device_?id|handset|equipment", ["35-845422-110932-2", "352318504122227"], # pragma: allowlist secret ), _rule( "ICCID", "SIM card serial (ICCID): 19-20 digits starting 89 with a Luhn check digit.", @@ -129,9 +137,15 @@ def _rule(name, description, patterns, context, validator=None, field_hint=None, ), _rule( "GEO_COORDINATES", "Geographic coordinates: latitude, longitude pair with 4+ decimals; reported next to location context.", - [Pattern("lat,lon", r"(?[A-Z]{1,3}-?\d{5,10})\b", 0.85), + Pattern("MRN (alphanumeric, weak)", r"\b[A-Z]{1,3}-?\d{6,10}\b", 0.15), + Pattern("MRN (labelled words)", r"\b(?:medical record(?: number| no\.?| #)?|record number|patient (?:id|number|no\.?)|chart (?:number|no\.?))\s*[:#|-]?\s*(?P(?:MRN-?)?\d{6,12})\b", 0.85), + Pattern("MRN (weak)", r"\b\d{6,10}\b", 0.01), + ], ["mrn", "medical record", "medical record number", "patient id", "chart number"], - None, r"(? bool: ], context=["passport", "indian passport", "passport number"], field_hint=r"passport", + min_score_with_context=0.8, # a specific 8-character shape next to "passport" is likely, not merely possible examples=("A3456781", "T3569075"), ), Rule( @@ -532,6 +533,7 @@ def _validate_th_tnin(text: str) -> bool: ], context=["voter", "epic", "elector photo identity card"], field_hint=r"voter|(? str: region="US", description="US passport number (UsPassportRecognizer): 9 digits or letter + 8 digits (next generation).", patterns=[ + Pattern("Passport (labelled)", r"\bpassport(?: number| no\.?| #|#)?\s*[:#-]?\s*(?P[A-Z]?[0-9]{8,9})\b", 0.85), Pattern("Passport (very weak)", r"(\b[0-9]{9}\b)", 0.05), Pattern("Passport Next Generation (very weak)", r"(\b[A-Z][0-9]{8}\b)", 0.1), ], context=["us", "united", "states", "passport", "passport#", "travel", "document"], field_hint=r"passport", + min_score_with_context=0.5, # "passport" next to a 9-digit run is a candidate, not silence examples=["912803456", "A12803456"], #pragma: allowlist secret ), Rule( @@ -305,10 +307,12 @@ def _labelled(labels: Sequence[str], body: str) -> str: r"\b([A-Z][0-9]{3,6}|[A-Z][0-9]{5,9}|[A-Z][0-9]{6,8}|[A-Z][0-9]{4,8}|[A-Z][0-9]{9,11}|[A-Z]{1,2}[0-9]{5,6}|H[0-9]{8}|V[0-9]{6}|X[0-9]{8}|A-Z]{2}[0-9]{2,5}|[A-Z]{2}[0-9]{3,7}|[0-9]{2}[A-Z]{3}[0-9]{5,6}|[A-Z][0-9]{13,14}|[A-Z][0-9]{18}|[A-Z][0-9]{6}R|[A-Z][0-9]{9}|[A-Z][0-9]{1,12}|[0-9]{9}[A-Z]|[A-Z]{2}[0-9]{6}[A-Z]|[0-9]{8}[A-Z]{2}|[0-9]{3}[A-Z]{2}[0-9]{4}|[A-Z][0-9][A-Z][0-9][A-Z]|[0-9]{7,8}[A-Z])\b", 0.3, ), + Pattern("Driver License (labelled)", r"\bdriver'?s?[ -]licen[cs]e(?: number| no\.?| #|#)?\s*[:#-]?\s*(?P[A-Z0-9]{5,14})\b", 0.85), Pattern("Driver License - Digits (very weak)", r"\b([0-9]{6,14}|[0-9]{16})\b", 0.01), ], context=["driver", "license", "permit", "lic", "identification", "dls", "cdls", "lic#", "driving"], field_hint=r"driv(er|ing)s?_?licen[cs]e|(? str: r"(?=[A-Z0-9-]*\d)[A-Z]{1,5}-?[A-Z0-9]{5,14}\b", 0.1, ), + Pattern( # the label itself is the evidence: "health plan beneficiary number is 8429 301 745 MN" + "Health plan id (labelled)", + r"\b(?:health plan beneficiary (?:number|id|no\.?)|beneficiary (?:number|id|no\.?)|member (?:id|number|no\.?)|" + r"subscriber (?:id|number|no\.?)|plan member (?:id|number))\s*(?:is|:|#|-)?\s*" + r"(?P(?:(?=[A-Z0-9-]{5,20}(?![A-Z0-9-]))(?=[A-Z0-9-]*\d)[A-Z0-9][A-Z0-9-]{4,19}|\d{3,4}(?:[ -]\d{3,4}){1,3})(?:\s(?-i:[A-Z]{2}))?)(?![A-Z0-9-])", + 0.85, + ), ], context=["member", "subscriber", "insurance", "policy"], - field_hint=r"member_?(id|num|no|number)|subscriber_?(id|num|no|number)|insurance_?(id|num|no|number)|policy_?(num|no|number)", + field_hint=r"member_?(id|num|no|number)|subscriber_?(id|num|no|number)|insurance_?(id|num|no|number)|policy_?(num|no|number)|beneficiary", examples=["ABC123456789", "ZX-987654321", "HPN12345A9"], #pragma: allowlist secret ), Rule( diff --git a/src/engine/recognizers/za_ng_ph_generic.py b/src/engine/recognizers/za_ng_ph_generic.py index 6a16ad4..6cb491d 100644 --- a/src/engine/recognizers/za_ng_ph_generic.py +++ b/src/engine/recognizers/za_ng_ph_generic.py @@ -1252,7 +1252,7 @@ def _invalidate_ip_address(pattern_text: str) -> bool: # MAC address (mac_recognizer.py) # --------------------------------------------------------------------------- def _invalidate_mac_address(pattern_text: str) -> bool: - cleaned = re.sub(r"[:\-.]", "", pattern_text) + cleaned = re.sub(r"[:\-. ]", "", pattern_text) # All characters must be valid hex if re.fullmatch(r"[0-9A-Fa-f]{12}", cleaned) is None: @@ -1274,7 +1274,7 @@ def _invalidate_mac_address(pattern_text: str) -> bool: patterns=[ Pattern( "MAC_COLON_OR_HYPHEN", - r"\b[0-9A-Fa-f]{2}([:-])(?:[0-9A-Fa-f]{2}\1){4}[0-9A-Fa-f]{2}\b", + r"\b[0-9A-Fa-f]{2}([:\- ])(?:[0-9A-Fa-f]{2}\1){4}[0-9A-Fa-f]{2}\b", 0.6, ), Pattern( diff --git a/src/engine/rules.py b/src/engine/rules.py index ea71ec9..7a937e6 100644 --- a/src/engine/rules.py +++ b/src/engine/rules.py @@ -138,10 +138,50 @@ def can_reach(self, threshold: float, words: frozenset, lowered: str, field_name return best >= threshold +# Words a run-together field name may be made of: 'familynamefemale' -> family name female, +# 'keystorepassword' -> key store password. A token is split only when every piece is a known word. +_FIELD_VOCAB = frozenset( + """ +account address api auth authorization bank bearer beneficiary billing birth card cardholder cell chart city client +code contact country county credit current customer date debit device dob driver email employee family fax female +first full gender given health holder home iban id identity imei insurance key last license licence login mac male +master medical member middle mobile mrn name national new nonbinary number old owner passport password patient +phone pin plan policy postal pwd record routing secret security shipping social ssn state store street subscriber +sur swift tax telephone token user visa work zip +""".split(), +) +_SEGMENT_MIN = 3 + + +def _segment(token: str) -> Optional[List[str]]: + """Greedy-with-backtracking split of a run-together token into known words, or None.""" + if len(token) < 2 * _SEGMENT_MIN or not token.isalpha(): + return None + best: Dict[int, Optional[List[str]]] = {len(token): []} + + def walk(pos: int) -> Optional[List[str]]: + if pos in best: + return best[pos] + result = None + for end in range(len(token), pos + 1, -1): + piece = token[pos:end] + if len(piece) < 2 or piece not in _FIELD_VOCAB: + continue + rest = walk(end) + if rest is not None: + result = [piece] + rest + break + best[pos] = result + return result + + words = walk(0) + return words if words and len(words) >= 2 else None + + def tokenize_field_name(field_name: Optional[str]) -> List[str]: """ 'api_event.http.request.headers.authorization' -> ['api', 'event', 'http', ...] - 'customerSSN' -> ['customer', 'ssn']; 'zipCode' -> ['zip', 'code'] + 'customerSSN' -> ['customer', 'ssn']; 'zipCode' -> ['zip', 'code']; 'keystorepassword' -> ['key', 'store', 'password'] """ if not field_name: return [] @@ -151,7 +191,8 @@ def tokenize_field_name(field_name: Optional[str]) -> List[str]: continue for token in _CAMEL_RE.split(chunk): if token: - parts.append(token.lower()) + lowered = token.lower() + parts.extend(_segment(lowered) or [lowered]) return parts @@ -211,7 +252,8 @@ def run_rule(rule: Rule, text: str, field_name: Optional[str] = None) -> List[Di for pattern in rule.patterns: for match in pattern.compiled.finditer(text): - start, end = match.span() + # a labelled pattern names its value with (?P...): the label is matched, the value is reported + start, end = match.span("v") if "v" in match.groupdict() and match.group("v") else match.span() value = text[start:end] if not value: continue @@ -227,6 +269,8 @@ def run_rule(rule: Rule, text: str, field_name: Optional[str] = None) -> List[Di continue context_word = find_context_word(rule, text, start, end, field_name) + if context_word is None and (start, end) != match.span(): + context_word = text[match.start():start].strip(" :#-|\t") or pattern.name # the label is the context if context_word is not None: score = min(1.0, max(score + rule.context_boost, rule.min_score_with_context)) if field_hint_hit: diff --git a/tests/fixtures/accuracy_cases.json b/tests/fixtures/accuracy_cases.json index ddd99fc..b5bfcc1 100644 --- a/tests/fixtures/accuracy_cases.json +++ b/tests/fixtures/accuracy_cases.json @@ -395,6 +395,207 @@ "purpose": "Clean prose remains clean.", "text": "The deployment completed successfully. No action is required.", "expected": [] + }, + { + "id": "phone_labelled_in_prose", + "purpose": "A national-format number next to its own keyword is a phone wherever it sits in the text.", + "text": "Notes follow. Notes follow. Notes follow. Notes follow. Notes follow. Notes follow. Notes follow. Notes follow. Notes follow. Notes follow. Notes follow. Notes follow. Notes follow. Notes follow. Notes follow. Notes follow. Notes follow. Notes follow. Notes follow. Notes follow. Phone Number: 205-897-7208", + "expected": [ + { + "detector": "Phone Number", + "value": "205-897-7208", + "location": "text" + } + ] + }, + { + "id": "ledger_reference_is_not_a_phone", + "purpose": "A bare national-format digit run with no phone keyword is not reported.", + "text": "ref 205-897-7208 in the ledger", + "expected": [] + }, + { + "id": "dob_day_first_and_written", + "purpose": "Dates of birth in day-first and written-month forms carry the birth keyword.", + "text": "Date of Birth: 12/08/1985. Born on 3 March 1979.", + "expected": [ + { + "detector": "Date of Birth", + "value": "12/08/1985", + "location": "text" + }, + { + "detector": "Date of Birth", + "value": "3 March 1979", + "location": "text" + } + ] + }, + { + "id": "invoice_date_is_not_a_birth_date", + "purpose": "A day-first date without a birth keyword stays silent.", + "text": "invoice 12/08/1985 total 40.00", + "expected": [] + }, + { + "id": "medical_record_number_labelled", + "purpose": "A medical record number labelled in words is reported as the number alone.", + "text": "Medical Record Number: 0004973621 was reviewed", + "expected": [ + { + "detector": "MEDICAL_RECORD_NUMBER", + "value": "0004973621", + "location": "text" + } + ] + }, + { + "id": "health_plan_member_id_labelled", + "purpose": "A health plan beneficiary number, digits or letters, is reported when its label precedes it.", + "text": "Health Plan Beneficiary Number: 245 876 1234. Also beneficiary number G172864593 at the hospital.", + "expected": [ + { + "detector": "US_HEALTH_INSURANCE_MEMBER_ID", + "value": "245 876 1234", + "location": "text" + }, + { + "detector": "US_HEALTH_INSURANCE_MEMBER_ID", + "value": "G172864593", + "location": "text" + } + ] + }, + { + "id": "passport_and_licence_labelled", + "purpose": "Labelled passport and driver licence numbers are reported as the number alone.", + "text": "Passport Number: B31498009. Driver's License Number: H8477831.", + "expected": [ + { + "detector": "US_PASSPORT", + "value": "B31498009", + "location": "text" + }, + { + "detector": "US_DRIVER_LICENSE", + "value": "H8477831", + "location": "text" + } + ] + }, + { + "id": "names_labelled_in_prose", + "purpose": "Unquoted name labels in prose and markdown report the name.", + "text": "**First Name**: Facundo **Last Name**: Fernandez. Borrower: Natasha Justin Ellis.", + "expected": [ + { + "detector": "PII.PersonName", + "value": "Facundo", + "location": "text" + }, + { + "detector": "PII.PersonName", + "value": "Fernandez", + "location": "text" + }, + { + "detector": "PII.PersonName", + "value": "Natasha Justin Ellis", + "location": "text" + } + ] + }, + { + "id": "uppercase_word_is_not_a_swift_code", + "purpose": "An 8-letter English word next to a generic bank word is not a SWIFT code; an explicit label is.", + "text": "bank transfer from BUSINESS account. SWIFT: DEUTDEFF", + "expected": [ + { + "detector": "SWIFT/BIC", + "value": "DEUTDEFF", + "location": "text" + } + ] + }, + { + "id": "run_together_field_names", + "purpose": "Run-together field names are segmented into the words the field rules know.", + "rows": [ + { + "fullnamefemale": "Whitney Dixion", + "keystorepassword": "eigahYe7se", + "latitude": "-28.323485", + "version": "1.4.2" + } + ], + "expected": [ + { + "detector": "PII.PersonName", + "value": "Whitney Dixion", + "location": "row:0:fullnamefemale" + }, + { + "detector": "Password Pattern", + "value": "eigahYe7se", + "location": "row:0:keystorepassword" + }, + { + "detector": "GEO_COORDINATES", + "value": "-28.323485", + "location": "row:0:latitude" + } + ] + }, + { + "id": "coordinate_pair_forms", + "purpose": "Space-separated and hemisphere coordinate pairs count next to a location keyword.", + "text": "location 40.7128 -74.0060; coordinates 35.7790 N, 78.6382 W", + "expected": [ + { + "detector": "GEO_COORDINATES", + "value": "40.7128 -74.0060", + "location": "text" + }, + { + "detector": "GEO_COORDINATES", + "value": "35.7790 N, 78.6382 W", + "location": "text" + } + ] + }, + { + "id": "single_names_under_plain_labels", + "purpose": "Unstarred first and last name labels report single names; a thing-owned Name label does not.", + "text": "First Name: Facundo\nLast Name: Fernandez\nBank Name: First National Bank\nCustomer Name: Priya Sharma", + "expected": [ + { + "detector": "PII.PersonName", + "value": "Facundo", + "location": "text" + }, + { + "detector": "PII.PersonName", + "value": "Fernandez", + "location": "text" + }, + { + "detector": "PII.PersonName", + "value": "Priya Sharma", + "location": "text" + } + ] + }, + { + "id": "passwords_stated_in_prose", + "purpose": "A password stated in a sentence is reported; a sentence about passwords is not.", + "text": "Your temporary password is Xk9#pq2L and expires tonight. The password is required on every login. Password was reset by the admin.", + "expected": [ + { + "detector": "Password Pattern", + "value": "Xk9#pq2L", + "location": "text" + } + ] } ] } diff --git a/tests/sample_dataset_builder.py b/tests/sample_dataset_builder.py index 4d7872b..731513b 100644 --- a/tests/sample_dataset_builder.py +++ b/tests/sample_dataset_builder.py @@ -155,7 +155,7 @@ def detected(value): def build(): - mapping = json.loads((ROOT / "fixtures" / "findings-mapping.json").read_text())[0] + mapping = json.loads((ROOT / "fixtures" / "findings-mapping-v2.json").read_text())[0] rows = [] used = set() engine = DetectionEngine({"enabled_regions": regions(), "ner": False}) diff --git a/tests/test_detector_names.py b/tests/test_detector_names.py index 9f79d16..08234a6 100644 --- a/tests/test_detector_names.py +++ b/tests/test_detector_names.py @@ -13,7 +13,7 @@ def _mapping(): - data = json.loads((ROOT / "fixtures" / "findings-mapping.json").read_text()) + data = json.loads((ROOT / "fixtures" / "findings-mapping-v2.json").read_text()) return data[0] if isinstance(data, list) else data diff --git a/tests/test_policy_confidence.py b/tests/test_policy_confidence.py index 0651760..913a851 100644 --- a/tests/test_policy_confidence.py +++ b/tests/test_policy_confidence.py @@ -56,7 +56,7 @@ def test_engine_reporting_tier_and_legacy_threshold(): def test_every_detector_resolves_to_a_policy(): - mapping = json.loads((ROOT / "fixtures" / "findings-mapping.json").read_text()) + mapping = json.loads((ROOT / "fixtures" / "findings-mapping-v2.json").read_text()) mapping = mapping[0] if isinstance(mapping, list) else mapping for name, entry in mapping.items(): policy = policy_for(name, entry["category"]) diff --git a/tests/test_public_benchmark.py b/tests/test_public_benchmark.py new file mode 100644 index 0000000..e6e4a89 --- /dev/null +++ b/tests/test_public_benchmark.py @@ -0,0 +1,64 @@ +"""Unit tests for scripts/benchmark_public_datasets.py: validators, payload sniffing and the scorer. +No dataset download is needed; the loaders are not exercised here.""" +from scripts.benchmark_public_datasets import ( + GROUPS, Record, _payload_format, card_valid, iban_valid, imei_valid, routing_valid, score, ssn_valid, +) + + +def test_validators_accept_checksum_valid_values_only(): + assert card_valid("4111 1111 1111 1111") and not card_valid("4111111111111112") + assert iban_valid("GB82 WEST 1234 5698 7654 32") and not iban_valid("GB82 WEST 1234 5698 7654 33") + assert routing_valid("021000021") and not routing_valid("271210785") + assert ssn_valid("219-09-9999") + assert not ssn_valid("666-12-3456") and not ssn_valid("000-12-3456") and not ssn_valid("219-00-9999") + assert imei_valid("490154203237518") and not imei_valid("490154203237519") + + +def test_payload_format_sniffing(): + assert _payload_format('{"a": 1}') == "json" + assert _payload_format("") == "xml/html" + assert _payload_format("INSERT INTO t VALUES (1)") == "sql" + assert _payload_format("plain words") == "other" + + +SPEC = {"email": ("target", "email"), "ssn": ("target", "ssn"), "url": ("ambiguous", "url")} + + +def _hit(detector, start, end, confidence="likely", value=""): + return {"detector": detector, "start": start, "end": end, "confidence": confidence, "value": value} + + +def test_scorer_counts_overlap_hits_wrong_types_and_tiers(): + text = "mail a@b.com ssn 219-09-9999 site http://x.io free 123-45-6789 company Acme" + record = Record("r1", text, [ + ("email", 5, 12, "a@b.com"), # target, hit by Email + ("ssn", 17, 28, "219-09-9999"), # target, hit only by a Phone Number (wrong type) + ("url", 34, 45, "http://x.io"), # ambiguous, URL hit is fine + ("company", 71, 75, "Acme"), # unscored label + ], {}) + hits = [ + _hit("Email", 5, 12), + _hit("Phone Number", 17, 28), + _hit("URL", 34, 45), + _hit("US SSN", 51, 62, "possible"), # no gold span: a false positive, but only at `possible` + _hit("PII.PersonName", 71, 75), # on an unscored span: left out of precision + ] + likely = score([record], [hits], SPEC, "likely", by_value=False) + assert likely["summary"]["accepted_hits"] == 1 + assert likely["summary"]["fp_wrong_type"] == 1 and likely["summary"]["fp_no_gold"] == 0 + assert likely["summary"]["hits_on_unscored_spans"] == 1 + assert likely["per_label"]["email"]["recall"] == 1.0 and likely["per_label"]["ssn"]["recall"] == 0.0 + assert likely["per_label"]["ssn"]["gold_valid"] == 1 and likely["per_label"]["ssn"]["recall_on_valid"] == 0.0 + assert likely["unscored_label_spans"] == {"company": 1} + assert likely["summary"]["precision"] == 0.5 and likely["summary"]["recall"] == 0.5 + + possible = score([record], [hits], SPEC, "possible", by_value=False) + assert possible["summary"]["fp_no_gold"] == 1 + assert possible["false_positives_by_detector"] == {"Phone Number": 1, "US SSN": 1} + + +def test_scorer_matches_by_value_in_document_mode(): + record = Record("r2", '{"email": "a@b.com"}', [("email", 11, 18, "a@b.com")], {"format": "json"}) + hits = [_hit("Email", -1, -1, value="a@b.com")] + assert score([record], [hits], SPEC, "likely", by_value=True)["per_label"]["email"]["tp"] == 1 + assert "Email" in GROUPS["email"] diff --git a/tests/test_recognizers_de_se_fi_pl.py b/tests/test_recognizers_de_se_fi_pl.py index 48b406c..358aad3 100644 --- a/tests/test_recognizers_de_se_fi_pl.py +++ b/tests/test_recognizers_de_se_fi_pl.py @@ -84,7 +84,7 @@ def test_all_expected_rules_present_and_unique(): def test_rules_match_findings_mapping(): - with open(ROOT / "fixtures" / "findings-mapping.json", encoding="utf-8") as fh: + with open(ROOT / "fixtures" / "findings-mapping-v2.json", encoding="utf-8") as fh: mapping = json.load(fh)[0] for rule in RULES: assert rule.name in mapping, rule.name diff --git a/tests/test_recognizers_gb_es_it_tr.py b/tests/test_recognizers_gb_es_it_tr.py index b82661e..2ea08ed 100644 --- a/tests/test_recognizers_gb_es_it_tr.py +++ b/tests/test_recognizers_gb_es_it_tr.py @@ -21,7 +21,7 @@ from src.engine.rules import run_rule EPS = 1e-6 -MAPPING_PATH = Path(__file__).resolve().parent.parent / "fixtures" / "findings-mapping.json" +MAPPING_PATH = Path(__file__).resolve().parent.parent / "fixtures" / "findings-mapping-v2.json" def _rule(name): @@ -224,7 +224,7 @@ def test_uk_passport(): ("AB1234567", [(0, 9, 0.1)]), ("XY9876543", [(0, 9, 0.1)]), ("ab1234567", [(0, 9, 0.1)]), - ("My passport number is CD7654321 and it expires soon", [(22, 31, 0.45)]), # context: "passport" + ("My passport number is CD7654321 and it expires soon", [(22, 31, 0.8)]), # context: "passport" ("Passports: AB1234567 and XY9876543", [(11, 20, 0.1), (25, 34, 0.1)]), # "passports" is not a context word ] invalid = [ @@ -260,8 +260,8 @@ def test_uk_postcode(): ("EC1A1BB", [(0, 7, 0.1)]), ("DN551PT", [(0, 7, 0.1)]), ("GIR0AA", [(0, 6, 0.1)]), - ("My address is SW1A 1AA in London", [(14, 22, 0.45)]), # context: "address" - ("Send to postcode EC2A 1NT please", [(17, 25, 0.45)]), # context: "postcode" + ("My address is SW1A 1AA in London", [(14, 22, 0.8)]), # context: "address" + ("Send to postcode EC2A 1NT please", [(17, 25, 0.8)]), # context: "postcode" ("From SW1A 1AA to EC1A 1BB", [(5, 13, 0.1), (17, 25, 0.1)]), ] invalid = [ @@ -403,15 +403,15 @@ def test_es_passport(): valid = [ ("AAA123456", [(0, 9, 0.05)]), ("XYZ987654", [(0, 9, 0.05)]), - ("Mi pasaporte es AAA123456", [(16, 25, 0.4)]), # context: "pasaporte" + ("Mi pasaporte es AAA123456", [(16, 25, 0.8)]), # keyword floor # context: "pasaporte" ("AAA123456 es mi número de pasaporte", [(0, 9, 0.05)]), # "pasaporte" is beyond the 3-word suffix window ("aaa123456", [(0, 9, 0.05)]), ("xyz987654", [(0, 9, 0.05)]), - ("Mi pasaporte es aaa123456", [(16, 25, 0.4)]), + ("Mi pasaporte es aaa123456", [(16, 25, 0.8)]), ("aaa123456 es mi número de pasaporte", [(0, 9, 0.05)]), ("AaA123456", [(0, 9, 0.05)]), ("XyZ987654", [(0, 9, 0.05)]), - ("Mi pasaporte es AaA123456", [(16, 25, 0.4)]), + ("Mi pasaporte es AaA123456", [(16, 25, 0.8)]), ("AaA123456 es mi número de pasaporte", [(0, 9, 0.05)]), ] invalid = [ diff --git a/tests/test_recognizers_in_sg_au_kr_th.py b/tests/test_recognizers_in_sg_au_kr_th.py index f305f4c..aaad764 100644 --- a/tests/test_recognizers_in_sg_au_kr_th.py +++ b/tests/test_recognizers_in_sg_au_kr_th.py @@ -29,7 +29,7 @@ _EPS = 0.00001 _BY_NAME = {r.name: r for r in RULES} -_MAPPING_PATH = os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "fixtures", "findings-mapping.json") +_MAPPING_PATH = os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "fixtures", "findings-mapping-v2.json") def _rule(name): diff --git a/tests/test_recognizers_us_ca.py b/tests/test_recognizers_us_ca.py index db3cf7a..acbb827 100644 --- a/tests/test_recognizers_us_ca.py +++ b/tests/test_recognizers_us_ca.py @@ -63,6 +63,11 @@ def _single(rule, text, value, lo, hi=None): _check(rule, text, [(start, start + len(value), lo, hi)]) +def _admin_score(name): + """Labelled claim, prior-authorization and referral numbers reach the `likely` floor (0.8); the others keep upstream's 0.7.""" + return 0.8 if name in ("US_PRIOR_AUTHORIZATION_NUMBER", "US_CLAIM_NUMBER", "US_REFERRAL_NUMBER") else 0.7 + + def _below(rule, text, threshold): """Every result scores below `threshold` (the upstream recognizer's analyzer would drop it).""" for res in _found(rule, text): @@ -81,7 +86,7 @@ def _field(rule, value, field_name): def _load_mapping(): """fixtures/findings-mapping.json: {detector: entry} (historically wrapped in a one-element list).""" - with open(ROOT / "fixtures" / "findings-mapping.json", encoding="utf-8") as fh: + with open(ROOT / "fixtures" / "findings-mapping-v2.json", encoding="utf-8") as fh: data = json.load(fh) return data[0] if isinstance(data, list) else data @@ -197,10 +202,12 @@ def test_us_passport_upstream_cases(): _check(rule, "912803456", [(0, 9, 0.0, 0.1)]) _check(rule, "Z12803456", [(0, 9, 0.0, 0.15)]) _check(rule, "A12803456", [(0, 9, 0.0, 0.15)]) - # "travel" / "passport" are context words: upstream expects >= 0.0, the engine boosts to 0.45 + # "travel" / "passport" are context words: upstream expects >= 0.0, the engine lifts a keyword-backed + # number to the `possible` floor (0.5) so column density or a record can promote it; a label reports the value _check(rule, "my travel document is A12803456", [(22, 31, 0.0, MAX)]) _check(rule, "my travel passport is A12803456", [(22, 31, 0.0, MAX)]) - _single(rule, "my travel passport is A12803456", "A12803456", 0.45) + _single(rule, "my travel passport is A12803456", "A12803456", 0.5) + _single(rule, "Passport Number: B31498009", "B31498009", 0.85, 1.0) def test_us_passport_field_name(): @@ -343,16 +350,16 @@ def test_us_health_insurance_member_id_with_context(): ("Policy ID CIGNA123456 belongs to the patient", ((10, 21),)), ("The insurance card lists subscriber number K123456789", ((43, 53),)), ): - _check(rule, text, [(s, e, 0.45, 0.45) for s, e in spans]) + _check(rule, text, [(s, e, 0.45, 1.0) for s, e in spans]) # a label pattern plus its context word scores 1.0, context alone 0.45 # case-insensitive, trailing punctuation outside the span - _single(rule, "member id abc123456", "abc123456", 0.45) - _single(rule, "MeMbEr Id AbC123456", "AbC123456", 0.45) - _single(rule, "Subscriber ID zx-987654321.", "zx-987654321", 0.45) + _single(rule, "member id abc123456", "abc123456", 0.85, 1.0) + _single(rule, "MeMbEr Id AbC123456", "AbC123456", 0.45, 1.0) + _single(rule, "Subscriber ID zx-987654321.", "zx-987654321", 0.45, 1.0) # multiple IDs - _check(rule, "Member ID ABC123456 and subscriber ID ZX-987654321.", [(10, 19, 0.45, 0.45), (38, 50, 0.45, 0.45)]) + _check(rule, "Member ID ABC123456 and subscriber ID ZX-987654321.", [(10, 19, 0.45, 1.0), (38, 50, 0.45, 1.0)]) # 6 and 20 character boundaries - _single(rule, "Member ID A12345", "A12345", 0.45) - _single(rule, "Member ID ABCDE-12345678901234", "ABCDE-12345678901234", 0.45) + _single(rule, "Member ID A12345", "A12345", 0.45, 1.0) + _single(rule, "Member ID ABCDE-12345678901234", "ABCDE-12345678901234", 0.45, 1.0) def test_us_health_insurance_member_id_without_context(): @@ -365,9 +372,11 @@ def test_us_health_insurance_member_id_without_context(): "ICD10CM123", "ABC-1234567", ): _below(rule, text, 0.4) - # implausible: numeric-only, too short, too long - for text in ("Member ID 1234567890", "Subscriber ID A123", "Member ID ABCDE-123456789012345"): + # implausible: too short, too long; a numeric-only id counts only with its label in front + for text in ("Subscriber ID A123", "Member ID ABCDE-123456789012345", "1234567890"): _none(rule, text) + _single(rule, "Member ID 1234567890", "1234567890", 0.85, 1.0) + _single(rule, "health plan beneficiary number is 8429 301 745 MN.", "8429 301 745 MN", 0.85, 1.0) # raw pattern score _check(rule, "ABC123456789", [(0, 12, 0.1, 0.1)]) @@ -391,7 +400,7 @@ def test_us_health_insurance_member_id_field_name(): def test_healthcare_admin_id_with_context_is_detected(): for name, text, value in _ADMIN: - _single(_rule(name), text, value, 0.7) + _single(_rule(name), text, value, _admin_score(name)) def test_healthcare_admin_id_matching_is_case_insensitive(): @@ -402,7 +411,7 @@ def test_healthcare_admin_id_matching_is_case_insensitive(): ("US_REFERRAL_NUMBER", "rEfErRaL inf123456", "inf123456"), ("US_PROVIDER_TAX_ID", "bIlLiNg PrOvIdEr eIn: 12-3456789", "12-3456789"), ): - _single(_rule(name), text, value, 0.7) + _single(_rule(name), text, value, _admin_score(name)) def test_healthcare_admin_multiple_ids_are_all_detected(): @@ -415,7 +424,7 @@ def test_healthcare_admin_multiple_ids_are_all_detected(): ): results = _found(_rule(name), text) assert [r["value"] for r in results] == values, (name, text, results) - assert all(abs(r["score"] - 0.7) < EPS for r in results), (name, results) + assert all(abs(r["score"] - _admin_score(name)) < EPS for r in results), (name, results) def test_healthcare_admin_id_ignores_trailing_punctuation(): @@ -426,7 +435,7 @@ def test_healthcare_admin_id_ignores_trailing_punctuation(): ("US_REFERRAL_NUMBER", "Referral REF123456.", "REF123456"), ("US_PROVIDER_TAX_ID", "Provider EIN 12-3456789.", "12-3456789"), ): - _single(_rule(name), text, value, 0.7) + _single(_rule(name), text, value, _admin_score(name)) def test_healthcare_admin_id_length_boundaries(): @@ -440,7 +449,7 @@ def test_healthcare_admin_id_length_boundaries(): ("US_REFERRAL_NUMBER", "Referral REF123456", "REF123456"), ("US_REFERRAL_NUMBER", "Referral INF123456789012", "INF123456789012"), ): - _single(_rule(name), text, value, 0.7) + _single(_rule(name), text, value, _admin_score(name)) for name, text in ( ("US_PRIOR_AUTHORIZATION_NUMBER", "PA-12345"), ("US_PRIOR_AUTHORIZATION_NUMBER", "PA-1234567890123"), @@ -458,13 +467,13 @@ def test_healthcare_admin_id_length_boundaries(): def test_healthcare_admin_label_enables_bare_numeric_id(): for name, text, value, score in ( - ("US_PRIOR_AUTHORIZATION_NUMBER", "Prior authorization number: 987654321 approved.", "987654321", 0.7), - ("US_CLAIM_NUMBER", "Claim number: 1234567890123 was paid.", "1234567890123", 0.7), - ("US_CLAIM_NUMBER", "Claim ID 123456789012345 was paid.", "123456789012345", 0.7), + ("US_PRIOR_AUTHORIZATION_NUMBER", "Prior authorization number: 987654321 approved.", "987654321", 0.8), + ("US_CLAIM_NUMBER", "Claim number: 1234567890123 was paid.", "1234567890123", 0.8), + ("US_CLAIM_NUMBER", "Claim ID 123456789012345 was paid.", "123456789012345", 0.8), ("US_PRESCRIPTION_NUMBER", "Rx #1234567", "1234567", 0.6), ("US_PRESCRIPTION_NUMBER", "Prescription number: 7654321", "7654321", 0.7), ("US_PRESCRIPTION_NUMBER", "prescription 4455667", "4455667", 0.7), - ("US_REFERRAL_NUMBER", "Infusion referral number: 2025001234", "2025001234", 0.7), + ("US_REFERRAL_NUMBER", "Infusion referral number: 2025001234", "2025001234", 0.8), ): _single(_rule(name), text, value, score) # a claim label does not support a prescription number match @@ -575,7 +584,7 @@ def test_ca_postal_code_upstream_cases(): _check(rule, "K1A0A1", [(0, 6, 0.1, 0.1)]) # "postal code" is a context phrase: upstream expects 0.3, the engine boosts to 0.65 _check(rule, "My postal code is K1A 0A1 thanks", [(18, 25, 0.3, MAX)]) - _single(rule, "My postal code is K1A 0A1 thanks", "K1A 0A1", 0.65) + _single(rule, "My postal code is K1A 0A1 thanks", "K1A 0A1", 0.8) # keyword floor _check(rule, "From K1A 0A1 to M5V 3A8", [(5, 12, 0.3, 0.3), (16, 23, 0.3, 0.3)]) diff --git a/tests/test_recognizers_world.py b/tests/test_recognizers_world.py index ec2de5c..0b2dc56 100644 --- a/tests/test_recognizers_world.py +++ b/tests/test_recognizers_world.py @@ -49,7 +49,7 @@ def test_examples_validate_and_mutations_are_rejected(): def test_packs_are_region_gated_and_mapped(): - mapping = json.loads((Path(__file__).resolve().parent.parent / "fixtures" / "findings-mapping.json").read_text())[0] + mapping = json.loads((Path(__file__).resolve().parent.parent / "fixtures" / "findings-mapping-v2.json").read_text())[0] names = {r.name for r in load_all()} assert len(names) == len(load_all()) and len(regions()) >= 60 for rule in _new_rules(): @@ -118,7 +118,7 @@ def test_column_of_national_ids_classifies_with_every_pack(): def test_new_vendor_secret_formats(): from src.engine import tokens as tk - mapping = json.loads((Path(__file__).resolve().parent.parent / "fixtures" / "findings-mapping.json").read_text())[0] + mapping = json.loads((Path(__file__).resolve().parent.parent / "fixtures" / "findings-mapping-v2.json").read_text())[0] engine = DetectionEngine() checked = 0 for name in ( @@ -136,7 +136,7 @@ def test_new_vendor_secret_formats(): def test_every_vendor_format_passes_its_prefilter(): from src.engine import tokens as tk - mapping = json.loads((Path(__file__).resolve().parent.parent / "fixtures" / "findings-mapping.json").read_text())[0] + mapping = json.loads((Path(__file__).resolve().parent.parent / "fixtures" / "findings-mapping-v2.json").read_text())[0] misses = [] for name in {n for n, _ in tk.VENDOR_TOKEN_RULES}: hits = {d for d, *_ in tk.find_vendor_tokens("x = " + mapping[name]["sample_value"])} diff --git a/tests/test_recognizers_za_ng_ph_generic.py b/tests/test_recognizers_za_ng_ph_generic.py index a80688e..04614db 100644 --- a/tests/test_recognizers_za_ng_ph_generic.py +++ b/tests/test_recognizers_za_ng_ph_generic.py @@ -107,7 +107,7 @@ def test_rule_names_and_regions(): def test_category_and_severity_follow_findings_mapping(): - with open(os.path.join(_ROOT, "fixtures", "findings-mapping.json")) as fh: + with open(os.path.join(_ROOT, "fixtures", "findings-mapping-v2.json")) as fh: mapping = json.load(fh) if isinstance(mapping, list): # historical shape: a list wrapping one dict mapping = mapping[0] @@ -541,7 +541,7 @@ def test_ph_passport(): ], ) _check_exact("PH_PASSPORT", "P1234567A", [(0, 9, 0.1)]) - _check_exact("PH_PASSPORT", "Passport: EB1234567 is valid.", [(10, 19, 0.45)]) + _check_exact("PH_PASSPORT", "Passport: EB1234567 is valid.", [(10, 19, 0.8)]) # keyword floor def test_ph_mobile_number(): From c2345dcf225638759a1355b9badec34c7e03fd71 Mon Sep 17 00:00:00 2001 From: AyushAggarwal1 Date: Sat, 26 Sep 2026 18:42:58 +0530 Subject: [PATCH 2/3] lint --- .github/workflows/linting.yaml | 1 + 1 file changed, 1 insertion(+) diff --git a/.github/workflows/linting.yaml b/.github/workflows/linting.yaml index 382cde1..6959ba0 100644 --- a/.github/workflows/linting.yaml +++ b/.github/workflows/linting.yaml @@ -8,6 +8,7 @@ on: - synchronize - edited branches: + - mainline - main jobs: From 7f52c6f40e16e3cefdaeb47254f7c37811175588 Mon Sep 17 00:00:00 2001 From: AyushAggarwal1 Date: Sat, 26 Sep 2026 18:48:36 +0530 Subject: [PATCH 3/3] fixes --- scripts/benchmark_public_datasets.py | 54 +++++++++++++++++----------- src/engine/layers.py | 25 +++++++++++++ src/engine/policy.py | 2 +- src/engine/recognizers/us_ca.py | 2 +- tests/fixtures/accuracy_cases.json | 35 ++++++++++++++++++ tests/test_public_benchmark.py | 14 ++++---- 6 files changed, 104 insertions(+), 28 deletions(-) diff --git a/scripts/benchmark_public_datasets.py b/scripts/benchmark_public_datasets.py index 4fcfe16..28e44e8 100644 --- a/scripts/benchmark_public_datasets.py +++ b/scripts/benchmark_public_datasets.py @@ -81,10 +81,14 @@ def _load_categories() -> Dict[str, str]: "swift": {"SWIFT/BIC"}, "ssn": {"US SSN"}, "itin": {"US_ITIN", "US SSN"}, - "passport": {"US_PASSPORT", "UK_PASSPORT", "PASSPORT_MRZ", "IN PASSPORT", "DE_PASSPORT", "ES_PASSPORT", - "IT_PASSPORT", "FR_PASSPORT", "JP_PASSPORT", "KR_PASSPORT", "PH_PASSPORT", "ZA_PASSPORT"}, - "driver": {"US_DRIVER_LICENSE", "UK_DRIVING_LICENCE", "IT_DRIVER_LICENSE", "KR_DRIVER_LICENSE", - "ZA_DRIVER_LICENSE", "DE_FUEHRERSCHEIN"}, + "passport": { + "US_PASSPORT", "UK_PASSPORT", "PASSPORT_MRZ", "IN PASSPORT", "DE_PASSPORT", "ES_PASSPORT", + "IT_PASSPORT", "FR_PASSPORT", "JP_PASSPORT", "KR_PASSPORT", "PH_PASSPORT", "ZA_PASSPORT", + }, + "driver": { + "US_DRIVER_LICENSE", "UK_DRIVING_LICENCE", "IT_DRIVER_LICENSE", "KR_DRIVER_LICENSE", + "ZA_DRIVER_LICENSE", "DE_FUEHRERSCHEIN", + }, "ip": {"PII.IPAddress"}, "mac": {"MAC_ADDRESS"}, "imei": {"IMEI"}, @@ -95,8 +99,10 @@ def _load_categories() -> Dict[str, str]: "mrn": {"MEDICAL_RECORD_NUMBER"}, "health_member": {"US_HEALTH_INSURANCE_MEMBER_ID", "US_MBI"}, "vin": {"VIN"}, - "plate": {"UK_VEHICLE_REGISTRATION", "TR_LICENSE_PLATE", "DE_KFZ", "ZA_LICENSE_PLATE", - "IN_VEHICLE_REGISTRATION", "NG_VEHICLE_REGISTRATION"}, + "plate": { + "UK_VEHICLE_REGISTRATION", "TR_LICENSE_PLATE", "DE_KFZ", "ZA_LICENSE_PLATE", + "IN_VEHICLE_REGISTRATION", "NG_VEHICLE_REGISTRATION", + }, "user": {"PII.UserIdentifier"}, "postcode": {"UK_POSTCODE", "CA_POSTAL_CODE", "DE_PLZ", "Address"}, "regional": set(REGIONAL_DETECTORS), @@ -492,8 +498,10 @@ def score(records: List[Record], results: List[List[Dict[str, Any]]], spec: Dict f1 = (2 * precision * recall / (precision + recall)) if precision and recall else None def label_row(counters: Counter) -> Dict[str, Any]: - row = {"gold": counters["gold"], "tp": counters["tp"], "fn": counters["fn"], - "recall": round(counters["tp"] / counters["gold"], 3) if counters["gold"] else None} + row = { + "gold": counters["gold"], "tp": counters["tp"], "fn": counters["fn"], + "recall": round(counters["tp"] / counters["gold"], 3) if counters["gold"] else None, + } if counters["demo_domain_gold"]: row["demo_domain_gold"] = counters["demo_domain_gold"] if counters["gold_valid"] or counters["gold_invalid"]: @@ -513,8 +521,10 @@ def label_row(counters: Counter) -> Dict[str, Any]: "hits_on_unscored_spans": fp_kind["on_unscored_span"], }, "per_label": {label: label_row(c) for label, c in sorted(per_label.items())}, - "per_group": {group: {"gold": c["gold"], "recall": round(c["tp"] / c["gold"], 3) if c["gold"] else None} - for group, c in sorted(per_group.items())}, + "per_group": { + group: {"gold": c["gold"], "recall": round(c["tp"] / c["gold"], 3) if c["gold"] else None} + for group, c in sorted(per_group.items()) + }, "false_positives_by_detector": dict(fp_by_detector.most_common(20)), "unscored_label_spans": dict(unscored_labels.most_common()), } @@ -546,11 +556,13 @@ def run(dataset: str, args, engine, config) -> Dict[str, Any]: suffix = "-json" if by_value else "" with (args.output_dir / f"{dataset}{suffix}-hits.jsonl").open("w", encoding="utf-8") as handle: for record, hits in zip(kept, results): - handle.write(json.dumps({ - "id": record.id, "meta": record.meta, - "gold": [(label, start, end, value if by_value else "") for label, start, end, value in record.gold], - "hits": [{k: h[k] for k in ("detector", "start", "end", "confidence", *(("value",) if by_value else ()))} for h in hits], - }) + "\n") + handle.write( + json.dumps({ + "id": record.id, "meta": record.meta, + "gold": [(label, start, end, value if by_value else "") for label, start, end, value in record.gold], + "hits": [{k: h[k] for k in ("detector", "start", "end", "confidence", *(("value",) if by_value else ()))} for h in hits], + }) + "\n", + ) if args.show_misses: shown = 0 @@ -623,11 +635,13 @@ def summarize(output_dir: Path) -> Path: data = json.loads(path.read_text(encoding="utf-8")) if "tiers" in data: reports.append(data) - lines = ["# Public-dataset benchmark results", "", - "Engine: src/engine + src/pipeline, aggregation off, NER model per report. Matching: lenient span overlap " - "(value match in JSON-document mode). Precision counts accepted hits against false positives of kind no_gold and " - "wrong_type; recall is over target labels only. `recall_on_valid` is recall on gold values that pass their own " - "checksum, the fair column for LLM-generated numbers.", ""] + lines = [ + "# Public-dataset benchmark results", "", + "Engine: src/engine + src/pipeline, aggregation off, NER model per report. Matching: lenient span overlap " + "(value match in JSON-document mode). Precision counts accepted hits against false positives of kind no_gold and " + "wrong_type; recall is over target labels only. `recall_on_valid` is recall on gold values that pass their own " + "checksum, the fair column for LLM-generated numbers.", "", + ] lines += markdown_summary(reports).splitlines() lines += ["", "## Runs", ""] for report in reports: diff --git a/src/engine/layers.py b/src/engine/layers.py index 2347a42..951a4bd 100644 --- a/src/engine/layers.py +++ b/src/engine/layers.py @@ -205,6 +205,16 @@ def validate_email(email: str) -> Optional[str]: LOOSE_PHONE_RE = re.compile(r"\+?\(?\d[\d\s().\-/]{5,22}\d") +_PLACEHOLDER_PHONE_DIGITS = ("1234567890", "0123456789", "0000000000", "1111111111", "800123456") + + +def _placeholder_phone(text: str, start: int, end: int) -> bool: + """Documented example numbers (123-456-7890, 0800 123 456, all-zero runs) and decimal fragments (12.3456 78).""" + digits = re.sub(r"\D", "", text[start:end]) + if any(digits.endswith(k) for k in _PLACEHOLDER_PHONE_DIGITS): + return True + # 555-01xx numbers are fictional but stay reportable: the regression corpus holds real scans that contain them + return start > 1 and text[start - 1] == "." and text[start - 2].isdigit() _MONTHS = r"(?:jan|feb|mar|apr|may|jun|jul|aug|sep|sept|oct|nov|dec)[a-z]*\.?" DOB_REGEX = re.compile( r"\b(?:(?:19|20)\d{2}[-/.](?:0[1-9]|1[0-2])[-/.](?:0[1-9]|[12]\d|3[01])" # 1985-08-12 @@ -340,6 +350,8 @@ def scan_pii( for match in (phonenumbers.PhoneNumberMatcher(text, None) if digit_count >= 7 else ()): raw = match.raw_string seen_spans.add((match.start, match.end)) + if _placeholder_phone(text, match.start, match.end): + continue score = 0.85 if raw.lstrip().startswith(("+", "00")) else 0.5 if phone_context or near(text, match.start, match.end, CONTEXT_WORDS["Phone Number"]): score = max(score, 0.85) @@ -357,6 +369,8 @@ def scan_pii( # a bare national-format digit run is left to the checksum detectors (NHS, SSN, accounts) if not (phone_context or near(text, match.start, match.end, CONTEXT_WORDS["Phone Number"])): continue + if _placeholder_phone(text, match.start, match.end): + continue findings.append(_finding("Phone Number", _CATEGORY_PII, "Medium", match.raw_string, 0.85, match.start, match.end, region=region.upper())) except Exception: continue @@ -472,6 +486,15 @@ def scan_addresses(text: str, field_name: Optional[str] = None) -> list: "team", "group", "plan", "policy", "file", "user", "host", "domain", "server", "table", "column", "field", "display", "screen", "model", "device", "app", "application", "service", "agency", "firm", "corporation", "entity", "branch", "merchant", "issuer", "carrier", "insurer", "ship", "vessel", "property", "estate", "building", "street", "road", + "chemical", "strategy", "scheme", "programme", "program", "portfolio", "index", "asset", "security", "bond", "stock", + "ticker", "template", "report", "document", "dataset", "method", "process", "procedure", "task", "event", "item", + "material", "substance", "compound", "drug", "medication", "species", "variety", "breed", "trade", "dba", +} +# "Guarantor: Not Applicable", "Customer: Sure, ..." - a capitalised opener that is not a name +NON_NAME_OPENERS = { + "not", "applicable", "none", "sure", "yes", "no", "unknown", "see", "please", "thank", "thanks", "dear", "hello", "hi", + "the", "this", "that", "same", "other", "various", "multiple", "any", "all", "each", "per", "as", "to", "be", "n/a", + "ok", "okay", "yes,", "no,", "pending", "confirmed", "approved", "declined", "unavailable", "available", "required", } NAME_KEYS_PROSE_RE = re.compile( r"(?first name|last name|surname|given name|family name|middle name|full name|patient name|" @@ -507,6 +530,8 @@ def scan_person_names(text: str, field_name: Optional[str] = None) -> list: if qualifier and qualifier[-1].lower() in NON_PERSON_NAME_QUALIFIERS: continue key_single = key != "name" and _SINGLE_NAME_FIELD_RE.search(key) is not None + if val.split()[0].lower().rstrip(",.") in NON_NAME_OPENERS: + continue if looks_like_person_name(val) and (key_single or len(val.split()) >= 2): findings.append(_finding("PII.PersonName", _CATEGORY_PII, "Low", val, 0.85, match.start("val"), match.end("val"), masked=_mask_name(val), key=match.group("key"))) return findings diff --git a/src/engine/policy.py b/src/engine/policy.py index 199ed0c..273e968 100644 --- a/src/engine/policy.py +++ b/src/engine/policy.py @@ -180,7 +180,7 @@ def has_sibling(self, field_tokens: str) -> bool: "IMEI": DetectorPolicy(context=CONTEXT_REQUIRED, identity=False, negative_fields=_TECHNICAL_NUMBER_FIELDS), "ICCID": DetectorPolicy(context=CONTEXT_REQUIRED, identity=False, negative_fields=_TECHNICAL_NUMBER_FIELDS), "VIN": DetectorPolicy(context=CONTEXT_REQUIRED, identity=False, count_promotion=False), - "GEO_COORDINATES": DetectorPolicy(context=CONTEXT_REQUIRED, identity=False), + "GEO_COORDINATES": DetectorPolicy(context=CONTEXT_REQUIRED, identity=False, count_promotion=False), # exchange-rate tables "PASSPORT_MRZ": DetectorPolicy(context=CONTEXT_NONE, identity_corroboration=False), "IN_IFSC": DetectorPolicy(context=CONTEXT_REQUIRED, column_ratio=0.8, count_promotion=False), "AU_BSB": DetectorPolicy(context=CONTEXT_REQUIRED, column_ratio=0.8, count_promotion=False), diff --git a/src/engine/recognizers/us_ca.py b/src/engine/recognizers/us_ca.py index b76b526..5f63e36 100644 --- a/src/engine/recognizers/us_ca.py +++ b/src/engine/recognizers/us_ca.py @@ -307,7 +307,7 @@ def _labelled(labels: Sequence[str], body: str) -> str: r"\b([A-Z][0-9]{3,6}|[A-Z][0-9]{5,9}|[A-Z][0-9]{6,8}|[A-Z][0-9]{4,8}|[A-Z][0-9]{9,11}|[A-Z]{1,2}[0-9]{5,6}|H[0-9]{8}|V[0-9]{6}|X[0-9]{8}|A-Z]{2}[0-9]{2,5}|[A-Z]{2}[0-9]{3,7}|[0-9]{2}[A-Z]{3}[0-9]{5,6}|[A-Z][0-9]{13,14}|[A-Z][0-9]{18}|[A-Z][0-9]{6}R|[A-Z][0-9]{9}|[A-Z][0-9]{1,12}|[0-9]{9}[A-Z]|[A-Z]{2}[0-9]{6}[A-Z]|[0-9]{8}[A-Z]{2}|[0-9]{3}[A-Z]{2}[0-9]{4}|[A-Z][0-9][A-Z][0-9][A-Z]|[0-9]{7,8}[A-Z])\b", 0.3, ), - Pattern("Driver License (labelled)", r"\bdriver'?s?[ -]licen[cs]e(?: number| no\.?| #|#)?\s*[:#-]?\s*(?P[A-Z0-9]{5,14})\b", 0.85), + Pattern("Driver License (labelled)", r"\bdriver'?s?[ -]licen[cs]e(?: number| no\.?| #|#)?(?: of| is)?\s*[:#(-]?\s*(?P(?=[A-Z0-9-]*\d)[A-Z0-9][A-Z0-9-]{4,15})(?![A-Z0-9])", 0.85), Pattern("Driver License - Digits (very weak)", r"\b([0-9]{6,14}|[0-9]{16})\b", 0.01), ], context=["driver", "license", "permit", "lic", "identification", "dls", "cdls", "lic#", "driving"], diff --git a/tests/fixtures/accuracy_cases.json b/tests/fixtures/accuracy_cases.json index b5bfcc1..5137851 100644 --- a/tests/fixtures/accuracy_cases.json +++ b/tests/fixtures/accuracy_cases.json @@ -596,6 +596,41 @@ "location": "text" } ] + }, + { + "id": "labelled_licence_reports_the_number", + "purpose": "The driver licence label never reports its own word 'number' as the value.", + "text": "Verified with a driver's license number of G577-7354-4; the user's driver license number is: 39-147784-4.", + "expected": [ + { + "detector": "US_DRIVER_LICENSE", + "value": "G577-7354-4", + "location": "text" + }, + { + "detector": "US_DRIVER_LICENSE", + "value": "39-147784-4", + "location": "text" + } + ] + }, + { + "id": "placeholder_phones_stay_silent", + "purpose": "Documented example numbers next to a phone keyword are not phones.", + "text": "Phone Number: (123) 456-7890. Phone: +44-800-123-456. Position 12.3456 7890123 logged.", + "expected": [] + }, + { + "id": "capitalised_openers_are_not_names", + "purpose": "Role labels followed by non-name words stay silent; thing-owned name labels too.", + "text": "Guarantor: Not Applicable. Customer: Sure, it's fine. Chemical name: Ethyl Acetate. Strategy Name: Optimizing Liquidity", + "expected": [] + }, + { + "id": "rate_table_is_not_coordinates", + "purpose": "A table of four-decimal pairs without a location keyword is not promoted to coordinates by count.", + "text": "USD, MXN, 19.5234, 20.3456; GBP, BRL, 6.4521, 6.7890; CAD, RUB, 0.1321, 0.1456; EUR, TRY, 0.1789, 0.1923; SGD, CNY, 1.5213, 1.6547; CHF, ZAR, 0.1123, 0.1234; JPY, PLN, 0.0312, 0.0345; AUD, SEK, 0.5123, 0.5432; NZD, NOK, 0.6123, 0.6321; HKD, DKK, 0.7123, 0.7321; INR, CZK, 0.8123, 0.8321", + "expected": [] } ] } diff --git a/tests/test_public_benchmark.py b/tests/test_public_benchmark.py index e6e4a89..b5ac95c 100644 --- a/tests/test_public_benchmark.py +++ b/tests/test_public_benchmark.py @@ -30,12 +30,14 @@ def _hit(detector, start, end, confidence="likely", value=""): def test_scorer_counts_overlap_hits_wrong_types_and_tiers(): text = "mail a@b.com ssn 219-09-9999 site http://x.io free 123-45-6789 company Acme" - record = Record("r1", text, [ - ("email", 5, 12, "a@b.com"), # target, hit by Email - ("ssn", 17, 28, "219-09-9999"), # target, hit only by a Phone Number (wrong type) - ("url", 34, 45, "http://x.io"), # ambiguous, URL hit is fine - ("company", 71, 75, "Acme"), # unscored label - ], {}) + record = Record( + "r1", text, [ + ("email", 5, 12, "a@b.com"), # target, hit by Email + ("ssn", 17, 28, "219-09-9999"), # target, hit only by a Phone Number (wrong type) + ("url", 34, 45, "http://x.io"), # ambiguous, URL hit is fine + ("company", 71, 75, "Acme"), # unscored label + ], {}, + ) hits = [ _hit("Email", 5, 12), _hit("Phone Number", 17, 28),