Skip to content

Commit f1a6fce

Browse files
Adopt locale-gated German regex defaults
Co-authored-by: Pranjal Parmar <pranjalrparmar@gmail.com>
1 parent cb71a59 commit f1a6fce

11 files changed

Lines changed: 86 additions & 56 deletions

File tree

‎README.md‎

Lines changed: 6 additions & 7 deletions
Original file line numberDiff line numberDiff line change
@@ -78,10 +78,9 @@ print(datafog.sanitize("Card: 4111-1111-1111-1111", engine="regex"))
7878

7979
## German Structured PII
8080

81-
German VAT IDs and German IBANs are detected by the default regex path because
82-
their country-code structure is specific enough for default-on screening.
83-
Broader German identifiers are available with explicit locale selection or
84-
explicit entity-type filtering.
81+
German structured PII is country-specific and opt-in. Use explicit locale
82+
selection or entity-type filtering when you want German VAT IDs, German IBANs,
83+
tax IDs, postal codes, passports, or residence permits.
8584

8685
```python
8786
import datafog
@@ -117,8 +116,8 @@ Use the engine that matches your accuracy and dependency constraints:
117116

118117
- `regex`:
119118
- Fastest and always available.
120-
- Best for structured entities: `EMAIL`, `PHONE`, `SSN`, `CREDIT_CARD`, `IP_ADDRESS`, `DE_VAT_ID`, `DE_IBAN`, `DATE`, `ZIP_CODE`.
121-
- Use `locales=["de"]` for broader German structured IDs such as `DE_TAX_ID`, `DE_POSTAL_CODE`, and passport or residence permit numbers.
119+
- Best for default structured entities: `EMAIL`, `PHONE`, `SSN`, `CREDIT_CARD`, `IP_ADDRESS`, `DATE`, `ZIP_CODE`.
120+
- Use `locales=["de"]` for German structured IDs such as `DE_VAT_ID`, `DE_IBAN`, `DE_TAX_ID`, `DE_POSTAL_CODE`, and passport or residence permit numbers.
122121
- `spacy`:
123122
- Requires `pip install datafog[nlp]`.
124123
- Useful for unstructured entities like person and organization names.
@@ -190,7 +189,7 @@ datafog replace-text "john@example.com"
190189
# Hash detected entities
191190
datafog hash-text "john@example.com"
192191

193-
# Enable broader German regex identifiers
192+
# Enable German regex identifiers
194193
datafog redact-text "Steuer-ID 12345678901" --locale de
195194
```
196195

‎datafog/core.py‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -221,7 +221,7 @@ def get_supported_entities(locales: List[str] | None = None) -> List[str]:
221221
Example:
222222
>>> entities = get_supported_entities()
223223
>>> print(entities)
224-
['EMAIL', 'PHONE', 'SSN', 'CREDIT_CARD', 'IP_ADDRESS', 'DE_VAT_ID', 'DE_IBAN', 'DATE', 'ZIP_CODE']
224+
['EMAIL', 'PHONE', 'SSN', 'CREDIT_CARD', 'IP_ADDRESS', 'DATE', 'ZIP_CODE']
225225
"""
226226
annotator = RegexAnnotator(locales=locales)
227227
legacy_map = {"DOB": "DATE", "ZIP": "ZIP_CODE"}

‎datafog/processing/text_processing/regex_annotator/regex_annotator.py‎

Lines changed: 8 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -38,9 +38,8 @@ class RegexAnnotator:
3838
"DE_PASSPORT_NUMBER",
3939
"DE_RESIDENCE_PERMIT_NUMBER",
4040
]
41-
DEFAULT_LOCALIZED_LABELS = ["DE_VAT_ID", "DE_IBAN"]
4241
LABELS = BASE_LABELS + GERMAN_LABELS
43-
DEFAULT_LABELS = BASE_LABELS + DEFAULT_LOCALIZED_LABELS
42+
DEFAULT_LABELS = BASE_LABELS
4443
SUPPORTED_LOCALES = {"de", "de-de", "de_de"}
4544
LOCALE_LABELS = {
4645
"de": GERMAN_LABELS,
@@ -103,13 +102,18 @@ def __init__(
103102
# Supports dashed and no-dash formats.
104103
"SSN": re.compile(
105104
r"""
106-
(?<!\d)
107105
(?:
106+
(?<!\d)
108107
(?!000|666)\d{3}-(?!00)\d{2}-(?!0000)\d{4}
108+
(?!\d)
109109
|
110+
(?<![A-Za-z0-9])
111+
(?<!DE)
112+
(?<!DE\s)
113+
(?<!DE-)
110114
(?!000|666)\d{3}(?!00)\d{2}(?!0000)\d{4}
115+
(?![A-Za-z0-9])
111116
)
112-
(?!\d)
113117
""",
114118
re.IGNORECASE | re.MULTILINE | re.VERBOSE,
115119
),

‎datafog/services/text_service.py‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -58,7 +58,7 @@ def __init__(
5858
- "smart": Try RegexAnnotator → GLiNER → SpaCy cascade (requires nlp-advanced extra)
5959
gliner_model: GLiNER model name to use when engine is "gliner" or "smart"
6060
locales: Optional locale tags for regex detection. Use ["de"] to enable
61-
broader German structured identifiers.
61+
German structured identifiers.
6262
6363
Raises:
6464
AssertionError: If an invalid engine type is provided

‎docs/cli.rst‎

Lines changed: 2 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -22,8 +22,8 @@ commands. Install ``datafog[distributed]`` when using ``SparkService``.
2222
German locale support
2323
---------------------
2424

25-
German VAT IDs and German IBANs are detected by the default regex path. Broader
26-
German identifiers are opt-in through ``--locale de`` on the core text commands:
25+
German structured PII is opt-in through ``--locale de`` on the core text
26+
commands:
2727

2828
.. code-block:: bash
2929

‎docs/getting-started.rst‎

Lines changed: 8 additions & 9 deletions
Original file line numberDiff line numberDiff line change
@@ -83,22 +83,21 @@ Agent-oriented helpers use the same lightweight text path:
8383
German Structured PII
8484
=====================
8585

86-
The core regex engine includes German VAT IDs and German IBANs by default
87-
because they carry strong country-code structure:
86+
German structured PII is country-specific and opt-in, including German VAT IDs
87+
and German IBANs:
8888

8989
.. code-block:: python
9090
9191
import datafog
9292
93-
result = datafog.scan("USt-IdNr DE 123456789", engine="regex")
93+
result = datafog.scan("USt-IdNr DE 123456789", engine="regex", locales=["de"])
9494
print([(entity.type, entity.text) for entity in result.entities])
9595
96-
Broader German identifiers such as ``DE_TAX_ID``,
97-
``DE_SOCIAL_SECURITY_NUMBER``, ``DE_POSTAL_CODE``,
98-
``DE_PASSPORT_NUMBER``, and ``DE_RESIDENCE_PERMIT_NUMBER`` require explicit
99-
German locale selection or explicit ``entity_types`` filtering. This keeps
100-
ordinary ticket, SKU, order, and invoice IDs from becoming default-on false
101-
positives.
96+
German identifiers such as ``DE_VAT_ID``, ``DE_IBAN``, ``DE_TAX_ID``,
97+
``DE_SOCIAL_SECURITY_NUMBER``, ``DE_POSTAL_CODE``, ``DE_PASSPORT_NUMBER``, and
98+
``DE_RESIDENCE_PERMIT_NUMBER`` require explicit German locale selection or
99+
explicit ``entity_types`` filtering. This keeps ordinary ticket, SKU, order,
100+
and invoice IDs from becoming default-on false positives.
102101

103102
.. code-block:: python
104103

‎docs/python-sdk.rst‎

Lines changed: 7 additions & 8 deletions
Original file line numberDiff line numberDiff line change
@@ -31,11 +31,11 @@ German locale coverage
3131
----------------------
3232

3333
DataFog 4.5 includes regex-only German structured PII support without adding
34-
dependencies. German VAT IDs and German IBANs are active in the default regex
35-
path. Broader German-only identifiers are opt-in because their raw shapes are
36-
common in ordinary product, ticket, invoice, and order data.
34+
dependencies. German-only identifiers are opt-in because their raw shapes are
35+
country-specific or common in ordinary product, ticket, invoice, and order
36+
data.
3737

38-
Use ``locales=["de"]`` to enable the broader German set:
38+
Use ``locales=["de"]`` to enable the German set:
3939

4040
.. code-block:: python
4141
@@ -55,10 +55,9 @@ You can also request one German entity type directly:
5555
entity_types=["DE_TAX_ID"],
5656
)
5757
58-
The opt-in German set currently covers ``DE_TAX_ID``,
59-
``DE_SOCIAL_SECURITY_NUMBER``, ``DE_POSTAL_CODE``,
60-
``DE_PASSPORT_NUMBER``, and ``DE_RESIDENCE_PERMIT_NUMBER``. The default set
61-
also covers ``DE_VAT_ID`` and ``DE_IBAN``.
58+
The opt-in German set currently covers ``DE_VAT_ID``, ``DE_IBAN``,
59+
``DE_TAX_ID``, ``DE_SOCIAL_SECURITY_NUMBER``, ``DE_POSTAL_CODE``,
60+
``DE_PASSPORT_NUMBER``, and ``DE_RESIDENCE_PERMIT_NUMBER``.
6261

6362
Optional services
6463
-----------------

‎tests/corpus/structured_pii.json‎

Lines changed: 4 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -735,8 +735,9 @@
735735
]
736736
},
737737
{
738-
"id": "de-vat-id-default",
738+
"id": "de-vat-id-locale",
739739
"input": "USt-IdNr DE 123456789 ist gesetzt.",
740+
"locales": ["de"],
740741
"expected_entities": [
741742
{
742743
"type": "DE_VAT_ID",
@@ -747,8 +748,9 @@
747748
]
748749
},
749750
{
750-
"id": "de-iban-default",
751+
"id": "de-iban-locale",
751752
"input": "IBAN DE44 5001 0517 5407 3249 31 ist gueltig.",
753+
"locales": ["de"],
752754
"expected_entities": [
753755
{
754756
"type": "DE_IBAN",

‎tests/test_de_pii_regex.py‎

Lines changed: 33 additions & 14 deletions
Original file line numberDiff line numberDiff line change
@@ -24,12 +24,17 @@
2424
),
2525
],
2626
)
27-
def test_high_specificity_german_regex_default_cases(
27+
def test_german_regex_cases_require_german_locale_or_explicit_entity_type(
2828
label: str, text: str, expected: str
2929
) -> None:
30-
annotator = RegexAnnotator()
31-
result = annotator.annotate(text)
32-
assert expected in result[label]
30+
default_result = RegexAnnotator().annotate(text)
31+
assert expected not in default_result[label]
32+
33+
german_result = RegexAnnotator(locales=["de"]).annotate(text)
34+
assert expected in german_result[label]
35+
36+
explicit_result = RegexAnnotator(enabled_labels=[label]).annotate(text)
37+
assert expected in explicit_result[label]
3338

3439

3540
@pytest.mark.parametrize(
@@ -118,25 +123,39 @@ def test_redaction_and_service_locale_support() -> None:
118123
assert service_result["DE_PASSPORT_NUMBER"] == ["C12345678"]
119124

120125

121-
def test_german_vat_redaction_suppresses_inner_generic_ssn_match() -> None:
122-
text = "USt-IdNr DE123456789 ist gesetzt."
123-
126+
@pytest.mark.parametrize(
127+
"text,vat_text",
128+
[
129+
("USt-IdNr DE123456789 ist gesetzt.", "DE123456789"),
130+
("USt-IdNr DE 123456789 ist gesetzt.", "DE 123456789"),
131+
("USt-IdNr DE-123456789 ist gesetzt.", "DE-123456789"),
132+
],
133+
)
134+
def test_german_vat_redaction_suppresses_inner_generic_ssn_match(
135+
text: str, vat_text: str
136+
) -> None:
124137
scan_result = scan(text, engine="regex")
125-
assert [(entity.type, entity.text) for entity in scan_result.entities] == [
126-
("DE_VAT_ID", "DE123456789")
138+
assert scan_result.entities == []
139+
140+
locale_scan_result = scan(text, engine="regex", locales=["de"])
141+
assert [(entity.type, entity.text) for entity in locale_scan_result.entities] == [
142+
("DE_VAT_ID", vat_text)
127143
]
128144

129-
redaction = scan_and_redact(text, engine="regex")
130-
assert redaction.redacted_text == "USt-IdNr [DE_VAT_ID_1] ist gesetzt."
145+
default_redaction = scan_and_redact(text, engine="regex")
146+
assert default_redaction.redacted_text == text
147+
148+
redaction = scan_and_redact(text, engine="regex", locales=["de"])
149+
assert redaction.redacted_text == text.replace(vat_text, "[DE_VAT_ID_1]")
131150

132151

133152
def test_top_level_helpers_and_supported_entities_respect_locale() -> None:
134153
default_entities = get_supported_entities()
135-
assert "DE_VAT_ID" in default_entities
136-
assert "DE_IBAN" in default_entities
137-
assert "DE_TAX_ID" not in default_entities
154+
assert all(not entity.startswith("DE_") for entity in default_entities)
138155

139156
german_entities = get_supported_entities(locales=["de"])
157+
assert "DE_VAT_ID" in german_entities
158+
assert "DE_IBAN" in german_entities
140159
assert "DE_TAX_ID" in german_entities
141160
assert "DE_RESIDENCE_PERMIT_NUMBER" in german_entities
142161

‎tests/test_detection_accuracy.py‎

Lines changed: 10 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -280,9 +280,11 @@ def _canon_type(entity_type: str) -> str:
280280
return TYPE_ALIASES.get(raw, raw)
281281

282282

283-
def _extract_entities(text: str, engine: str) -> list[dict[str, Any]]:
283+
def _extract_entities(
284+
text: str, engine: str, locales: list[str] | None = None
285+
) -> list[dict[str, Any]]:
284286
try:
285-
result = scan(text=text, engine=engine)
287+
result = scan(text=text, engine=engine, locales=locales)
286288
except (ImportError, EngineNotAvailable) as exc:
287289
pytest.skip(f"{engine} engine unavailable in this environment: {exc}")
288290

@@ -347,7 +349,7 @@ def _assert_expected_found(
347349
case: dict[str, Any], engine: str, corpus_kind: str
348350
) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]:
349351
text = case["input"]
350-
actual = _extract_entities(text, engine)
352+
actual = _extract_entities(text, engine, locales=case.get("locales"))
351353
expected = _required_expected(case["expected_entities"], engine, corpus_kind)
352354

353355
for exp in expected:
@@ -403,7 +405,9 @@ def _compute_metrics(
403405
for engine in engines:
404406
for corpus_kind, cases in corpora:
405407
for case in cases:
406-
actual = _extract_entities(case["input"], engine)
408+
actual = _extract_entities(
409+
case["input"], engine, locales=case.get("locales")
410+
)
407411
expected = _required_expected(
408412
case["expected_entities"], engine, corpus_kind
409413
)
@@ -490,7 +494,7 @@ def test_structured_pii_detection_slow(case: dict[str, Any], engine: str) -> Non
490494
@pytest.mark.parametrize("engine", FAST_ENGINES)
491495
def test_negative_cases_fast(case: dict[str, Any], engine: str) -> None:
492496
_xfail_if_known_limitation(case, engine, "negative")
493-
actual = _extract_entities(case["input"], engine)
497+
actual = _extract_entities(case["input"], engine, locales=case.get("locales"))
494498
assert not actual, f"{case['id']} ({engine}) false positives: {actual}"
495499

496500

@@ -501,7 +505,7 @@ def test_negative_cases_fast(case: dict[str, Any], engine: str) -> None:
501505
@pytest.mark.parametrize("engine", SLOW_ENGINES)
502506
def test_negative_cases_slow(case: dict[str, Any], engine: str) -> None:
503507
_xfail_if_known_limitation(case, engine, "negative")
504-
actual = _extract_entities(case["input"], engine)
508+
actual = _extract_entities(case["input"], engine, locales=case.get("locales"))
505509
assert not actual, f"{case['id']} ({engine}) false positives: {actual}"
506510

507511

0 commit comments

Comments
 (0)