From 56010b78b232030780f0d40dfa47bcf84e5f4c1a Mon Sep 17 00:00:00 2001 From: Harsh Raj Singhania Date: Tue, 15 Sep 2026 18:17:36 +0530 Subject: [PATCH 1/3] Remove duplicate NN-NNNNNNN pattern from the SSN filter The EIN-shaped form is already detected by EINFilter. Leaving a copy in SSNFilter mislabels the value as an SSN when ein is not enabled. Fixes #84. --- phileas/filters/ssn_filter.py | 14 +------------- 1 file changed, 1 insertion(+), 13 deletions(-) diff --git a/phileas/filters/ssn_filter.py b/phileas/filters/ssn_filter.py index 19e5185..4ec9991 100644 --- a/phileas/filters/ssn_filter.py +++ b/phileas/filters/ssn_filter.py @@ -36,22 +36,10 @@ ), ] -# TIN: NN-NNNNNNN, the shape the ein filter also detects. Scored below the SSN -# forms so `ein` wins the span wherever both filters are enabled. -_TIN_PATTERNS = [ - re.compile(r"(? List[Span]: - spans = self._detect_patterns(_PATTERNS, text, context) - spans.extend( - self._detect_patterns(_TIN_PATTERNS, text, context, confidence=_TIN_CONFIDENCE) - ) - return spans + return self._detect_patterns(_PATTERNS, text, context) From 458dd95683e748ceeebd4f12c746e1359c84e009 Mon Sep 17 00:00:00 2001 From: Harsh Raj Singhania Date: Tue, 15 Sep 2026 18:18:16 +0530 Subject: [PATCH 2/3] Document that NN-NNNNNNN is an EIN, not an SSN --- RELEASE_NOTES.md | 4 ++++ docs/index.md | 2 +- 2 files changed, 5 insertions(+), 1 deletion(-) diff --git a/RELEASE_NOTES.md b/RELEASE_NOTES.md index 81de374..bcfbeda 100644 --- a/RELEASE_NOTES.md +++ b/RELEASE_NOTES.md @@ -2,6 +2,10 @@ Notable changes to the `phileas-redact` package, most recent first. +## Unreleased + +* The `ssn` filter no longer detects the `NN-NNNNNNN` form. That shape is an Employer Identification Number, not a Social Security Number, and is already covered by the `ein` filter. A policy that enables only `ssn` therefore no longer detects values such as `12-3456789`; enable `ein` to detect them. SSN forms (`NNN-NN-NNNN`, `NNN NN NNNN`, and nine digits with no hyphen) are unchanged. + ## Version 1.1.0 * The `url` filter no longer absorbs the punctuation that ends a sentence. The host and path character sets include `.`, `,`, `;`, `!`, and `'`, so `Visit https://example.com/page, then` redacted the comma along with the URL and `Visit (https://example.com/page). Then` redacted the closing parenthesis and period. A trailing run of such characters is now left out of the span, while punctuation inside a path, query, or fragment is kept, as is a percent-encoded delimiter. A match that trims down to nothing but its scheme is dropped. diff --git a/docs/index.md b/docs/index.md index bb573d4..5dc7d5b 100644 --- a/docs/index.md +++ b/docs/index.md @@ -40,7 +40,7 @@ print(result.filtered_text) | `age` | `age` | Age references, numeric or spelled out | | `emailAddress` | `email-address` | Email addresses | | `creditCard` | `credit-card` | Credit card numbers | -| `ssn` | `ssn` | Social Security Numbers and TINs | +| `ssn` | `ssn` | Social Security Numbers | | `phoneNumber` | `phone-number` | Phone numbers, international and US | | `ipAddress` | `ip-address` | IPv4 and IPv6 addresses | | `url` | `url` | HTTP/HTTPS URLs | From 94aeeb2ed322a15ed96f90f845c27e071e3450b0 Mon Sep 17 00:00:00 2001 From: Harsh Raj Singhania Date: Tue, 15 Sep 2026 18:19:36 +0530 Subject: [PATCH 3/3] Update SSN/EIN tests and filter docs for issue #84 --- docs/filters.md | 4 +-- tests/test_ein_detection.py | 21 +++++++++------ tests/test_ssn_detection.py | 52 +++++++++++++++++-------------------- 3 files changed, 39 insertions(+), 38 deletions(-) diff --git a/docs/filters.md b/docs/filters.md index bffaa98..dea75dc 100644 --- a/docs/filters.md +++ b/docs/filters.md @@ -100,9 +100,9 @@ Detects major credit card number formats (Visa, Mastercard, American Express, Di ## ssn -Detects US Social Security Numbers in `NNN-NN-NNNN`, `NNN NN NNNN`, and `NNNNNNNNN` formats, and Taxpayer Identification Numbers in `NN-NNNNNNN`. +Detects US Social Security Numbers in `NNN-NN-NNNN`, `NNN NN NNNN`, and `NNNNNNNNN` formats. -A TIN span carries confidence `0.90`, below the `1.0` of the SSN forms, so the [`ein`](#ein) filter wins that shape wherever both filters are enabled. +The `NN-NNNNNNN` form is an EIN, not an SSN. Enable the [`ein`](#ein) filter to detect it. A policy that enables only `ssn` does not detect that shape. ```python "identifiers": { diff --git a/tests/test_ein_detection.py b/tests/test_ein_detection.py index f8d6a29..bce9144 100644 --- a/tests/test_ein_detection.py +++ b/tests/test_ein_detection.py @@ -94,14 +94,12 @@ def test_not_detected(self, text): class TestSSNDistinction: - """Both filters claim ``NN-NNNNNNN``; ein outranks ssn on it. See issue #64.""" + """``NN-NNNNNNN`` is an EIN only. The SSN filter does not claim it (issue #84).""" - def test_ssn_claims_the_tin_form_at_lower_confidence(self): - spans = SSNFilter().detect("12-3456789") - assert [s.text for s in spans] == ["12-3456789"] - assert spans[0].confidence == 0.90 + def test_ssn_does_not_claim_the_ein_form(self): + assert SSNFilter().detect("12-3456789") == [] - def test_ein_claims_the_same_form_at_full_confidence(self): + def test_ein_claims_the_form_at_full_confidence(self): spans = EINFilter().detect("12-3456789") assert [s.text for s in spans] == ["12-3456789"] assert spans[0].confidence == 1.0 @@ -135,11 +133,18 @@ def test_ein_wins_the_tin_form_when_both_are_enabled(self): ) assert [(s.filter_type, s.text) for s in r.spans] == [("ein", "12-3456789")] - def test_ssn_alone_still_redacts_the_tin_form(self): + def test_ssn_alone_does_not_detect_the_ein_form(self): r = run({"ssn": {"ssnFilterStrategies": [{"strategy": "REDACT"}]}}, "Tax ID 12-3456789.") - assert [(s.filter_type, s.text) for s in r.spans] == [("ssn", "12-3456789")] + assert r.spans == [] + assert r.filtered_text == "Tax ID 12-3456789." + + def test_ein_alone_redacts_the_form(self): + r = run({"ein": {"einFilterStrategies": [{"strategy": "REDACT"}]}}, + "Tax ID 12-3456789.") + assert [(s.filter_type, s.text) for s in r.spans] == [("ein", "12-3456789")] assert "12-3456789" not in r.filtered_text + assert "{{{REDACTED-ein}}}" in r.filtered_text def test_both_enabled_bare_run_is_ssn(self): r = run( diff --git a/tests/test_ssn_detection.py b/tests/test_ssn_detection.py index f848d5b..41a4c85 100644 --- a/tests/test_ssn_detection.py +++ b/tests/test_ssn_detection.py @@ -240,38 +240,34 @@ def test_redacted_end_to_end(self): assert "{{{REDACTED-ssn}}}" in r.filtered_text -class TestTINForm: - """`NN-NNNNNNN`, ported from Java's SsnFilter. See issue #64.""" +class TestEINFormNotClaimed: + """`NN-NNNNNNN` is an EIN. The SSN filter no longer claims it (issue #84).""" @pytest.mark.parametrize("value", ["12-3456789", "98-7654321", "07-1234567"]) - def test_tin_detected(self, value): - spans = SSNFilter().detect(f"Tax ID {value} on file") - assert [s.text for s in spans] == [value] - assert spans[0].confidence == 0.90 + def test_ein_shape_not_detected_as_ssn(self, value): + assert SSNFilter().detect(f"Tax ID {value} on file") == [] def test_ssn_forms_keep_full_confidence(self): for value in ["123-45-6789", "123456789", "123 45 6789"]: assert SSNFilter().detect(value)[0].confidence == 1.0 - @pytest.mark.parametrize( - "text", - [ - "12-3456789-01", - "ID-12-3456789", - "2026-12-3456789", - "123-45-6789123-45-6789", - "12-34567890", - "112-3456789", - "12 3456789", - ], - ) - def test_tin_hyphen_boundaries(self, text): - assert [s.text for s in SSNFilter().detect(text) if s.confidence == 0.90] == [] - - def test_tin_does_not_overlap_the_ssn_forms(self): - for text in ["123-45-6789", "123456789", "123 45 6789"]: - spans = SSNFilter().detect(text) - for i, a in enumerate(spans): - for b in spans[i + 1:]: - assert not (a.character_start < b.character_end - and b.character_start < a.character_end) + def test_ssn_only_policy_does_not_redact_ein_shape(self): + from phileas.policy.policy import Policy + from phileas.services.filter_service import FilterService + + policy = Policy.from_dict({"name": "t", "identifiers": {"ssn": { + "ssnFilterStrategies": [{"strategy": "REDACT"}]}}}) + r = FilterService().filter(policy, "c", "d", "Tax ID 12-3456789.") + assert r.spans == [] + assert r.filtered_text == "Tax ID 12-3456789." + + def test_ein_policy_redacts_the_shape(self): + from phileas.policy.policy import Policy + from phileas.services.filter_service import FilterService + + policy = Policy.from_dict({"name": "t", "identifiers": {"ein": { + "einFilterStrategies": [{"strategy": "REDACT"}]}}}) + r = FilterService().filter(policy, "c", "d", "Tax ID 12-3456789.") + assert [(s.filter_type, s.text) for s in r.spans] == [("ein", "12-3456789")] + assert "12-3456789" not in r.filtered_text + assert "{{{REDACTED-ein}}}" in r.filtered_text