Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions RELEASE_NOTES.md
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,10 @@

Notable changes to the `phileas-redact` package, most recent first.

## Unreleased

* The `ssn` filter no longer detects the `NN-NNNNNNN` form. That shape is an Employer Identification Number, not a Social Security Number, and is already covered by the `ein` filter. A policy that enables only `ssn` therefore no longer detects values such as `12-3456789`; enable `ein` to detect them. SSN forms (`NNN-NN-NNNN`, `NNN NN NNNN`, and nine digits with no hyphen) are unchanged.

## Version 1.1.0

* The `url` filter no longer absorbs the punctuation that ends a sentence. The host and path character sets include `.`, `,`, `;`, `!`, and `'`, so `Visit https://example.com/page, then` redacted the comma along with the URL and `Visit (https://example.com/page). Then` redacted the closing parenthesis and period. A trailing run of such characters is now left out of the span, while punctuation inside a path, query, or fragment is kept, as is a percent-encoded delimiter. A match that trims down to nothing but its scheme is dropped.
Expand Down
4 changes: 2 additions & 2 deletions docs/filters.md
Original file line number Diff line number Diff line change
Expand Up @@ -100,9 +100,9 @@ Detects major credit card number formats (Visa, Mastercard, American Express, Di

## ssn

Detects US Social Security Numbers in `NNN-NN-NNNN`, `NNN NN NNNN`, and `NNNNNNNNN` formats, and Taxpayer Identification Numbers in `NN-NNNNNNN`.
Detects US Social Security Numbers in `NNN-NN-NNNN`, `NNN NN NNNN`, and `NNNNNNNNN` formats.

A TIN span carries confidence `0.90`, below the `1.0` of the SSN forms, so the [`ein`](#ein) filter wins that shape wherever both filters are enabled.
The `NN-NNNNNNN` form is an EIN, not an SSN. Enable the [`ein`](#ein) filter to detect it. A policy that enables only `ssn` does not detect that shape.

```python
"identifiers": {
Expand Down
2 changes: 1 addition & 1 deletion docs/index.md
Original file line number Diff line number Diff line change
Expand Up @@ -40,7 +40,7 @@ print(result.filtered_text)
| `age` | `age` | Age references, numeric or spelled out |
| `emailAddress` | `email-address` | Email addresses |
| `creditCard` | `credit-card` | Credit card numbers |
| `ssn` | `ssn` | Social Security Numbers and TINs |
| `ssn` | `ssn` | Social Security Numbers |
| `phoneNumber` | `phone-number` | Phone numbers, international and US |
| `ipAddress` | `ip-address` | IPv4 and IPv6 addresses |
| `url` | `url` | HTTP/HTTPS URLs |
Expand Down
14 changes: 1 addition & 13 deletions phileas/filters/ssn_filter.py
Original file line number Diff line number Diff line change
Expand Up @@ -36,22 +36,10 @@
),
]

# TIN: NN-NNNNNNN, the shape the ein filter also detects. Scored below the SSN
# forms so `ein` wins the span wherever both filters are enabled.
_TIN_PATTERNS = [
re.compile(r"(?<![\w-])\d{2}-\d{7}(?![\w-])"),
]

_TIN_CONFIDENCE = 0.90


class SSNFilter(BaseFilter):
def __init__(self, config=None):
super().__init__(FilterType.SSN, config)

def detect(self, text: str, context: str = "default") -> List[Span]:
spans = self._detect_patterns(_PATTERNS, text, context)
spans.extend(
self._detect_patterns(_TIN_PATTERNS, text, context, confidence=_TIN_CONFIDENCE)
)
return spans
return self._detect_patterns(_PATTERNS, text, context)
21 changes: 13 additions & 8 deletions tests/test_ein_detection.py
Original file line number Diff line number Diff line change
Expand Up @@ -94,14 +94,12 @@ def test_not_detected(self, text):


class TestSSNDistinction:
"""Both filters claim ``NN-NNNNNNN``; ein outranks ssn on it. See issue #64."""
"""``NN-NNNNNNN`` is an EIN only. The SSN filter does not claim it (issue #84)."""

def test_ssn_claims_the_tin_form_at_lower_confidence(self):
spans = SSNFilter().detect("12-3456789")
assert [s.text for s in spans] == ["12-3456789"]
assert spans[0].confidence == 0.90
def test_ssn_does_not_claim_the_ein_form(self):
assert SSNFilter().detect("12-3456789") == []

def test_ein_claims_the_same_form_at_full_confidence(self):
def test_ein_claims_the_form_at_full_confidence(self):
spans = EINFilter().detect("12-3456789")
assert [s.text for s in spans] == ["12-3456789"]
assert spans[0].confidence == 1.0
Expand Down Expand Up @@ -135,11 +133,18 @@ def test_ein_wins_the_tin_form_when_both_are_enabled(self):
)
assert [(s.filter_type, s.text) for s in r.spans] == [("ein", "12-3456789")]

def test_ssn_alone_still_redacts_the_tin_form(self):
def test_ssn_alone_does_not_detect_the_ein_form(self):
r = run({"ssn": {"ssnFilterStrategies": [{"strategy": "REDACT"}]}},
"Tax ID 12-3456789.")
assert [(s.filter_type, s.text) for s in r.spans] == [("ssn", "12-3456789")]
assert r.spans == []
assert r.filtered_text == "Tax ID 12-3456789."

def test_ein_alone_redacts_the_form(self):
r = run({"ein": {"einFilterStrategies": [{"strategy": "REDACT"}]}},
"Tax ID 12-3456789.")
assert [(s.filter_type, s.text) for s in r.spans] == [("ein", "12-3456789")]
assert "12-3456789" not in r.filtered_text
assert "{{{REDACTED-ein}}}" in r.filtered_text

def test_both_enabled_bare_run_is_ssn(self):
r = run(
Expand Down
52 changes: 24 additions & 28 deletions tests/test_ssn_detection.py
Original file line number Diff line number Diff line change
Expand Up @@ -240,38 +240,34 @@ def test_redacted_end_to_end(self):
assert "{{{REDACTED-ssn}}}" in r.filtered_text


class TestTINForm:
"""`NN-NNNNNNN`, ported from Java's SsnFilter. See issue #64."""
class TestEINFormNotClaimed:
"""`NN-NNNNNNN` is an EIN. The SSN filter no longer claims it (issue #84)."""

@pytest.mark.parametrize("value", ["12-3456789", "98-7654321", "07-1234567"])
def test_tin_detected(self, value):
spans = SSNFilter().detect(f"Tax ID {value} on file")
assert [s.text for s in spans] == [value]
assert spans[0].confidence == 0.90
def test_ein_shape_not_detected_as_ssn(self, value):
assert SSNFilter().detect(f"Tax ID {value} on file") == []

def test_ssn_forms_keep_full_confidence(self):
for value in ["123-45-6789", "123456789", "123 45 6789"]:
assert SSNFilter().detect(value)[0].confidence == 1.0

@pytest.mark.parametrize(
"text",
[
"12-3456789-01",
"ID-12-3456789",
"2026-12-3456789",
"123-45-6789123-45-6789",
"12-34567890",
"112-3456789",
"12 3456789",
],
)
def test_tin_hyphen_boundaries(self, text):
assert [s.text for s in SSNFilter().detect(text) if s.confidence == 0.90] == []

def test_tin_does_not_overlap_the_ssn_forms(self):
for text in ["123-45-6789", "123456789", "123 45 6789"]:
spans = SSNFilter().detect(text)
for i, a in enumerate(spans):
for b in spans[i + 1:]:
assert not (a.character_start < b.character_end
and b.character_start < a.character_end)
def test_ssn_only_policy_does_not_redact_ein_shape(self):
from phileas.policy.policy import Policy
from phileas.services.filter_service import FilterService

policy = Policy.from_dict({"name": "t", "identifiers": {"ssn": {
"ssnFilterStrategies": [{"strategy": "REDACT"}]}}})
r = FilterService().filter(policy, "c", "d", "Tax ID 12-3456789.")
assert r.spans == []
assert r.filtered_text == "Tax ID 12-3456789."

def test_ein_policy_redacts_the_shape(self):
from phileas.policy.policy import Policy
from phileas.services.filter_service import FilterService

policy = Policy.from_dict({"name": "t", "identifiers": {"ein": {
"einFilterStrategies": [{"strategy": "REDACT"}]}}})
r = FilterService().filter(policy, "c", "d", "Tax ID 12-3456789.")
assert [(s.filter_type, s.text) for s in r.spans] == [("ein", "12-3456789")]
assert "12-3456789" not in r.filtered_text
assert "{{{REDACTED-ein}}}" in r.filtered_text