diff --git a/CHANGELOG.md b/CHANGELOG.md index 5e40d9dab5..dd9e50be4a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,8 @@ All notable changes to this project will be documented in this file. ### Analyzer #### Added +- Added a disabled-by-default US health insurance member ID (`US_HEALTH_INSURANCE_MEMBER_ID`) recognizer requiring healthcare or insurance context. +- Added disabled-by-default US healthcare administrative ID recognizers for claim numbers, prior authorization numbers, prescription numbers, provider tax IDs, and referral numbers. - Added `UuidRecognizer` (generic, entity type `UUID`) to detect UUIDs in the standard 8-4-4-4-12 hyphenated hexadecimal format, covering RFC 4122 versions 1-5 and RFC 9562 versions 6-8. Validates version and variant nibbles and filters the nil UUID to reduce false positives. - South African ID number (`ZA_ID_NUMBER`) recognizer for the 13-digit national identity number, using pattern matching, context words, birth-date validation, and Luhn checksum validation. Disabled by default. - South African recognizers for `ZA_PASSPORT`, `ZA_INCOME_TAX_NUMBER`, `ZA_DRIVER_LICENSE`, `ZA_VAT_NUMBER`, `ZA_COMPANY_REGISTRATION`, `ZA_TRAFFIC_REGISTER_NUMBER`, `ZA_LICENSE_PLATE`, `ZA_MOBILE_NUMBER`, and `ZA_TELEPHONE_NUMBER`. All disabled by default. diff --git a/docs/supported_entities.md b/docs/supported_entities.md index 3e14d1af6a..209fbc06ea 100644 --- a/docs/supported_entities.md +++ b/docs/supported_entities.md @@ -34,9 +34,15 @@ For more information, refer to the [adding new recognizers documentation](analyz |US_BANK_NUMBER|A US bank account number is between 8 to 17 digits.|Pattern match and context| |US_DRIVER_LICENSE|A US driver license according to |Pattern match and context| |US_ITIN | US Individual Taxpayer Identification Number (ITIN). Nine digits that start with a "9" and contain a "7" or "8" as the 4 digit.|Pattern match and context| +|US_CLAIM_NUMBER|A US healthcare claim identifier used in billing and claims processing.|Pattern match, context enhancement, and entity threshold| +|US_HEALTH_INSURANCE_MEMBER_ID|A US health insurance member or subscriber identifier printed on an insurance card. Healthcare or insurance context increases detection confidence.|Pattern match, context enhancement, and entity threshold| |US_MBI|A US Medicare Beneficiary Identifier (MBI) with 11 alphanumeric characters.|Pattern match and context| |US_NPI|A US National Provider Identifier (NPI) is a 10-digit number issued to healthcare providers by CMS under HIPAA.|Pattern match, context and checksum| |US_PASSPORT |A US passport number with 9 digits.|Pattern match and context| +|US_PRESCRIPTION_NUMBER|A US prescription or pharmacy order identifier.|Pattern match, context enhancement, and entity threshold| +|US_PRIOR_AUTHORIZATION_NUMBER|A US prior authorization identifier used for treatment or drug approval requests.|Pattern match, context enhancement, and entity threshold| +|US_PROVIDER_TAX_ID|A US provider organization tax identifier (TIN/EIN) used in healthcare billing workflows.|Pattern match, context enhancement, and entity threshold| +|US_REFERRAL_NUMBER|A US healthcare referral identifier, including specialty or infusion referral numbers.|Pattern match, context enhancement, and entity threshold| |US_SSN|A US Social Security Number (SSN) with 9 digits.|Pattern match and context| ### UK diff --git a/presidio-analyzer/presidio_analyzer/conf/default_recognizers.yaml b/presidio-analyzer/presidio_analyzer/conf/default_recognizers.yaml index 35f7f6adf9..90fc678560 100644 --- a/presidio-analyzer/presidio_analyzer/conf/default_recognizers.yaml +++ b/presidio-analyzer/presidio_analyzer/conf/default_recognizers.yaml @@ -91,6 +91,48 @@ recognizers: enabled: false country_code: us + - name: UsHealthInsuranceMemberIdRecognizer + supported_languages: + - en + type: predefined + enabled: false + country_code: us + + - name: UsPriorAuthorizationNumberRecognizer + supported_languages: + - en + type: predefined + enabled: false + country_code: us + + - name: UsClaimNumberRecognizer + supported_languages: + - en + type: predefined + enabled: false + country_code: us + + - name: UsPrescriptionNumberRecognizer + supported_languages: + - en + type: predefined + enabled: false + country_code: us + + - name: UsReferralNumberRecognizer + supported_languages: + - en + type: predefined + enabled: false + country_code: us + + - name: UsProviderTaxIdRecognizer + supported_languages: + - en + type: predefined + enabled: false + country_code: us + - name: NhsRecognizer supported_languages: - en diff --git a/presidio-analyzer/presidio_analyzer/predefined_recognizers/__init__.py b/presidio-analyzer/presidio_analyzer/predefined_recognizers/__init__.py index a4aa2f8abc..285a5a21d2 100644 --- a/presidio-analyzer/presidio_analyzer/predefined_recognizers/__init__.py +++ b/presidio-analyzer/presidio_analyzer/predefined_recognizers/__init__.py @@ -157,6 +157,16 @@ from .country_specific.us.medical_license_recognizer import MedicalLicenseRecognizer from .country_specific.us.us_bank_recognizer import UsBankRecognizer from .country_specific.us.us_driver_license_recognizer import UsLicenseRecognizer +from .country_specific.us.us_health_insurance_member_id_recognizer import ( + UsHealthInsuranceMemberIdRecognizer, +) +from .country_specific.us.us_healthcare_admin_recognizers import ( + UsClaimNumberRecognizer, + UsPrescriptionNumberRecognizer, + UsPriorAuthorizationNumberRecognizer, + UsProviderTaxIdRecognizer, + UsReferralNumberRecognizer, +) from .country_specific.us.us_itin_recognizer import UsItinRecognizer from .country_specific.us.us_mbi_recognizer import UsMbiRecognizer from .country_specific.us.us_npi_recognizer import UsNpiRecognizer @@ -229,11 +239,17 @@ "SgFinRecognizer", "UrlRecognizer", "UsBankRecognizer", + "UsClaimNumberRecognizer", + "UsHealthInsuranceMemberIdRecognizer", "UsItinRecognizer", "UsLicenseRecognizer", "UsMbiRecognizer", "UsNpiRecognizer", "UsPassportRecognizer", + "UsPrescriptionNumberRecognizer", + "UsPriorAuthorizationNumberRecognizer", + "UsProviderTaxIdRecognizer", + "UsReferralNumberRecognizer", "UsSsnRecognizer", "EsNifRecognizer", "SpacyRecognizer", diff --git a/presidio-analyzer/presidio_analyzer/predefined_recognizers/country_specific/us/__init__.py b/presidio-analyzer/presidio_analyzer/predefined_recognizers/country_specific/us/__init__.py index 6e80dbbaef..df013a80d0 100644 --- a/presidio-analyzer/presidio_analyzer/predefined_recognizers/country_specific/us/__init__.py +++ b/presidio-analyzer/presidio_analyzer/predefined_recognizers/country_specific/us/__init__.py @@ -4,6 +4,16 @@ from .medical_license_recognizer import MedicalLicenseRecognizer from .us_bank_recognizer import UsBankRecognizer from .us_driver_license_recognizer import UsLicenseRecognizer +from .us_health_insurance_member_id_recognizer import ( + UsHealthInsuranceMemberIdRecognizer, +) +from .us_healthcare_admin_recognizers import ( + UsClaimNumberRecognizer, + UsPrescriptionNumberRecognizer, + UsPriorAuthorizationNumberRecognizer, + UsProviderTaxIdRecognizer, + UsReferralNumberRecognizer, +) from .us_itin_recognizer import UsItinRecognizer from .us_mbi_recognizer import UsMbiRecognizer from .us_npi_recognizer import UsNpiRecognizer @@ -15,9 +25,15 @@ "UsItinRecognizer", "UsBankRecognizer", "UsLicenseRecognizer", + "UsClaimNumberRecognizer", + "UsHealthInsuranceMemberIdRecognizer", "UsMbiRecognizer", "UsNpiRecognizer", "UsPassportRecognizer", + "UsPrescriptionNumberRecognizer", + "UsPriorAuthorizationNumberRecognizer", + "UsProviderTaxIdRecognizer", + "UsReferralNumberRecognizer", "AbaRoutingRecognizer", "UsSsnRecognizer", ] diff --git a/presidio-analyzer/presidio_analyzer/predefined_recognizers/country_specific/us/us_health_insurance_member_id_recognizer.py b/presidio-analyzer/presidio_analyzer/predefined_recognizers/country_specific/us/us_health_insurance_member_id_recognizer.py new file mode 100644 index 0000000000..0a8e43993f --- /dev/null +++ b/presidio-analyzer/presidio_analyzer/predefined_recognizers/country_specific/us/us_health_insurance_member_id_recognizer.py @@ -0,0 +1,75 @@ +"""Recognizer for US health insurance member identifiers.""" + +from typing import Dict, List, Optional + +from presidio_analyzer import Pattern, PatternRecognizer + + +class UsHealthInsuranceMemberIdRecognizer(PatternRecognizer): + """Recognize US health insurance member/subscriber IDs with context. + + US health insurance member identifiers are payer-specific and do not have a + single universal checksum or format. To avoid broad matching of generic + alphanumeric IDs, this recognizer requires both: + - a plausible alphanumeric member ID pattern, and + - nearby healthcare/insurance context. + + CMS consumer guidance explicitly labels the payer-assigned member number on + a sample insurance card. Medicaid T-MSIS defines MEMBER-ID as the value shown + on the insurance carrier's card and permits up to 20 characters. These + sources establish the identifier and upper bound, not a universal syntax; + the default regex is therefore a conservative, replaceable heuristic. + Presidio applies ``re.IGNORECASE`` through its default global regex flags, + so the uppercase character classes also match lowercase and mixed-case IDs. + + CMS card reference: https://www.cms.gov/files/document/11818-sample-insurance-card-english.pdf + Medicaid data reference: https://www.medicaid.gov/tmsis/dataguide/v4/data-elements/tpl003036/ + + :param patterns: List of patterns to be used by this recognizer + :param context: List of context words which increase detection confidence + :param supported_language: Language this recognizer supports + :param supported_entity: The entity this recognizer can detect + :param score_thresholds: Optional default and entity-specific score thresholds + """ + + COUNTRY_CODE = "us" + + PATTERNS = [ + Pattern( + "Health insurance member ID (weak)", + r"\b(?=[A-Z0-9-]{6,20}\b)(?=[A-Z0-9-]*[A-Z])" + r"(?=[A-Z0-9-]*\d)[A-Z]{1,5}-?[A-Z0-9]{5,14}\b", + 0.1, + ), + ] + + CONTEXT = [ + "member", + "subscriber", + "insurance", + "policy", + ] + + def __init__( + self, + patterns: Optional[List[Pattern]] = None, + context: Optional[List[str]] = None, + supported_language: str = "en", + supported_entity: str = "US_HEALTH_INSURANCE_MEMBER_ID", + name: Optional[str] = None, + score_thresholds: Optional[Dict[str, float]] = None, + ): + patterns = patterns if patterns else self.PATTERNS + context = context if context else self.CONTEXT + super().__init__( + supported_entity=supported_entity, + patterns=patterns, + context=context, + supported_language=supported_language, + name=name, + ) + self.score_thresholds = ( + score_thresholds + if score_thresholds is not None + else {supported_entity: 0.4} + ) diff --git a/presidio-analyzer/presidio_analyzer/predefined_recognizers/country_specific/us/us_healthcare_admin_recognizers.py b/presidio-analyzer/presidio_analyzer/predefined_recognizers/country_specific/us/us_healthcare_admin_recognizers.py new file mode 100644 index 0000000000..b66a4401ff --- /dev/null +++ b/presidio-analyzer/presidio_analyzer/predefined_recognizers/country_specific/us/us_healthcare_admin_recognizers.py @@ -0,0 +1,317 @@ +"""Recognizers for US healthcare administrative identifiers.""" + +from typing import Dict, List, Optional + +from presidio_analyzer import Pattern, PatternRecognizer + + +class UsPriorAuthorizationNumberRecognizer(PatternRecognizer): + """Recognize US healthcare prior authorization numbers with context. + + CMS identifies prior authorization and referral numbers as payer-assigned + values. There is no universal US syntax. The primary pattern anchors a + numeric identifier on its label, while a weak prefixed pattern supports + structured data containing values such as ``PA-987654321``. + + Reference: https://www.cms.gov/outreach-and-education/mln/wbt/mln4462429-mln-wbt-1500/1500/lesson04/18/index.html + """ + + COUNTRY_CODE = "us" + + PATTERNS = [ + Pattern( + "Prior authorization number (labelled)", + r"(?<=\b(?:prior\s+authorization|prior\s+auth|preauthorization|" + r"pre-auth|authorization)(?:\s*(?:#|no\.?|number|id)\s*:?\s*|" + r"\s*:\s*|\s+))" + r"(?:PA-?)?\d{6,12}\b", + 0.35, + ), + Pattern( + "Prior authorization number (weak prefixed)", + r"\bPA-?\d{6,12}\b", + 0.1, + ), + ] + + CONTEXT = [ + "authorization", + "auth", + "preauthorization", + "approval", + ] + + def __init__( + self, + patterns: Optional[List[Pattern]] = None, + context: Optional[List[str]] = None, + supported_language: str = "en", + supported_entity: str = "US_PRIOR_AUTHORIZATION_NUMBER", + name: Optional[str] = None, + score_thresholds: Optional[Dict[str, float]] = None, + ): + patterns = patterns if patterns else self.PATTERNS + context = context if context else self.CONTEXT + super().__init__( + supported_entity=supported_entity, + patterns=patterns, + context=context, + supported_language=supported_language, + name=name, + ) + self.score_thresholds = ( + score_thresholds + if score_thresholds is not None + else {supported_entity: 0.6} + ) + + +class UsClaimNumberRecognizer(PatternRecognizer): + """Recognize US healthcare claim numbers with billing/claims context. + + CMS describes a claim number as the reference number shown on an + explanation of benefits, but does not prescribe a universal syntax. The + primary pattern anchors a numeric identifier on its claim label, while a + weak prefixed pattern supports structured data containing ``CLM`` values. + + Reference: https://www.cms.gov/medical-bill-rights/help/guides/explanation-of-benefits + """ + + COUNTRY_CODE = "us" + + PATTERNS = [ + Pattern( + "Claim number (labelled)", + r"(?<=\b(?:claim|medical\s+claim|healthcare\s+claim)" + r"(?:\s*(?:#|no\.?|number|id)\s*:?\s*|\s*:\s*|\s+))" + r"(?:CLM-?)?\d{6,15}\b", + 0.35, + ), + Pattern( + "Claim number (weak prefixed)", + r"\bCLM-?\d{6,15}\b", + 0.1, + ), + ] + + CONTEXT = [ + "claim", + "billing", + ] + + def __init__( + self, + patterns: Optional[List[Pattern]] = None, + context: Optional[List[str]] = None, + supported_language: str = "en", + supported_entity: str = "US_CLAIM_NUMBER", + name: Optional[str] = None, + score_thresholds: Optional[Dict[str, float]] = None, + ): + patterns = patterns if patterns else self.PATTERNS + context = context if context else self.CONTEXT + super().__init__( + supported_entity=supported_entity, + patterns=patterns, + context=context, + supported_language=supported_language, + name=name, + ) + self.score_thresholds = ( + score_thresholds + if score_thresholds is not None + else {supported_entity: 0.6} + ) + + +class UsPrescriptionNumberRecognizer(PatternRecognizer): + """Recognize US prescription numbers with pharmacy context. + + CMS defines the prescription/service reference number as a pharmacy-assigned + alphanumeric value. Because there is no universal syntax, the primary + pattern anchors a numeric identifier on an ``Rx`` or ``prescription`` label. + A weak prefixed pattern remains available for structured data. + + Reference: https://www.cms.gov/files/document/cms-medicare-part-d-340b-repository-companion-guide-v-1.pdf + """ + + COUNTRY_CODE = "us" + + PATTERNS = [ + Pattern( + "Prescription number (Rx labelled)", + r"(?<=\brx(?:\s*(?:#|no\.?|number|id)\s*:?\s*|\s*:\s*|\s+))" + r"(?:RX-?)?\d{6,12}\b", + 0.6, + ), + Pattern( + "Prescription number (labelled)", + r"(?<=\bprescription" + r"(?:\s*(?:#|no\.?|number|id)\s*:?\s*|\s*:\s*|\s+))" + r"(?:RX-?)?\d{6,12}\b", + 0.35, + ), + Pattern( + "Prescription number (weak prefixed)", + r"\bRX-?\d{6,12}\b", + 0.1, + ), + ] + + CONTEXT = [ + "prescription", + "pharmacy", + "medication", + ] + + def __init__( + self, + patterns: Optional[List[Pattern]] = None, + context: Optional[List[str]] = None, + supported_language: str = "en", + supported_entity: str = "US_PRESCRIPTION_NUMBER", + name: Optional[str] = None, + score_thresholds: Optional[Dict[str, float]] = None, + ): + patterns = patterns if patterns else self.PATTERNS + context = context if context else self.CONTEXT + super().__init__( + supported_entity=supported_entity, + patterns=patterns, + context=context, + supported_language=supported_language, + name=name, + ) + self.score_thresholds = ( + score_thresholds + if score_thresholds is not None + else {supported_entity: 0.6} + ) + + +class UsReferralNumberRecognizer(PatternRecognizer): + """Recognize US healthcare referral numbers with referral context. + + CMS documents referral numbers as payer-assigned values reported in the same + CMS-1500 field as prior authorization numbers. There is no universal syntax. + The primary pattern anchors a numeric identifier on its referral label, and + a weak prefixed pattern supports structured ``REF`` or ``INF`` values. + + Reference: https://www.cms.gov/outreach-and-education/mln/wbt/mln4462429-mln-wbt-1500/1500/lesson04/18/index.html + """ + + COUNTRY_CODE = "us" + + PATTERNS = [ + Pattern( + "Referral number (labelled)", + r"(?<=\b(?:referral|infusion\s+referral)" + r"(?:\s*(?:#|no\.?|number|id)\s*:?\s*|\s*:\s*|\s+))" + r"(?:(?:REF|INF)-?)?\d{6,12}\b", + 0.35, + ), + Pattern( + "Referral number (weak prefixed)", + r"\b(?:REF|INF)-?\d{6,12}\b", + 0.1, + ), + ] + + CONTEXT = [ + "referral", + "infusion", + "specialty", + "referring", + ] + + def __init__( + self, + patterns: Optional[List[Pattern]] = None, + context: Optional[List[str]] = None, + supported_language: str = "en", + supported_entity: str = "US_REFERRAL_NUMBER", + name: Optional[str] = None, + score_thresholds: Optional[Dict[str, float]] = None, + ): + patterns = patterns if patterns else self.PATTERNS + context = context if context else self.CONTEXT + super().__init__( + supported_entity=supported_entity, + patterns=patterns, + context=context, + supported_language=supported_language, + name=name, + ) + self.score_thresholds = ( + score_thresholds + if score_thresholds is not None + else {supported_entity: 0.6} + ) + + +class UsProviderTaxIdRecognizer(PatternRecognizer): + """Recognize US provider TIN/EIN values with healthcare provider context. + + CMS uses a provider's EIN or SSN as the billing provider tax ID. This + recognizer intentionally matches only the IRS-defined EIN format and valid + two-digit EIN prefixes to avoid treating SSNs as provider organization IDs. + + CMS reference: https://www.cms.gov/outreach-and-education/mln/wbt/mln4462429-mln-wbt-1500/1500/lesson04/12/index.html + IRS prefix reference: https://www.irs.gov/businesses/small-businesses-self-employed/valid-eins + """ + + COUNTRY_CODE = "us" + + # The IRS prefix list excludes 00, 07-09, 17-19, 28-29, 49, 69-70, + # 78-79, 89, and 96-97. + VALID_EIN_PREFIX = ( + r"(?:0[1-6]|1[0-6]|2[0-7]|3[0-9]|4[0-8]|5[0-9]|6[0-8]|" + r"7[1-7]|8[0-8]|9[0-5]|9[89])" + ) + + PATTERNS = [ + Pattern( + "Provider tax ID (labelled)", + r"(?<=\b(?:(?:(?:billing|rendering|healthcare)\s+provider|" + r"provider\s+organization|provider)\s+(?:tax\s*(?:id|number|" + r"identification\s+number)|tin|ein)|billing\s+provider)" + r"(?:\s*:\s*|\s+))" + VALID_EIN_PREFIX + r"-\d{7}\b", + 0.35, + ), + Pattern( + "Provider tax ID (weak valid EIN)", + r"\b" + VALID_EIN_PREFIX + r"-\d{7}\b", + 0.1, + ), + ] + + CONTEXT = [ + "tax", + "tin", + "ein", + "billing", + ] + + def __init__( + self, + patterns: Optional[List[Pattern]] = None, + context: Optional[List[str]] = None, + supported_language: str = "en", + supported_entity: str = "US_PROVIDER_TAX_ID", + name: Optional[str] = None, + score_thresholds: Optional[Dict[str, float]] = None, + ): + patterns = patterns if patterns else self.PATTERNS + context = context if context else self.CONTEXT + super().__init__( + supported_entity=supported_entity, + patterns=patterns, + context=context, + supported_language=supported_language, + name=name, + ) + self.score_thresholds = ( + score_thresholds + if score_thresholds is not None + else {supported_entity: 0.6} + ) diff --git a/presidio-analyzer/tests/mocks/__init__.py b/presidio-analyzer/tests/mocks/__init__.py index 3c5c37a7be..8420442476 100644 --- a/presidio-analyzer/tests/mocks/__init__.py +++ b/presidio-analyzer/tests/mocks/__init__.py @@ -1,5 +1,9 @@ -from .nlp_engine_mock import NlpEngineMock from .app_tracer_mock import AppTracerMock +from .nlp_engine_mock import NlpEngineMock from .recognizer_registry_mock import RecognizerRegistryMock -__all__ = ["NlpEngineMock", "AppTracerMock", "RecognizerRegistryMock"] +__all__ = [ + "NlpEngineMock", + "AppTracerMock", + "RecognizerRegistryMock", +] diff --git a/presidio-analyzer/tests/mocks/nlp_engine_mock.py b/presidio-analyzer/tests/mocks/nlp_engine_mock.py index 5065ecb514..bf11f0e0a5 100644 --- a/presidio-analyzer/tests/mocks/nlp_engine_mock.py +++ b/presidio-analyzer/tests/mocks/nlp_engine_mock.py @@ -1,6 +1,6 @@ -from typing import Iterable, Iterator, Tuple, Dict, List +from typing import Dict, Iterable, Iterator, List, Tuple -from presidio_analyzer.nlp_engine import NlpEngine, NlpArtifacts +from presidio_analyzer.nlp_engine import NlpArtifacts, NlpEngine class NlpEngineMock(NlpEngine): diff --git a/presidio-analyzer/tests/test_context_support.py b/presidio-analyzer/tests/test_context_support.py index f3cf3b9e2a..f3f94307e4 100644 --- a/presidio-analyzer/tests/test_context_support.py +++ b/presidio-analyzer/tests/test_context_support.py @@ -166,7 +166,7 @@ def test_when_text_with_only_additional_context_lemma_based_context_enhancer_the results_with_additional_context[0].analysis_explanation.supportive_context_word == "driver" ) - assert results_with_additional_context[0].score == 0.6499999999999999 + assert results_with_additional_context[0].score == pytest.approx(0.65) def test_when_text_with_context_then_improves_score( diff --git a/presidio-analyzer/tests/test_us_health_insurance_member_id_recognizer.py b/presidio-analyzer/tests/test_us_health_insurance_member_id_recognizer.py new file mode 100644 index 0000000000..983f6d430c --- /dev/null +++ b/presidio-analyzer/tests/test_us_health_insurance_member_id_recognizer.py @@ -0,0 +1,194 @@ +import pytest +from presidio_analyzer import AnalyzerEngine, RecognizerRegistry +from presidio_analyzer.predefined_recognizers import ( + UsHealthInsuranceMemberIdRecognizer, +) + +from tests import assert_result + + +@pytest.fixture(scope="module") +def recognizer(): + """Return an instance of the US health insurance member ID recognizer.""" + return UsHealthInsuranceMemberIdRecognizer() + + +@pytest.fixture(scope="module") +def entity(): + """Return the US health insurance member ID entity name.""" + return "US_HEALTH_INSURANCE_MEMBER_ID" + + +@pytest.fixture(scope="module") +def analyze_member_id(spacy_nlp_engine): + """Return a member ID analyzer using production spaCy tokenization.""" + + def analyze(text, recognizer, entity, score_threshold=None): + registry = RecognizerRegistry() + registry.add_recognizer(recognizer) + analyzer = AnalyzerEngine(registry=registry, nlp_engine=spacy_nlp_engine) + return analyzer.analyze( + text=text, + language="en", + entities=[entity], + score_threshold=score_threshold, + ) + + return analyze + + +@pytest.mark.parametrize( + "text, expected_positions", + [ + # fmt: off + ("Member ID ABC123456789", ((10, 22),)), + ("member number ZX-987654321 appears on the card", ((14, 26),)), + ("Subscriber ID HPN12345A9 is active", ((14, 24),)), + ("Insurance ID BCBSM1234567 was verified", ((13, 25),)), + ("Insurance plan ID UHC-12345AB covers the visit", ((18, 29),)), + ("Plan member ID AET987654 for this policy", ((15, 24),)), + ("Policy ID CIGNA123456 belongs to the patient", ((10, 21),)), + ("The insurance card lists subscriber number K123456789", ((43, 53),)), + # fmt: on + ], +) +def test_when_member_id_has_context_then_detected( + text, expected_positions, recognizer, entity, analyze_member_id +): + """Test context raises plausible member IDs above the threshold.""" + results = analyze_member_id(text, recognizer, entity) + results = sorted(results, key=lambda result: result.start) + assert len(results) == len(expected_positions) + for result, (start, end) in zip(results, expected_positions): + assert result.entity_type == entity + assert result.start == start + assert result.end == end + assert result.score == pytest.approx(0.45) + + +@pytest.mark.parametrize( + "text, expected_value", + [ + ("member id abc123456", "abc123456"), + ("MeMbEr Id AbC123456", "AbC123456"), + ("Subscriber ID zx-987654321.", "zx-987654321"), + ], +) +def test_member_id_matching_is_case_insensitive_and_ignores_trailing_punctuation( + text, expected_value, recognizer, entity, analyze_member_id +): + """Test casing and punctuation do not change a plausible member ID match.""" + results = analyze_member_id(text, recognizer, entity) + start = text.index(expected_value) + assert len(results) == 1 + assert results[0].entity_type == entity + assert results[0].start == start + assert results[0].end == start + len(expected_value) + assert results[0].score == pytest.approx(0.45) + + +def test_when_text_has_multiple_member_ids_then_all_are_detected( + recognizer, entity, analyze_member_id +): + """Test every contextual member ID in one input is returned.""" + text = "Member ID ABC123456 and subscriber ID ZX-987654321." + expected_values = ["ABC123456", "ZX-987654321"] + results = sorted( + analyze_member_id(text, recognizer, entity), + key=lambda result: result.start, + ) + assert [text[result.start : result.end] for result in results] == expected_values + assert all(result.score == pytest.approx(0.45) for result in results) + + +@pytest.mark.parametrize( + "text, expected_value", + [ + ("Member ID A12345", "A12345"), + ("Member ID ABCDE-12345678901234", "ABCDE-12345678901234"), + ], +) +def test_member_id_minimum_and_maximum_lengths_are_detected( + text, expected_value, recognizer, entity, analyze_member_id +): + """Test the documented 6-to-20-character member ID boundaries.""" + results = analyze_member_id(text, recognizer, entity) + start = text.index(expected_value) + assert len(results) == 1 + assert results[0].entity_type == entity + assert results[0].start == start + assert results[0].end == start + len(expected_value) + assert results[0].score == pytest.approx(0.45) + + +@pytest.mark.parametrize( + "text", + [ + "ABC123456789", + "Please store HPN12345A9 in the table", + "Order number ABC123456789 shipped yesterday", + "Tracking number ZX-987654321 is in transit", + "Case number HPN12345A9 is pending review", + "Claim number BCBSM1234567 was denied", + "covid19", + "sha256", + "iphone15pro", + "rfc2119", + "gpt4turbo", + "ICD10CM123", + "ABC-1234567", + ], +) +def test_when_member_id_lacks_insurance_context_then_below_threshold( + text, recognizer, entity, analyze_member_id +): + """Test pattern-only and unrelated-context values are suppressed.""" + assert analyze_member_id(text, recognizer, entity) == [] + + +@pytest.mark.parametrize( + "text", + [ + "Member ID 1234567890", + "Subscriber ID A123", + "Member ID ABCDE-123456789012345", + ], +) +def test_when_member_id_pattern_is_implausible_then_not_detected( + text, recognizer, entity, analyze_member_id +): + """Test numeric-only and short values do not match the base pattern.""" + assert ( + analyze_member_id( + text, + recognizer, + entity, + score_threshold=0, + ) + == [] + ) + + +def test_explicit_request_threshold_can_return_pattern_only_member_id( + recognizer, entity, analyze_member_id +): + """Test structured callers can opt into the raw pattern match.""" + text = "ABC123456789" + results = analyze_member_id( + text, + recognizer, + entity, + score_threshold=0, + ) + assert len(results) == 1 + assert_result(results[0], entity, 0, len(text), 0.1) + + +def test_us_health_insurance_member_id_recognizer_metadata(recognizer, entity): + """Test entity metadata, context, and recognizer threshold.""" + assert recognizer.supported_entities == [entity] + assert recognizer.supported_language == "en" + assert recognizer.context == ["member", "subscriber", "insurance", "policy"] + assert recognizer.patterns[0].name == "Health insurance member ID (weak)" + assert recognizer.patterns[0].score == 0.1 + assert recognizer.score_thresholds == {entity: 0.4} diff --git a/presidio-analyzer/tests/test_us_healthcare_admin_recognizers.py b/presidio-analyzer/tests/test_us_healthcare_admin_recognizers.py new file mode 100644 index 0000000000..584a27f2e6 --- /dev/null +++ b/presidio-analyzer/tests/test_us_healthcare_admin_recognizers.py @@ -0,0 +1,643 @@ +import pytest +from presidio_analyzer import AnalyzerEngine, PatternRecognizer, RecognizerRegistry +from presidio_analyzer.predefined_recognizers import ( + UsClaimNumberRecognizer, + UsPrescriptionNumberRecognizer, + UsPriorAuthorizationNumberRecognizer, + UsProviderTaxIdRecognizer, + UsReferralNumberRecognizer, +) + +from tests import assert_result + + +@pytest.fixture(scope="module") +def analyze_with_recognizer(spacy_nlp_engine): + """Return an administrative ID analyzer using production spaCy tokenization.""" + + def analyze(text, entity, recognizer, score_threshold=None): + registry = RecognizerRegistry() + registry.add_recognizer(recognizer) + analyzer = AnalyzerEngine(registry=registry, nlp_engine=spacy_nlp_engine) + return analyzer.analyze( + text=text, + language="en", + entities=[entity], + score_threshold=score_threshold, + ) + + return analyze + + +@pytest.mark.parametrize( + "recognizer, entity, text, expected_positions", + [ + # fmt: off + ( + UsPriorAuthorizationNumberRecognizer(), + "US_PRIOR_AUTHORIZATION_NUMBER", + "Prior authorization PA-987654321 approved for treatment.", + ((20, 32),), + ), + ( + UsClaimNumberRecognizer(), + "US_CLAIM_NUMBER", + "Processed healthcare claim CLM456789123 was paid.", + ((27, 39),), + ), + ( + UsPrescriptionNumberRecognizer(), + "US_PRESCRIPTION_NUMBER", + "Prescription number RX789456123 was filled by the pharmacy.", + ((20, 31),), + ), + ( + UsReferralNumberRecognizer(), + "US_REFERRAL_NUMBER", + "Infusion referral number INF2025001234 is ready for scheduling.", + ((25, 38),), + ), + ( + UsProviderTaxIdRecognizer(), + "US_PROVIDER_TAX_ID", + "Provider Tax ID 12-3456789 belongs to the billing provider.", + ((16, 26),), + ), + # fmt: on + ], +) +def test_when_us_healthcare_admin_id_has_context_then_detected( + recognizer, entity, text, expected_positions, analyze_with_recognizer +): + """Test context enhancement raises matches above the recognizer threshold.""" + results = analyze_with_recognizer(text, entity, recognizer) + results = sorted(results, key=lambda result: result.start) + assert len(results) == len(expected_positions) + for result, (start, end) in zip(results, expected_positions): + assert_result(result, entity, start, end, 0.7) + + +@pytest.mark.parametrize( + "recognizer, entity, text, expected_value", + [ + # fmt: off + ( + UsPriorAuthorizationNumberRecognizer(), + "US_PRIOR_AUTHORIZATION_NUMBER", + "pRiOr AuThOrIzAtIoN pa-123456", + "pa-123456", + ), + ( + UsClaimNumberRecognizer(), + "US_CLAIM_NUMBER", + "cLaIm clm123456", + "clm123456", + ), + ( + UsPrescriptionNumberRecognizer(), + "US_PRESCRIPTION_NUMBER", + "pReScRiPtIoN rX123456", + "rX123456", + ), + ( + UsReferralNumberRecognizer(), + "US_REFERRAL_NUMBER", + "rEfErRaL inf123456", + "inf123456", + ), + ( + UsProviderTaxIdRecognizer(), + "US_PROVIDER_TAX_ID", + "bIlLiNg PrOvIdEr eIn: 12-3456789", + "12-3456789", + ), + # fmt: on + ], +) +def test_admin_id_matching_is_case_insensitive( + recognizer, entity, text, expected_value, analyze_with_recognizer +): + """Test mixed-case labels and prefixes are detected.""" + results = analyze_with_recognizer(text, entity, recognizer) + start = text.index(expected_value) + assert len(results) == 1 + assert_result(results[0], entity, start, start + len(expected_value), 0.7) + + +@pytest.mark.parametrize( + "recognizer, entity, text, expected_values", + [ + # fmt: off + ( + UsPriorAuthorizationNumberRecognizer(), + "US_PRIOR_AUTHORIZATION_NUMBER", + "Prior authorization PA-123456; prior authorization PA-654321.", + ["PA-123456", "PA-654321"], + ), + ( + UsClaimNumberRecognizer(), + "US_CLAIM_NUMBER", + "Claim CLM123456 and claim CLM654321.", + ["CLM123456", "CLM654321"], + ), + ( + UsPrescriptionNumberRecognizer(), + "US_PRESCRIPTION_NUMBER", + "Prescription RX123456 and prescription RX654321.", + ["RX123456", "RX654321"], + ), + ( + UsReferralNumberRecognizer(), + "US_REFERRAL_NUMBER", + "Referral REF123456 and referral INF654321.", + ["REF123456", "INF654321"], + ), + ( + UsProviderTaxIdRecognizer(), + "US_PROVIDER_TAX_ID", + "Provider EIN 12-3456789 and provider TIN 20-1234567.", + ["12-3456789", "20-1234567"], + ), + # fmt: on + ], +) +def test_when_text_has_multiple_admin_ids_then_all_are_detected( + recognizer, entity, text, expected_values, analyze_with_recognizer +): + """Test every contextual administrative ID in one input is returned.""" + results = sorted( + analyze_with_recognizer(text, entity, recognizer), + key=lambda result: result.start, + ) + assert [text[result.start : result.end] for result in results] == expected_values + + +@pytest.mark.parametrize( + "recognizer, entity, text, expected_value", + [ + # fmt: off + ( + UsPriorAuthorizationNumberRecognizer(), + "US_PRIOR_AUTHORIZATION_NUMBER", + "Prior authorization PA-123456.", + "PA-123456", + ), + ( + UsClaimNumberRecognizer(), + "US_CLAIM_NUMBER", + "Claim CLM123456,", + "CLM123456", + ), + ( + UsPrescriptionNumberRecognizer(), + "US_PRESCRIPTION_NUMBER", + "Prescription RX123456;", + "RX123456", + ), + ( + UsReferralNumberRecognizer(), + "US_REFERRAL_NUMBER", + "Referral REF123456.", + "REF123456", + ), + ( + UsProviderTaxIdRecognizer(), + "US_PROVIDER_TAX_ID", + "Provider EIN 12-3456789.", + "12-3456789", + ), + # fmt: on + ], +) +def test_admin_id_matching_ignores_trailing_punctuation( + recognizer, entity, text, expected_value, analyze_with_recognizer +): + """Test trailing sentence punctuation stays outside the result span.""" + results = analyze_with_recognizer(text, entity, recognizer) + start = text.index(expected_value) + assert len(results) == 1 + assert_result(results[0], entity, start, start + len(expected_value), 0.7) + + +@pytest.mark.parametrize( + "recognizer, entity, text, expected_value", + [ + # fmt: off + ( + UsPriorAuthorizationNumberRecognizer(), + "US_PRIOR_AUTHORIZATION_NUMBER", + "Prior authorization PA-123456", + "PA-123456", + ), + ( + UsPriorAuthorizationNumberRecognizer(), + "US_PRIOR_AUTHORIZATION_NUMBER", + "Prior authorization PA-123456789012", + "PA-123456789012", + ), + ( + UsClaimNumberRecognizer(), + "US_CLAIM_NUMBER", + "Claim CLM123456", + "CLM123456", + ), + ( + UsClaimNumberRecognizer(), + "US_CLAIM_NUMBER", + "Claim CLM123456789012345", + "CLM123456789012345", + ), + ( + UsPrescriptionNumberRecognizer(), + "US_PRESCRIPTION_NUMBER", + "Prescription RX123456", + "RX123456", + ), + ( + UsPrescriptionNumberRecognizer(), + "US_PRESCRIPTION_NUMBER", + "Prescription RX123456789012", + "RX123456789012", + ), + ( + UsReferralNumberRecognizer(), + "US_REFERRAL_NUMBER", + "Referral REF123456", + "REF123456", + ), + ( + UsReferralNumberRecognizer(), + "US_REFERRAL_NUMBER", + "Referral INF123456789012", + "INF123456789012", + ), + # fmt: on + ], +) +def test_admin_id_minimum_and_maximum_lengths_are_detected( + recognizer, entity, text, expected_value, analyze_with_recognizer +): + """Test each variable-length administrative ID at its exact boundaries.""" + results = analyze_with_recognizer(text, entity, recognizer) + start = text.index(expected_value) + assert len(results) == 1 + assert_result(results[0], entity, start, start + len(expected_value), 0.7) + + +@pytest.mark.parametrize( + "recognizer, entity, text", + [ + # fmt: off + ( + UsPriorAuthorizationNumberRecognizer(), + "US_PRIOR_AUTHORIZATION_NUMBER", + "PA-12345", + ), + ( + UsPriorAuthorizationNumberRecognizer(), + "US_PRIOR_AUTHORIZATION_NUMBER", + "PA-1234567890123", + ), + (UsClaimNumberRecognizer(), "US_CLAIM_NUMBER", "CLM12345"), + ( + UsClaimNumberRecognizer(), + "US_CLAIM_NUMBER", + "CLM1234567890123456", + ), + (UsPrescriptionNumberRecognizer(), "US_PRESCRIPTION_NUMBER", "RX12345"), + ( + UsPrescriptionNumberRecognizer(), + "US_PRESCRIPTION_NUMBER", + "RX1234567890123", + ), + (UsReferralNumberRecognizer(), "US_REFERRAL_NUMBER", "REF12345"), + ( + UsReferralNumberRecognizer(), + "US_REFERRAL_NUMBER", + "INF1234567890123", + ), + (UsProviderTaxIdRecognizer(), "US_PROVIDER_TAX_ID", "12-123456"), + (UsProviderTaxIdRecognizer(), "US_PROVIDER_TAX_ID", "12-12345678"), + # fmt: on + ], +) +def test_too_short_and_too_long_admin_ids_do_not_match( + recognizer, entity, text, analyze_with_recognizer +): + """Test values one digit outside each supported length do not match.""" + assert analyze_with_recognizer(text, entity, recognizer, score_threshold=0) == [] + + +@pytest.mark.parametrize( + "recognizer, entity, text, expected_value, expected_score", + [ + # fmt: off + ( + UsPriorAuthorizationNumberRecognizer(), + "US_PRIOR_AUTHORIZATION_NUMBER", + "Prior authorization number: 987654321 approved.", + "987654321", + 0.7, + ), + ( + UsClaimNumberRecognizer(), + "US_CLAIM_NUMBER", + "Claim number: 1234567890123 was paid.", + "1234567890123", + 0.7, + ), + ( + UsClaimNumberRecognizer(), + "US_CLAIM_NUMBER", + "Claim ID 123456789012345 was paid.", + "123456789012345", + 0.7, + ), + ( + UsPrescriptionNumberRecognizer(), + "US_PRESCRIPTION_NUMBER", + "Rx #1234567", + "1234567", + 0.6, + ), + ( + UsPrescriptionNumberRecognizer(), + "US_PRESCRIPTION_NUMBER", + "Prescription number: 7654321", + "7654321", + 0.7, + ), + ( + UsPrescriptionNumberRecognizer(), + "US_PRESCRIPTION_NUMBER", + "prescription 4455667", + "4455667", + 0.7, + ), + ( + UsReferralNumberRecognizer(), + "US_REFERRAL_NUMBER", + "Infusion referral number: 2025001234", + "2025001234", + 0.7, + ), + # fmt: on + ], +) +def test_when_admin_id_follows_label_then_identifier_only_is_detected( + recognizer, + entity, + text, + expected_value, + expected_score, + analyze_with_recognizer, +): + """Test labels enable bare numeric IDs without entering the result span.""" + results = analyze_with_recognizer(text, entity, recognizer) + start = text.index(expected_value) + assert len(results) == 1 + assert_result( + results[0], entity, start, start + len(expected_value), expected_score + ) + + +@pytest.mark.parametrize( + "text, expected_value", + [ + ("Billing provider EIN: 12-3456789", "12-3456789"), + ("Rendering provider TIN 20-1234567", "20-1234567"), + ("Healthcare provider tax number: 67-1234567", "67-1234567"), + ("Billing provider: 99-1234567", "99-1234567"), + ], +) +def test_when_provider_ein_has_provider_tax_label_then_detected( + text, expected_value, analyze_with_recognizer +): + """Test valid EINs immediately following provider tax labels are detected.""" + results = analyze_with_recognizer( + text, + "US_PROVIDER_TAX_ID", + UsProviderTaxIdRecognizer(), + ) + start = text.index(expected_value) + assert len(results) == 1 + assert_result( + results[0], + "US_PROVIDER_TAX_ID", + start, + start + len(expected_value), + 0.7, + ) + + +def test_when_number_has_different_workflow_label_then_prescription_not_detected( + analyze_with_recognizer, +): + """Test a claim label does not support a prescription number match.""" + recognizer = UsPrescriptionNumberRecognizer() + assert ( + analyze_with_recognizer( + "The claim 1234567 was paid", + "US_PRESCRIPTION_NUMBER", + recognizer, + ) + == [] + ) + + +@pytest.mark.parametrize( + "recognizer, entity, text", + [ + # fmt: off + ( + UsPriorAuthorizationNumberRecognizer(), + "US_PRIOR_AUTHORIZATION_NUMBER", + "PA-987654321", + ), + (UsClaimNumberRecognizer(), "US_CLAIM_NUMBER", "CLM456789123"), + (UsPrescriptionNumberRecognizer(), "US_PRESCRIPTION_NUMBER", "RX789456123"), + (UsReferralNumberRecognizer(), "US_REFERRAL_NUMBER", "INF2025001234"), + (UsProviderTaxIdRecognizer(), "US_PROVIDER_TAX_ID", "12-3456789"), + # fmt: on + ], +) +def test_when_us_healthcare_admin_id_lacks_context_then_below_threshold( + recognizer, entity, text, analyze_with_recognizer +): + """Test normal analyzer calls suppress pattern-only matches.""" + assert analyze_with_recognizer(text, entity, recognizer) == [] + + +@pytest.mark.parametrize( + "recognizer, entity, text", + [ + # fmt: off + ( + UsPriorAuthorizationNumberRecognizer(), + "US_PRIOR_AUTHORIZATION_NUMBER", + "Order number PA-987654321 is ready.", + ), + ( + UsClaimNumberRecognizer(), + "US_CLAIM_NUMBER", + "Tracking number CLM456789123 is active.", + ), + ( + UsPrescriptionNumberRecognizer(), + "US_PRESCRIPTION_NUMBER", + "Case number RX789456123 is pending.", + ), + ( + UsReferralNumberRecognizer(), + "US_REFERRAL_NUMBER", + "Claim number INF2025001234 was denied.", + ), + ( + UsProviderTaxIdRecognizer(), + "US_PROVIDER_TAX_ID", + "Invoice number 12-3456789 was posted.", + ), + # fmt: on + ], +) +def test_when_us_healthcare_admin_id_has_unrelated_context_then_not_detected( + recognizer, entity, text, analyze_with_recognizer +): + """Test similar-looking workflow IDs stay below the threshold.""" + assert analyze_with_recognizer(text, entity, recognizer) == [] + + +@pytest.mark.parametrize( + "text", + [ + "Provider phone extension 12-3456789", + "provider 00-0000000 listed", + "Employee tax ID 12-3456789", + ], +) +def test_when_ein_lacks_provider_tax_label_then_not_detected( + text, analyze_with_recognizer +): + """Test generic provider or tax wording cannot promote an EIN-shaped value.""" + assert ( + analyze_with_recognizer( + text, + "US_PROVIDER_TAX_ID", + UsProviderTaxIdRecognizer(), + ) + == [] + ) + + +@pytest.mark.parametrize( + "invalid_prefix", + [ + "00", + "07", + "08", + "09", + "17", + "18", + "19", + "28", + "29", + "49", + "69", + "70", + "78", + "79", + "89", + "96", + "97", + ], +) +def test_when_provider_ein_prefix_is_not_irs_valid_then_not_detected( + invalid_prefix, analyze_with_recognizer +): + """Test values outside the IRS-assigned EIN prefix set do not match.""" + text = f"Provider Tax ID {invalid_prefix}-1234567" + assert ( + analyze_with_recognizer( + text, + "US_PROVIDER_TAX_ID", + UsProviderTaxIdRecognizer(), + score_threshold=0, + ) + == [] + ) + + +@pytest.mark.parametrize( + "recognizer, entity, text, expected_score", + [ + # fmt: off + ( + UsPriorAuthorizationNumberRecognizer(), + "US_PRIOR_AUTHORIZATION_NUMBER", + "PA-987654321", + 0.1, + ), + (UsClaimNumberRecognizer(), "US_CLAIM_NUMBER", "CLM456789123", 0.1), + ( + UsPrescriptionNumberRecognizer(), + "US_PRESCRIPTION_NUMBER", + "RX789456123", + 0.1, + ), + (UsReferralNumberRecognizer(), "US_REFERRAL_NUMBER", "INF2025001234", 0.1), + (UsProviderTaxIdRecognizer(), "US_PROVIDER_TAX_ID", "12-3456789", 0.1), + # fmt: on + ], +) +def test_explicit_request_threshold_can_return_pattern_only_matches( + recognizer, entity, text, expected_score, analyze_with_recognizer +): + """Test callers can opt into raw pattern matches for structured analysis.""" + results = analyze_with_recognizer(text, entity, recognizer, score_threshold=0) + assert len(results) == 1 + assert_result(results[0], entity, 0, len(text), expected_score) + + +@pytest.mark.parametrize( + "recognizer, entity, expected_context", + [ + ( + UsPriorAuthorizationNumberRecognizer(), + "US_PRIOR_AUTHORIZATION_NUMBER", + ["authorization", "auth", "preauthorization", "approval"], + ), + ( + UsClaimNumberRecognizer(), + "US_CLAIM_NUMBER", + ["claim", "billing"], + ), + ( + UsPrescriptionNumberRecognizer(), + "US_PRESCRIPTION_NUMBER", + ["prescription", "pharmacy", "medication"], + ), + ( + UsReferralNumberRecognizer(), + "US_REFERRAL_NUMBER", + ["referral", "infusion", "specialty", "referring"], + ), + ( + UsProviderTaxIdRecognizer(), + "US_PROVIDER_TAX_ID", + ["tax", "tin", "ein", "billing"], + ), + ], +) +def test_us_healthcare_admin_recognizer_metadata(recognizer, entity, expected_context): + """Test entity metadata, context, and recognizer threshold.""" + custom_thresholds = {entity: 0.8} + customized_recognizer = type(recognizer)(score_thresholds=custom_thresholds) + + assert isinstance(recognizer, PatternRecognizer) + assert PatternRecognizer in type(recognizer).__bases__ + assert recognizer.COUNTRY_CODE == "us" + assert recognizer.supported_entities == [entity] + assert recognizer.supported_language == "en" + assert recognizer.context == expected_context + assert recognizer.score_thresholds == {entity: 0.6} + assert customized_recognizer.score_thresholds == custom_thresholds