2025-08-05 20:46:19 +00:00
|
|
|
# Delimiter Patterns
|
|
|
|
|
PIPE_PATTERN = r"\|([^|]+)\|"
|
|
|
|
|
BACKTICK_PATTERN = r"`([^`]+)`"
|
|
|
|
|
TRIPLE_BACKTICK_PATTERN = r"```([^`]+)```"
|
|
|
|
|
|
|
|
|
|
# TIN and NPI Regex Patterns
|
2025-10-08 15:30:19 +00:00
|
|
|
# TIN_PATTERN = (
|
|
|
|
|
# r"\b\d{9}\b|\b\d{2}[-.]\d{7}\b|\b\d{3}[-.]\d{2}[-.]\d{4}\b|\b\d{2}-\d{4}-\d{3}\b"
|
|
|
|
|
# )
|
|
|
|
|
# NPI_PATTERN = r"\b\d{10}\b"
|
|
|
|
|
TIN_PATTERN = r"\b(?:\d[\s\-.]*){9}\b"
|
2025-12-10 23:26:24 +00:00
|
|
|
FILE_NAME_TIN_PATTERN = r"(?<!\d)(?:\d{2}[-.]\d{7}|\d{9})(?!\d)"
|
2025-10-08 15:30:19 +00:00
|
|
|
NPI_PATTERN = r"\b(?:\d\s*){10}\b"
|
|
|
|
|
|
2025-09-05 15:15:39 +00:00
|
|
|
|
|
|
|
|
# OCR-Correctable Patterns
|
|
|
|
|
# Handles common OCR misinterpretations where letters are mistaken for digits.
|
|
|
|
|
# Common OCR substitutions: O/o→0, l/I→1, S/s→5, G/g→6, B/b→8
|
|
|
|
|
|
|
|
|
|
# OCR-Correctable TIN Pattern
|
|
|
|
|
# Strategy: Match 9-character sequences OR formatted sequences (with hyphens/dots)
|
|
|
|
|
# containing digits and OCR-prone letters, then apply corrections post-match.
|
|
|
|
|
#
|
|
|
|
|
# Examples that WILL be captured and corrected:
|
|
|
|
|
# - "12345678O" → "123456780" (unformatted, O at end)
|
|
|
|
|
# - "O23456789" → "023456789" (unformatted, O at start)
|
|
|
|
|
# - "1234O6789" → "123406789" (unformatted, O in middle)
|
|
|
|
|
# - "12-345678O" → "12-3456780" (formatted XX-XXXXXXX, O at end)
|
|
|
|
|
# - "O2-3456789" → "02-3456789" (formatted XX-XXXXXXX, O at start)
|
|
|
|
|
# - "12O-456789" → "120-456789" (formatted XX-XXXXXXX, O in first segment)
|
|
|
|
|
# - "123-4S-6789" → "123-45-6789" (formatted XXX-XX-XXXX, S in middle segment)
|
|
|
|
|
#
|
|
|
|
|
# Examples that will NOT match:
|
|
|
|
|
# - "12345678" (only 8 characters)
|
|
|
|
|
# - "1234567890" (10 characters)
|
|
|
|
|
# - "123ABC789" (non-OCR letters like A, C)
|
|
|
|
|
#
|
|
|
|
|
# Note: This pattern is intentionally permissive. Validation happens post-match
|
|
|
|
|
# by applying OCR_SUBSTITUTIONS and checking that result has exactly 9 digits.
|
|
|
|
|
TIN_OCR_PATTERN = r"\b[O0lI1S5G6B8gbo\d]{9}\b|\b[O0lI1S5G6B8gbo\d]{2}[-.][O0lI1S5G6B8gbo\d]{7}\b|\b[O0lI1S5G6B8gbo\d]{3}[-.][O0lI1S5G6B8gbo\d]{2}[-.][O0lI1S5G6B8gbo\d]{4}\b"
|
|
|
|
|
|
|
|
|
|
# OCR-Correctable NPI Pattern
|
|
|
|
|
# Strategy: Match 10-character sequences containing digits and OCR-prone letters,
|
|
|
|
|
# then apply corrections post-match to validate it becomes a proper 10-digit NPI.
|
|
|
|
|
#
|
|
|
|
|
# Examples that WILL be captured and corrected:
|
|
|
|
|
# - "123456789O" → "1234567890" (O at end)
|
|
|
|
|
# - "l234567890" → "1234567890" (l at start)
|
|
|
|
|
# - "12345O7890" → "1234507890" (O in middle)
|
|
|
|
|
# - "123456789l" → "1234567891" (l at end)
|
|
|
|
|
#
|
|
|
|
|
# Examples that will NOT match:
|
|
|
|
|
# - "123456789" (only 9 characters)
|
|
|
|
|
# - "12345678901" (11 characters)
|
|
|
|
|
# - "123ABC7890" (non-OCR letters like A, C)
|
|
|
|
|
#
|
|
|
|
|
# Note: NPIs are always unformatted 10-digit numbers, so no hyphen/dot variations.
|
|
|
|
|
# Validation happens post-match by applying OCR_SUBSTITUTIONS and checking
|
|
|
|
|
# that result is exactly 10 digits.
|
|
|
|
|
NPI_OCR_PATTERN = r"\b[O0lI1S5G6B8gbo\d]{10}\b"
|
|
|
|
|
|
|
|
|
|
# OCR Substitution Mapping
|
|
|
|
|
OCR_SUBSTITUTIONS = {
|
|
|
|
|
"O": "0", # Letter O to zero
|
|
|
|
|
"o": "0",
|
|
|
|
|
"l": "1", # Letter l/I to one
|
|
|
|
|
"I": "1",
|
|
|
|
|
"S": "5", # Letter S to five
|
|
|
|
|
"s": "5",
|
|
|
|
|
"G": "6", # Letter G to six
|
|
|
|
|
"g": "6",
|
|
|
|
|
"B": "8", # Letter B to eight
|
|
|
|
|
"b": "8",
|
|
|
|
|
}
|