From c0fa621aed2b390b7769e2543dc0ff10cb5ad43d Mon Sep 17 00:00:00 2001 From: Alex Galarce Date: Fri, 5 Sep 2025 21:21:54 +0000 Subject: [PATCH] Merged in bugfix/multi-tin-npi-fix (pull request #699) Bugfix/multi tin npi fix * Add debug logging for OCR candidates and provider info extraction * Update TIN and NPI extraction instructions to handle multiple values in a single record * Add validation layer to cross-check TINs and NPIs with regex findings in provider info extraction * Enhance exact match extraction by normalizing identifiers and validating lengths for TIN and NPI Approved-by: Katon Minhas --- .../src/investment/tin_npi_funcs.py | 76 ++++++++++++++++++- .../src/prompts/prompt_templates.py | 12 +-- 2 files changed, 78 insertions(+), 10 deletions(-) diff --git a/fieldExtraction/src/investment/tin_npi_funcs.py b/fieldExtraction/src/investment/tin_npi_funcs.py index 03c1c68..689837e 100644 --- a/fieldExtraction/src/investment/tin_npi_funcs.py +++ b/fieldExtraction/src/investment/tin_npi_funcs.py @@ -46,15 +46,25 @@ def get_all_matches_with_ocr( results = [] seen_values = set() - # Get exact matches + # Get exact matches and normalize them exact_matches = get_all_matches(text, exact_pattern) for match in exact_matches: - results.append((match, "EXACT")) - seen_values.add(match) + # Normalize: remove hyphens/dots, keep only digits + normalized = "".join(c for c in match if c.isdigit()) + # Validate length based on identifier type + if identifier_type == "TIN": + expected_length = 9 + elif identifier_type == "NPI": + expected_length = 10 + + if len(normalized) == expected_length and normalized not in seen_values: + results.append((normalized, "EXACT")) + seen_values.add(normalized) # Get OCR-correctable candidates (even if we found exact matches) ocr_candidates = get_all_matches(text, ocr_pattern) for candidate in ocr_candidates: + logging.debug(f"OCR candidate for {identifier_type}: {candidate}") corrected, is_valid = attempt_ocr_correction(candidate, identifier_type) if is_valid and corrected not in seen_values: results.append((corrected, "OCR_CORRECTED")) @@ -478,9 +488,14 @@ def get_provider_info(text_dict: dict, page_num: str, filename: str) -> list: prompt, "legacy_sonnet", filename, max_tokens=8192 ) # Sometimes rosters can have 50+ provider + logging.debug( + f"Raw LLM response for provider info on page {page_num}: {claude_answer_raw}" + ) + # Defensive programming for any json parsing errors try: providers = string_utils.universal_json_load(claude_answer_raw) + logging.debug(f"Parsed provider info on page {page_num}: {providers}") # Ensure we have a list of providers (not a dict or other type) if isinstance(providers, dict): # Single provider returned as dict - wrap in list @@ -499,6 +514,56 @@ def get_provider_info(text_dict: dict, page_num: str, filename: str) -> list: } ] + # VALIDATION LAYER - Cross-check with regex findings + page_text = text_dict[page_num] + + # Get all TINs and NPIs found via regex on this page + regex_tins = [ + tin + for tin, _ in get_all_matches_with_ocr( + page_text, + regex_patterns.TIN_PATTERN, + regex_patterns.TIN_OCR_PATTERN, + identifier_type="TIN", + ) + ] + regex_npis = [ + npi + for npi, _ in get_all_matches_with_ocr( + page_text, + regex_patterns.NPI_PATTERN, + regex_patterns.NPI_OCR_PATTERN, + identifier_type="NPI", + ) + ] + + # Collect all TINs/NPIs that LLM found + llm_found_tins = [] + llm_found_npis = [] + for provider in providers: + if provider.get("TIN") and provider["TIN"] != "UNKNOWN": + # Normalize for comparison (remove hyphens/dots) + raw_tins = [tin.strip() for tin in provider["TIN"].split("|")] + normalized_tins = [ + tin.replace("-", "").replace(".", "") for tin in raw_tins + ] + llm_found_tins.extend(normalized_tins) + if provider.get("NPI") and provider["NPI"] != "UNKNOWN": + raw_npis = [npi.strip() for npi in provider["NPI"].split("|")] + normalized_npis = [ + npi.replace("-", "").replace(".", "") for npi in raw_npis + ] + llm_found_npis.extend(normalized_npis) + + # Note: regex_tins and regex_npis are already normalized in get_all_matches_with_ocr + missed_tins = [tin for tin in regex_tins if tin not in llm_found_tins] + missed_npis = [npi for npi in regex_npis if npi not in llm_found_npis] + if missed_tins or missed_npis: + # WARN ONLY if there are missed identifiers + logging.warning( + f"LLM missed identifiers on page {page_num} of {filename}. Missed TINs: {missed_tins}, Missed NPIs: {missed_npis}" + ) + # Add IS_GROUP flag based on ON_SIGNATURE_PAGE and page_num for provider in providers: if float(page_num) <= 2.0 or provider.get("ON_SIGNATURE_PAGE", "N") == "Y": @@ -553,7 +618,7 @@ def run_provider_info_fields( npi_matches = get_all_matches_with_ocr( contract_text, regex_patterns.NPI_PATTERN, - regex_patterns.NPI_OCR_PATTERN, # no OCR correction for NPI right now + regex_patterns.NPI_OCR_PATTERN, identifier_type="NPI", ) @@ -561,6 +626,9 @@ def run_provider_info_fields( all_tins = [tin for tin, quality in tin_matches] all_npis = [npi for npi, quality in npi_matches] + logging.debug(f"Extracted TINs: {all_tins}") + logging.debug(f"Extracted NPIs: {all_npis}") + # Log OCR corrections for monitoring exact_tins = [tin for tin, quality in tin_matches if quality == "EXACT"] ocr_corrected_tins = [ diff --git a/fieldExtraction/src/prompts/prompt_templates.py b/fieldExtraction/src/prompts/prompt_templates.py index bb65407..7ed52cd 100644 --- a/fieldExtraction/src/prompts/prompt_templates.py +++ b/fieldExtraction/src/prompts/prompt_templates.py @@ -2,7 +2,6 @@ import json from functools import cache import pandas as pd - import src.utils.llm_utils as llm_utils from constants.delimiters import Delimiter from src.utils.string_utils import extract_text_from_delimiters @@ -840,6 +839,12 @@ IMPORTANT DISTINCTION: For EACH provider entity found (group or individual practitioner), extract: {questions} +CRITICAL: If a provider has multiple TINs or NPIs, include ALL of them separated by pipes (|) in the same provider record. Do NOT create separate records for the same provider entity. + +Examples: +- If you see "NPI: 1083087266 and 1396117412", return "NPI": "1083087266|1396117412" +- If you see "Tax ID: 12-3456789, also 98-7654321", return "TIN": "123456789|987654321" + Briefly explain your reasoning. After your explanation, include the heading "FINAL PROVIDER INFO:" followed by a properly formatted JSON array with your final output. [EXTRACTION AND FORMATTING INSTRUCTIONS] @@ -1206,8 +1211,3 @@ Here is the term to analyze: Briefly explain your answer, then put your final answer in JSON dictionary format. """ - - - - -