Files
doczyai-pipelines/fieldExtraction/src/utils/qa_qc_utils.py
T

1416 lines
46 KiB
Python
Raw Normal View History

import os
import re
import warnings
from datetime import datetime
import pandas as pd
import src.utils.io_utils as io_utils
import src.utils.string_utils as string_utils
from openpyxl import Workbook
from openpyxl.utils.dataframe import dataframe_to_rows
from src import config
from src.investment import preprocessing_funcs
# Suppress pandas regex UserWarnings about match groups
warnings.filterwarnings('ignore', message='This pattern is interpreted as a regular expression', category=UserWarning)
def detect_date_anomalies(series, field_name, is_merged=False):
total_count = len(series)
unique_values = series.fillna("").nunique()
date_fields = [
"CONTRACT_BASE_EFFECTIVE_DT",
"CONTRACT_EFFECTIVE_DT",
"CONTRACT_EFFECTIVE_DT_SIGNATORY",
"CONTRACT_TERMINATION_DT",
"CONTRACT_SIGNATORY_DT",
]
date_pattern = r"(\d{1,2}[/-]\d{1,2}[/-]\d{2,4}|\d{2,4}[/-]\d{1,2}[/-]\d{1,2})"
date_like_count = 0
if pd.api.types.is_string_dtype(series) or pd.api.types.is_object_dtype(series):
series = series.astype(str)
if field_name in date_fields:
date_like_count = series.str.count(date_pattern).sum()
else:
date_like_count = series.str.count(r"\d{1,2}/\d{1,2}/\d{2,4}").sum()
data_type = "float64" if pd.api.types.is_float_dtype(series) else "string"
dataset = "Merged DataFrame" if is_merged else "Original DataFrame"
result = {
f"Total entries ({dataset})": total_count,
f"Unique values ({dataset})": unique_values,
f"Date-like entries ({dataset})": date_like_count,
}
if is_merged:
result[f"Data type ({dataset})"] = data_type
return result
def tin_check(tin: str) -> tuple[str, bool]:
pattern = r"^\b\d{2}-\d{7}\b$"
if not isinstance(tin, str):
try:
tin = str(tin)
except Exception:
check = False
return tin, check
match = re.match(pattern, tin)
check = bool(match)
tin = match.group(0) if match else tin
return tin, check
def npi_check(npi: str) -> tuple[str, bool]:
pattern = r"^[1-2]\d{9}$"
if not isinstance(npi, str):
try:
npi = str(npi)
except Exception:
check = False
return npi, check
match = re.match(pattern, npi)
check = bool(match)
npi = match.group(0) if match else npi
return npi, check
def date_format_check(date: str) -> tuple:
format = "%m/%d/%Y"
if not isinstance(date, str):
try:
date = str(date)
except TypeError:
check = False
date = date
try:
dt = datetime.strptime(date, format)
date = dt.strftime("%-m/%-d/%Y")
check = True
except ValueError:
date = date
check = False
return date, check
def yn_check(yn: str) -> tuple[str, bool]:
if not isinstance(yn, str):
try:
yn = str(yn)
except TypeError:
check = False
return yn, check
return yn, (yn == "Y" or yn == "N")
def pages_pagenum_check(pages: int, pagenum: int) -> bool:
if not isinstance(pages, int):
try:
pages = int(pages)
except (TypeError, ValueError) as e:
check = False
return check
if not isinstance(pagenum, int):
try:
pagenum = int(pagenum)
except (TypeError, ValueError) as e:
check = False
return check
check = pages >= pagenum
return check
def blanks_check(df: pd.DataFrame, threshold=config.BLANKS_THRESHOLD) -> pd.DataFrame:
highly_blank_cols = []
for col in df.columns:
if (
df[col].apply(lambda x: string_utils.is_empty(x)).sum() / len(df[col])
> threshold
):
highly_blank_cols.append(col)
return "\n".join(highly_blank_cols)
def perform_qc_qa_tests_both(ac_df, b_df, merged_df, input_dict, processed_files):
file_name_column = config.FILE_NAME_COLUMN
total_input_files = len(input_dict) if input_dict else 0
processed_ac_files = len(
[f for f in processed_files if f in ac_df[file_name_column].values]
)
processed_b_files = len(
[f for f in processed_files if f in b_df[file_name_column].values]
)
failed_ac_files = (
total_input_files - processed_ac_files if total_input_files > 0 else 0
)
failed_b_files = (
total_input_files - processed_b_files if total_input_files > 0 else 0
)
yn_cols_b = [col for col in b_df.columns if "(Y/N)" in col]
yn_cols_ac = [col for col in ac_df.columns if "(Y/N)" in col]
general_stats = {
"Total Number of Contracts processed": total_input_files,
"# of contracts successfully processed for AC fields": processed_ac_files,
"# of contracts successfully processed for B fields": processed_b_files,
"# of contracts failed for AC fields": failed_ac_files,
"# of contracts failed for B fields": failed_b_files,
"# of rows for AC run": len(ac_df),
"# of rows for B run": len(b_df),
"Total rows in merged DataFrame": len(merged_df),
"Unique Filenames in AC DataFrame": ac_df[file_name_column].nunique(),
"Unique Filenames in B DataFrame": b_df[file_name_column].nunique(),
"Unique Filenames in merged DataFrame": merged_df[file_name_column].nunique(),
"Contracts with all AC fields missing": merged_df[
[col for col in ac_df.columns if col != file_name_column]
]
.isnull()
.all(axis=1)
.sum(),
"Contracts with all B fields missing": merged_df[
[col for col in b_df.columns if col != file_name_column]
]
.isnull()
.all(axis=1)
.sum(),
"# of duplicate rows for AC run": ac_df.duplicated().sum(),
"# of duplicate rows for B run": b_df.duplicated().sum(),
"Number of duplicate rows in merged DataFrame": merged_df.duplicated().sum(),
"# of fields for AC run": len(ac_df.columns),
"# of fields for B run": len(b_df.columns),
"Highly Blank Fields in merged DataFrame": blanks_check(merged_df),
"# of incorrectly formatted Provider Group TINs": ac_df["IRS #"]
.apply(lambda x: not tin_check(x)[1] if not string_utils.is_empty(x) else False)
.sum(),
"# of incorrectly formatted Provider Group NPIs": ac_df["NPI (10-digits)"]
.apply(lambda x: not npi_check(x)[1] if not string_utils.is_empty(x) else False)
.sum(),
"# of incorrectly formatted Provider Group Signatory TINs": ac_df[
"PROV_GROUP_TIN_SIGNATORY"
]
.apply(lambda x: not tin_check(x)[1] if not string_utils.is_empty(x) else False)
.sum(),
"# of incorrectly formatted Provider TINs (other)": ac_df["PROV_TIN_OTHER"]
.apply(lambda x: not tin_check(x)[1] if not string_utils.is_empty(x) else False)
.sum(),
"# of incorrectly formatted Provider NPIs (other)": ac_df["PROV_NPI_OTHER"]
.apply(lambda x: not npi_check(x)[1] if not string_utils.is_empty(x) else False)
.sum(),
"# of rows with Pages greater than Page Num": merged_df.apply(
lambda row: (
pages_pagenum_check(row["Pages"], row["Page_Num"])
if (
not string_utils.is_empty(row["Pages"])
and string_utils.is_empty(row["Page_Num"])
)
else False
),
axis=1,
).sum(),
"# of Payer Names found in Notice to Provider Name": ac_df.apply(
lambda row: (
row["PAYER NAME"] in row["Notice to Provider Name"]
if not string_utils.is_empty(row["Notice to Provider Name"])
and not string_utils.is_empty(row["PAYER NAME"])
else False
),
axis=1,
).sum(),
"# of invalid Y/N values for B run": b_df[yn_cols_b]
.apply(lambda x: yn_check(x)[1], axis=1)
.sum()
.sum(),
"# of invalid Y/N values for AC run": ac_df[yn_cols_ac]
.apply(lambda x: yn_check(x)[1], axis=1)
.sum()
.sum(),
"# of invalid Termination Dates": ac_df["Termination Date"]
.apply(
lambda x: not date_format_check(x) if not string_utils.is_empty(x) else x
)
.sum(),
"# of invalid Conract Effective Dates": ac_df["Contract Effective Date"]
.apply(
lambda x: not date_format_check(x) if not string_utils.is_empty(x) else x
)
.sum(),
}
ac_stats = {}
for field in [col for col in ac_df.columns if col != file_name_column]:
missing_contracts = merged_df[merged_df[field].isnull()][
file_name_column
].nunique()
ac_stats[field] = {
"Contracts missing this field (Merged DataFrame)": missing_contracts,
**detect_date_anomalies(ac_df[field], field, is_merged=False),
**detect_date_anomalies(merged_df[field], field, is_merged=True),
}
b_stats = {}
for field in [col for col in b_df.columns if col != file_name_column]:
missing_contracts = (
merged_df.groupby(file_name_column)[field]
.apply(lambda x: x.isnull().all())
.sum()
)
contracts_with_missing = (
merged_df.groupby(file_name_column)[field]
.apply(lambda x: x.isnull().any())
.sum()
)
total_missing = merged_df[field].isnull().sum()
b_stats[field] = {
"Contracts missing all entries (Merged DataFrame)": missing_contracts,
"Contracts with at least one missing entry (Merged DataFrame)": contracts_with_missing,
"Total missing values (Merged DataFrame)": total_missing,
**detect_date_anomalies(b_df[field], field, is_merged=False),
**detect_date_anomalies(merged_df[field], field, is_merged=True),
}
return general_stats, ac_stats, b_stats
def perform_qc_qa_tests_ac(ac_df, input_dict, processed_files):
file_name_column = config.FILE_NAME_COLUMN
total_input_files = len(input_dict) if input_dict else 0
processed_ac_files = len(
[f for f in processed_files if f in ac_df[file_name_column].values]
)
failed_ac_files = (
total_input_files - processed_ac_files if total_input_files > 0 else 0
)
yn_cols_ac = [col for col in ac_df.columns if "(Y/N)" in col]
general_stats = {
"Total Number of Contracts processed": total_input_files,
"# of contracts successfully processed for AC fields": processed_ac_files,
"# of contracts failed for AC fields": failed_ac_files,
"# of rows for AC run": len(ac_df),
"Unique Filenames in AC DataFrame": ac_df[file_name_column].nunique(),
"Contracts with all AC fields missing": ac_df[
[col for col in ac_df.columns if col != file_name_column]
]
.isnull()
.all(axis=1)
.sum(),
"# of duplicate rows for AC run": ac_df.duplicated().sum(),
"# of fields for AC run": len(ac_df.columns),
"Highly Blank Fields in AC DataFrame": blanks_check(ac_df),
"# of incorrectly formatted Provider Group TINs": ac_df["IRS #"]
.apply(lambda x: not tin_check(x)[1] if not string_utils.is_empty(x) else False)
.sum(),
"# of incorrectly formatted Provider Group NPIs": ac_df["NPI (10-digits)"]
.apply(lambda x: not npi_check(x)[1] if not string_utils.is_empty(x) else False)
.sum(),
"# of incorrectly formatted Provider Group Signatory TINs": ac_df[
"PROV_GROUP_TIN_SIGNATORY"
]
.apply(lambda x: not tin_check(x)[1] if not string_utils.is_empty(x) else False)
.sum(),
"# of incorrectly formatted Provider TINs (other)": ac_df["PROV_TIN_OTHER"]
.apply(lambda x: not tin_check(x)[1] if not string_utils.is_empty(x) else False)
.sum(),
"# of incorrectly formatted Provider NPIs (other)": ac_df["PROV_NPI_OTHER"]
.apply(lambda x: not npi_check(x)[1] if not string_utils.is_empty(x) else False)
.sum(),
"# of Payer Names found in Notice to Provider Name": ac_df.apply(
lambda row: (
row["PAYER NAME"] in row["Notice to Provider Name"]
if not string_utils.is_empty(row["Notice to Provider Name"])
else False
),
axis=1,
).sum(),
"# of invalid Y/N values for AC run": ac_df[yn_cols_ac]
.apply(lambda x: yn_check(x)[1], axis=1)
.sum()
.sum(),
"# of invalid Termination Dates": ac_df["Termination Date"]
.apply(
lambda x: not date_format_check(x) if not string_utils.is_empty(x) else x
)
.sum(),
"# of invalid Conract Effective Dates": ac_df["Contract Effective Date"]
.apply(
lambda x: not date_format_check(x) if not string_utils.is_empty(x) else x
)
.sum(),
}
ac_stats = {}
for field in [col for col in ac_df.columns if col != file_name_column]:
missing_contracts = ac_df[ac_df[field].isnull()][file_name_column].nunique()
ac_stats[field] = {
"Contracts missing this field (Merged DataFrame)": missing_contracts,
**detect_date_anomalies(ac_df[field], field, is_merged=False),
}
return general_stats, ac_stats
def perform_qc_qa_tests_b(b_df, input_dict, processed_files):
file_name_column = config.FILE_NAME_COLUMN
total_input_files = len(input_dict) if input_dict else 0
processed_b_files = len(
[f for f in processed_files if f in b_df[file_name_column].values]
)
failed_b_files = (
total_input_files - processed_b_files if total_input_files > 0 else 0
)
yn_cols_b = [col for col in b_df.columns if "(Y/N)" in col]
general_stats = {
"Total Number of Contracts processed": total_input_files,
"# of contracts successfully processed for B fields": processed_b_files,
"# of contracts failed for B fields": failed_b_files,
"# of rows for B run": len(b_df),
"Unique Filenames in B DataFrame": b_df[file_name_column].nunique(),
"Contracts with all B fields missing": b_df[
[col for col in b_df.columns if col != file_name_column]
]
.isnull()
.all(axis=1)
.sum(),
"# of duplicate rows for B run": b_df.duplicated().sum(),
"# of fields for B run": len(b_df.columns),
"Highly Blank Fields in B DataFrame": blanks_check(b_df),
"# of rows with Pages greater than Page Num": b_df.apply(
lambda row: pages_pagenum_check(row["Pages"], row["Page_Num"]), axis=1
).sum(),
"# of invalid Y/N values for B run": b_df[yn_cols_b]
.apply(lambda x: yn_check(x)[1], axis=1)
.sum()
.sum(),
}
b_stats = {}
for field in [col for col in b_df.columns if col != file_name_column]:
missing_contracts = (
b_df.groupby(file_name_column)[field]
.apply(lambda x: x.isnull().all())
.sum()
)
contracts_with_missing = (
b_df.groupby(file_name_column)[field]
.apply(lambda x: x.isnull().any())
.sum()
)
total_missing = b_df[field].isnull().sum()
b_stats[field] = {
"Contracts missing all entries (B DataFrame)": missing_contracts,
"Contracts with at least one missing entry (B DataFrame)": contracts_with_missing,
"Total missing values (Merged DataFrame)": total_missing,
**detect_date_anomalies(b_df[field], field, is_merged=False),
}
return general_stats, b_stats
def get_table_stats(text_dict):
table_count, num_table_pages, rate_count = 0, 0, 0
table_pages, rate_pages, rates_more_than_10 = [], [], []
for page_num, page_text in text_dict.items():
if string_utils.contains_reimbursement(page_text, page_num):
rate_pages.append(page_num)
rates_on_page = sum(1 for char in page_text if char in {"$", "%"})
rate_count += rates_on_page
if rates_on_page >= 10:
rates_more_than_10.append(page_num)
table_texts = re.findall(
r"-------Table Start--------(.*?)-------Table End--------",
page_text,
re.DOTALL,
)
table_page_count = len(table_texts)
if table_page_count > 0:
table_count += table_page_count
num_table_pages += 1
table_pages.append(page_num)
return {
"Table Count": table_count,
"Table Page Count": num_table_pages,
"Table Pages": table_pages,
"Rate Count": rate_count,
"Rate Pages": rate_pages,
"Num Rate Pages": len(rate_pages),
">=10 Rate Pages": rates_more_than_10,
"Num >=10 Rates": len(rates_more_than_10),
}
def table_stats_qaqc(input_dict, ac_df, b_df, file_name_column) -> list[dict]:
table_stats = []
if input_dict:
for filename, content in input_dict.items():
filename_without_ext = io_utils.remove_txt_extension(filename).lower()
text_dict = preprocessing_funcs.split_text(content)
stats = get_table_stats(text_dict)
table_stats.append(
{
"Contract Name": filename,
"Has B output": (
filename_without_ext in b_df[file_name_column].values
if b_df is not None
else "No B Output Given"
),
"Has AC output": (
filename_without_ext in ac_df[file_name_column].values
if ac_df is not None
else "No AC Output Given"
),
"# Tables": stats["Table Count"],
"# Reimb rates": stats["Rate Count"],
"Table Pages": ", ".join(map(str, stats["Table Pages"])),
"Rate Pages": ", ".join(map(str, stats["Rate Pages"])),
">=10 Rate Pages": ", ".join(map(str, stats[">=10 Rate Pages"])),
}
)
return table_stats
def perform_qc_qa(ac_final_df, b_final_df, abc_final_df, input_dict):
# Perform QA/QC analysis
wb = Workbook()
ws_general = wb.active
ws_general.title = "General Stats"
timestamp = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
# Get list of all processed files
processed_files = []
if ac_final_df is not None:
processed_files.extend(ac_final_df["Contract Name"].tolist())
if b_final_df is not None:
processed_files.extend(b_final_df["Contract Name"].tolist())
processed_files = list(set(processed_files))
# Generate QA/QC stats based on available dataframes
if ac_final_df is not None and b_final_df is not None:
general_stats, ac_stats, b_stats = perform_qc_qa_tests_both(
ac_final_df, b_final_df, abc_final_df, input_dict, processed_files
)
elif ac_final_df is not None:
general_stats, ac_stats = perform_qc_qa_tests_ac(
ac_final_df, input_dict, processed_files
)
else:
general_stats, b_stats = perform_qc_qa_tests_b(
b_final_df, input_dict, processed_files
)
# Add metadata to general stats
general_stats_with_metadata = {
"Report Generation Time": timestamp,
"Batch ID": config.BATCH_ID,
"AC Processing": (
"Yes"
if ("a" in config.FIELDS and "c" in config.FIELDS) or (config.AC_DF)
else "No"
),
"B Processing": "Yes" if ("b" in config.FIELDS) or (config.B_DF) else "No",
**general_stats,
}
# Write general stats
general_df = pd.DataFrame.from_dict(
general_stats_with_metadata, orient="index", columns=["Value"]
)
for r in dataframe_to_rows(general_df, index=True, header=True):
ws_general.append(r)
# Write AC stats if available
if ac_final_df is not None:
ws_ac = wb.create_sheet("AC Fields Stats")
ac_stats_df = pd.DataFrame(ac_stats).T
for r in dataframe_to_rows(ac_stats_df, index=True, header=True):
ws_ac.append(r)
# Write B stats if available
if b_final_df is not None:
ws_b = wb.create_sheet("B Fields Stats")
b_stats_df = pd.DataFrame(b_stats).T
for r in dataframe_to_rows(b_stats_df, index=True, header=True):
ws_b.append(r)
# Generate table statistics
table_stats = table_stats_qaqc(
input_dict, ac_final_df, b_final_df, config.FILE_NAME_COLUMN
)
ws_table_stats = wb.create_sheet("Table and Reimbursement Stats")
table_stats_df = pd.DataFrame(table_stats)
for r in dataframe_to_rows(table_stats_df, index=False, header=True):
ws_table_stats.append(r)
return wb
def save_qcqa_wb(wb, run_timestamp):
# Save QC report locally
report_filename = f'QC_QA_Report_{config.BATCH_ID}_{datetime.now().strftime("%Y%m%d_%H%M%S")}.xlsx'
report_path = os.path.join(config.CONSOLIDATED_OUTPUT_DIRECTORY, report_filename)
wb.save(report_path)
print(f"QC/QA Report saved to {report_path}")
# Upload to S3
if config.WRITE_TO_S3:
BUCKET_NAME = config.S3_OUTPUT_BUCKET
print(f"Uploading final outputs to s3://{BUCKET_NAME}/qcqa/{run_timestamp}")
# Upload consolidated outputs
consolidated_files = [
f"{config.BATCH_ID}-AC.csv",
f"{config.BATCH_ID}-B.csv",
f"{config.BATCH_ID}-ABC.csv",
report_filename, # Add QC Report to upload list
]
for filename in consolidated_files:
local_path = os.path.join(config.CONSOLIDATED_OUTPUT_DIRECTORY, filename)
if os.path.exists(local_path):
if filename == report_filename:
s3_key = f"{config.BATCH_ID}/{run_timestamp}/consolidated/qcqa/{filename}"
else:
s3_key = (
f"{config.BATCH_ID}/{run_timestamp}/consolidated/{filename}"
)
try:
print(f"uploading {local_path} to s3://{BUCKET_NAME}/{s3_key}")
config.S3_CLIENT.upload_file(local_path, BUCKET_NAME, s3_key)
except Exception as e:
print(f"Error uploading {local_path}: {str(e)}")
else:
print(f"File not found: {local_path}")
print(
f"Upload completed to s3://{BUCKET_NAME}/{config.BATCH_ID}/{run_timestamp}"
)
"""
QC/QA Validation Utilities.
This module provides comprehensive validation functions for healthcare contract data including:
- Date parsing and formatting
- Medical code validation (CPT, HCPCS, Revenue, Diagnosis codes)
- TIN/NPI validation and formatting
- AI hallucination detection
- Generic validation helpers
"""
import ast
import json
import re
from datetime import datetime, date
from typing import Any, List, Optional, Tuple
import pandas as pd
from dateutil import parser
from pandas.api.types import (
is_bool_dtype,
is_categorical_dtype,
is_datetime64_any_dtype,
is_numeric_dtype,
is_object_dtype,
is_string_dtype,
)
# ================================================================================================
# QC/QA CONSTANTS AND CONFIGURATION
# ================================================================================================
# Column name mappings
QC_CONTRACT_NAME = "FILE_NAME"
QC_LINE_OF_BUSINESS = "AARETE_DERIVED_LOB"
QC_AGREEMENT_NAME = "CONTRACT_TITLE"
QC_FILENAME_TIN = "FILENAME_TIN"
QC_PROV_GROUP_TIN = "PROV_GROUP_TIN"
QC_PROV_GROUP_NPI = "PROV_GROUP_NPI"
QC_PROV_OTHER_TIN = "PROV_OTHER_TIN"
QC_REIMB_PROV_TIN = "REIMB_PROV_TIN"
# JSON columns that should not be scanned for hallucinations
QC_JSON_OK_COLS = {"PROV_INFO_JSON", "PROV_INFO_JSON_FORMATTED"}
# Date pattern strings for extraction
QC_DATE_PATTERNS = [
r"(\d{1,2}[/-]\d{1,2}[/-]\d{2,4})", # 12/31/24 or 12-31-2024
r"(\d{1,2}[.]\d{1,2}[.]\d{2,4})", # 12.31.2024
r"(\d{1,2}[_]\d{1,2}[_]\d{2,4})", # 12_31_2024
r"(\d{1,2}[~]\d{1,2}[~]\d{2,4})", # 12~31~2024
r"(\d{1,2}\s+\d{1,2}\s+\d{2,4})", # 12 31 2024
r"(\d{4}\s+\d{4})", # 2024 2024
r"(\d{8})", # 12312024
r"(\d{6})", # 123024
r"([A-Za-z]+ ?\d{1,2}(?:st|nd|rd|th)?,?\s*\d{2,4})", # January 1st, 2024
]
# Medical code validation patterns (compiled regex)
QC_RE_CPT5 = re.compile(r"^\d{5}$")
QC_RE_HCPCS = re.compile(r"^[A-TV-Z]\d{4}$", re.I)
QC_RE_RANGE_CPT = re.compile(r"^\d{5}-\d{5}$")
QC_RE_RANGE_HCPCS = re.compile(r"^([A-TV-Z])(\d{4})-\1(\d{4})$", re.I)
QC_RE_RANGE_HCPCS_ANY = re.compile(r"^[A-TV-Z]\d{4}-[A-TV-Z]\d{4}$", re.I)
QC_RE_REVENUE4 = re.compile(r"^\d{4}$")
QC_RE_SMASHED_CPT_RANGE = re.compile(r"^\d{10}$")
QC_RE_RANGE_REV = re.compile(r"^\s*(\d{1,4})-(\d{1,4})\s*$")
QC_RE_REV4 = re.compile(r"^\d{4}$")
QC_DIGITS_9_RE = re.compile(r"^\d{9}$")
# Hallucination detection regex flags
QC_RE_FLAGS = re.IGNORECASE | re.MULTILINE
QC_RE_FLAGS_BLOCK = re.IGNORECASE | re.MULTILINE | re.DOTALL
# Hallucination detection patterns
QC_CANNED_RGX = re.compile(
r"(?:\bas an ai (?:language )?model,? i cannot provide\b|"
r"\bi am not certain,? but it may\b|"
r"\binformation is not specified in the contract\b|"
r"\bplease note (?:that )?this is an autogenerated summary\b|"
r"\bdata could not be extracted from (?:the )?source\b|"
r"\bno relevant section found in the document\b|"
r"\bi do not have access to that information\b|"
r"\bthis information is unavailable at this time\b|"
r"\bif you require more information,? consult the client\b|"
r"\bsection not present\b|"
r"\bunable to determine (?:the )?term from context\b)",
QC_RE_FLAGS,
)
QC_GENERIC_RGX = re.compile(
r"(?:\bthis agreement typically includes clauses about\b|"
r"\bsuch provisions are often negotiated\b|"
r"\bcontractual obligations vary depending on circumstances\b|"
r"\bthe answer depends on future negotiations\b|"
r"\bdetails may be outlined elsewhere in the document\b|"
r"\bthis field may include definitions or roles\b|"
r"\bthe contract is effective upon signatures\b|"
r"\bsee above for similar terms\b|"
r"\bmay be governed by applicable law\b)",
QC_RE_FLAGS,
)
QC_ARTIFACTS_RGX = re.compile(
r"(?:"
r"(^|\n)\s*---\s*(\n|$)|" # ---
r"(^|\n)\s*\*\*\*?\s*(\n|$)|" # *** or **
r"(^|\n)\s*~~\s*(\n|$)|" # ~~
r"(^|\n)\s*#{1,6}\s+\S+|" # markdown headers
r"(^|\n)\s*```.+?```|" # fenced code block
r"</?(?:b|i|ul|li|table|tr|td|th|br|p|em|strong|code|pre)[^>]*>" # HTML-ish
r")",
QC_RE_FLAGS_BLOCK,
)
QC_PHANTOM_RGX = re.compile(
r"(?:\bas shown (?:above|below)\b|"
r"\bsee attached\b|"
r"\brefer to (?:table|chart)\s+[A-Za-z0-9]+\b|"
r"\bper clause\s+\d+(?:\.\d+)*\b|"
r"\bsee (?:referenced )?(?:exhibit|appendix)\b|"
r"\bbased on earlier sections of this document\b|"
r"\bper legal citation\b)",
QC_RE_FLAGS,
)
# Yes/No indicator columns that should be normalized
QC_YES_NO_COLUMNS = [
"AUTO_RENEWAL_IND",
"DSH_IND",
"NTAP_IND",
"CARVEOUT_IND",
"DEFAULT_IND",
"RATE_ESCALATOR_IND",
"UC_IND",
"IME_IND",
"LESSER_OF_IND",
"GME_IND",
"GREATER_OF_IND",
"GROUPER_HAC_IND",
"GROUPER_READMISSIONS_IND",
"GROUPER_SEVERITY_IND",
"GROUPER_TRANSFER_IND",
"OUTLIER_FIRST_DOLLAR_IND",
"STOP_LOSS_FIRST_DOLLAR_IND",
"COMPOUND_SPLIT_IND",
"GLOBAL_LESSER_OF_APPLIED",
]
# Standard Line of Business values
QC_STANDARD_LOBS = {"MEDICARE", "MEDICAID", "COMMERCIAL", "HMO", "PPO", "ALL"}
# Values considered as blank/empty
QC_BLANK_VALUES = ["nan", "na", "none", "n/a", "null"]
# Date columns to validate
QC_DATE_COLUMNS = [
"AARETE_DERIVED_EFFECTIVE_DT",
"TERMINATION_DT",
"AARETE_DERIVED_TERMINATION_DT",
"REIMB_EFFECTIVE_DT",
"REIMB_TERMINATION_DT",
"DISCOUNT_START_DT",
"DISCOUNT_END_DT",
"PREMIUM_START_DT",
"PREMIUM_END_DT",
"SEQUESTRATION_START_DT",
"SEQUESTRATION_END_DT",
]
# ================================================================================================
# GENERIC VALIDATION UTILITIES
# ================================================================================================
def append_flag(df: pd.DataFrame, mask: pd.Series, col: str, msg: str) -> pd.DataFrame:
"""
Append flag message to specified column based on mask.
Args:
df: DataFrame to modify
mask: Boolean mask indicating which rows to flag
col: Column name for flags
msg: Message to append (string or Series)
Returns:
Modified DataFrame
"""
if col not in df.columns:
df[col] = ""
else:
df[col] = df[col].fillna("")
if isinstance(msg, pd.Series):
df.loc[mask, col] = df.loc[mask, col].combine(
msg[mask], lambda old, new: f"{old}|{new}" if old else new
)
else:
df.loc[mask, col] = df.loc[mask, col].apply(
lambda x: (str(x) + "|" + msg) if x else msg
)
return df
def normalize_series(s: pd.Series) -> pd.Series:
"""
Normalize a string series for comparison.
Args:
s: Series to normalize
Returns:
Normalized series
"""
s = s.astype("string")
return (
s.str.normalize("NFC")
.str.replace(r"\s+", " ", regex=True)
.str.strip()
.str.lower()
)
def is_filled(series: pd.Series) -> pd.Series:
"""Return boolean mask indicating which values are filled (not blank/NA)."""
return ~(series.astype(str).str.strip().isin(["", "nan", "NaN", "N/A"]))
def is_blank(series: pd.Series) -> pd.Series:
"""Return boolean mask indicating which values are blank/NA."""
return series.astype(str).str.strip().isin(["", "nan", "NaN", "N/A"])
def is_valid_json(value: Any) -> bool:
"""
Check if value is valid JSON (dict or list of dicts).
Args:
value: Value to validate
Returns:
True if valid JSON or blank
"""
if pd.isna(value) or str(value).strip() == "":
return True
try:
parsed = json.loads(value)
if isinstance(parsed, dict):
return True
elif isinstance(parsed, list) and all(
isinstance(item, dict) for item in parsed
):
return True
else:
return False
except (json.JSONDecodeError, TypeError):
return False
def smart_len(x: Any) -> int:
"""Calculate content length robustly for various data types."""
try:
if isinstance(x, (list, dict, set, tuple)):
return len(json.dumps(x))
if isinstance(x, pd.Series):
if pd.isnull(x).any():
return 0
return len(str(x))
if pd.isnull(x):
return 0
if isinstance(x, str):
return len(x)
return len(str(x))
except Exception:
return 0
# ================================================================================================
# DATE UTILITIES
# ================================================================================================
def clean_text(text):
"""
Remove non-ASCII characters except newlines from text.
Args:
text: String, Series, array, or list to clean
Returns:
Cleaned text in the same format as input
"""
if isinstance(text, pd.Series):
return text.apply(clean_text)
if isinstance(text, list):
return pd.Series(text).apply(clean_text).to_list()
if pd.isna(text):
return text
return re.sub(r"[^\x20-\x7E\n]", "", str(text))
def _strip_ordinals(s: str) -> str:
"""Remove ordinal suffixes (st, nd, rd, th) from numbers in a string."""
return re.sub(r"(\d{1,2})(?:st|nd|rd|th|s)\b", r"\1", s, flags=re.IGNORECASE)
def parse_date(date_str: Optional[str]) -> Optional[date]:
"""
Parse a date string into a date object, handling multiple formats.
Supports: MMDDYY, MMDDYYYY, MM/DD/YYYY, MM-DD-YYYY, MMYY-YYYY, natural language dates
Args:
date_str: String representation of a date
Returns:
date object if parsing successful, None otherwise
"""
if not date_str:
return None
try:
s = str(date_str).strip()
# "MMYY YYYY" format
if re.fullmatch(r"\d{4}\s+\d{4}", s):
part = s.split()[0]
mm, yy = int(part[:2]), int(part[2:])
yyyy = 2000 + yy if yy < 50 else 1900 + yy
return datetime(yyyy, mm, 1).date()
# MMDDYY
if re.fullmatch(r"\d{6}", s):
mm, dd, yy = int(s[:2]), int(s[2:4]), int(s[4:6])
yyyy = 2000 + yy if yy < 50 else 1900 + yy
return datetime(yyyy, mm, dd).date()
# MMDDYYYY
if re.fullmatch(r"\d{8}", s):
mm, dd, yyyy = int(s[:2]), int(s[2:4]), int(s[4:8])
return datetime(yyyy, mm, dd).date()
# Strip ordinals and normalize separators
clean = _strip_ordinals(s)
clean = re.sub(r"[-._~]", "/", clean)
return parser.parse(clean, dayfirst=False, yearfirst=False).date()
except Exception:
return None
def extract_date(text: Optional[str]) -> Optional[str]:
"""
Extract a date substring from text using regex patterns.
Args:
text: Text that may contain a date
Returns:
Extracted date string if found, None otherwise
"""
if text is None or (isinstance(text, float) and pd.isna(text)):
return None
s = str(text)
for pattern in QC_DATE_PATTERNS:
m = re.search(pattern, s)
if m:
return m.group(0)
return None
def normalize_to_date(val) -> Optional[date]:
"""
Normalize any value to a date object.
Args:
val: Value to normalize
Returns:
date object if successful, None otherwise
"""
if val is None or (isinstance(val, float) and pd.isna(val)):
return None
if isinstance(val, datetime):
return val.date()
if isinstance(val, date):
return val
s = str(val).strip()
d = parse_date(s)
if d:
return d
sub = extract_date(s)
if sub:
return parse_date(sub)
return None
def format_date_series(series: pd.Series, col_name: str) -> Tuple[pd.Series, pd.Series]:
"""
Format a pandas Series of dates and generate validation flags.
Args:
series: Series containing date values
col_name: Column name for error messages
Returns:
Tuple of (formatted_series, flags_series)
"""
parsed_dates = series.apply(normalize_to_date)
# Output format: YYYY-MM-DD
formatted = parsed_dates.apply(
lambda x: x.strftime("%Y-%m-%d") if pd.notna(x) else ""
)
# Invalid = non-empty original but parse failed
invalid_mask = (series.astype(str).str.strip() != "") & parsed_dates.isna()
# Outlier = parsed but absurd year (<1900 or > today)
today = datetime.now().date()
outlier_mask = (~parsed_dates.isna()) & (
(parsed_dates.apply(lambda d: d.year if d else 0) < 1900) | (parsed_dates > today)
)
flags = pd.Series("", index=series.index, dtype="object")
flags.loc[invalid_mask] = f"{col_name} Invalid"
flags.loc[outlier_mask] = f"{col_name} Outlier"
return formatted, flags
# ================================================================================================
# TIN/NPI VALIDATION
# ================================================================================================
def safe_token_to_str(tok: Any) -> str:
"""
Clean single token: handles float-as-string like '123456789.0',
strips whitespace, pads to 9 digits if needed for TINs.
Args:
tok: Token to clean
Returns:
Cleaned token string
"""
tok = str(tok).strip()
if re.fullmatch(r"\d+\.0+", tok):
tok = str(int(float(tok)))
if tok.isdigit() and len(tok) < 9:
tok = tok.zfill(9)
return tok
def format_or_preserve_tins(val: Any) -> Tuple[str, bool]:
"""
Format TINs to 9 digits, preserving pipe-delimited multiple values.
Args:
val: TIN value (may be pipe-delimited)
Returns:
Tuple of (formatted_tins, has_invalid)
"""
if pd.isna(val) or val is None or str(val).strip() == "":
return ("", True)
parts = [
part.strip()
for part in str(val).split("|")
if part is not None and part.strip() != ""
]
out, any_invalid = [], False
for part in parts:
part = safe_token_to_str(part)
if part is None:
part = ""
elif len(part) == 9 and part.isdigit():
out.append(part)
elif QC_DIGITS_9_RE.fullmatch(part):
out.append(part)
elif len(part) == 8 and part.isdigit():
out.append("0" + part)
else:
out.append(part)
any_invalid = True
return ("|".join([x if x is not None else "" for x in out]), any_invalid)
def is_all_valid_npi(cell: Any) -> bool:
"""
Check if all pipe-delimited tokens in cell are valid 10-digit NPIs.
Args:
cell: Cell value to validate
Returns:
True if all tokens are valid NPIs, or if blank
"""
tokens = [t for t in str(cell).split("|") if t.strip()]
if not tokens:
return True
for tok in tokens:
if not re.fullmatch(r"\d{10}", tok):
return False
return True
# ================================================================================================
# MEDICAL CODE VALIDATION (CPT, HCPCS, REVENUE, DIAGNOSIS)
# ================================================================================================
def safe_listify(cell) -> List[str]:
"""Parse stringified lists; always returns a list."""
s = str(cell).strip()
if s.startswith("[") and s.endswith("]"):
try:
return [str(x) for x in ast.literal_eval(s)]
except Exception:
return [s]
return [s]
def normalize_token(t: str) -> str:
"""Normalize a token: uppercase, trim, remove brackets/quotes, compress spaces."""
return str(t).strip().upper().strip("[](){}'\"\n").replace(" ", "")
def tokenize(raw: str) -> List[str]:
"""Split cell into tokens, normalize, keeping hyphens inside code/range."""
s = str(raw).strip("[](){}'\"\n")
tokens = re.split(r"[|,;/\s]+", s)
return [normalize_token(t) for t in tokens if t]
def dedup_preserve_order(seq: List[str]) -> List[str]:
"""Remove duplicates from list while preserving order."""
seen: set[str] = set()
result = []
for x in seq:
if x not in seen:
seen.add(x)
result.append(x)
return result
def unsmash_cpt_range(tok: str) -> Optional[str]:
"""Convert smashed 10-digit CPT range to proper format. Example: '9047190474' -> '90471-90474'"""
t = normalize_token(tok)
if QC_RE_SMASHED_CPT_RANGE.fullmatch(t):
a, b = t[:5], t[5:]
if QC_RE_CPT5.fullmatch(a) and QC_RE_CPT5.fullmatch(b):
return f"{a}-{b}"
return None
def _normalize_cell(raw: str) -> List[str]:
"""Flatten, normalize, and split input per standardization rules."""
s = str(raw).upper().strip("[](){}'\"\n")
s = re.sub(r"[|;/]", ",", s)
s = re.sub(r"\s*,\s*", ",", s)
s = re.sub(r"\s+", " ", s)
parts = []
for piece in s.split(","):
piece = piece.strip()
if not piece:
continue
if "-" in piece:
parts.append(piece)
else:
parts.extend(piece.split())
return [p for p in parts if p]
def parse_cpt_hcpcs_or_revenue(cell) -> List[str]:
"""
Main parser for CPT/HCPCS/Revenue codes.
Returns only valid codes/ranges.
"""
if pd.isna(cell):
return []
tokens = _normalize_cell(cell)
only_revenue = len(tokens) > 0 and all(
QC_RE_REVENUE4.fullmatch(t) for t in tokens
)
if only_revenue:
return tokens
kept = []
for t in tokens:
clean = t.replace(" ", "")
if QC_RE_RANGE_CPT.fullmatch(
clean
) or QC_RE_RANGE_HCPCS.fullmatch(clean):
kept.append(clean)
elif QC_RE_CPT5.fullmatch(clean) or QC_RE_HCPCS.fullmatch(clean):
kept.append(clean)
return dedup_preserve_order(kept)
def parse_cpt_adapter(cell) -> List[str]:
"""Handle stringified list, add HCPCS any-letter ranges and smashed CPT ranges."""
src = ",".join(safe_listify(cell))
kept = set(parse_cpt_hcpcs_or_revenue(src))
for t in tokenize(src):
smash = unsmash_cpt_range(t)
if smash and QC_RE_RANGE_CPT.fullmatch(smash):
kept.add(smash)
if "-" in t and QC_RE_RANGE_HCPCS_ANY.fullmatch(t):
kept.add(t)
return dedup_preserve_order(kept)
def _tokenize_for_flagging(raw: str) -> List[str]:
"""Tokenize for flagging, normalize, expand unsmashed CPT5 ranges."""
tokens = []
for p in tokenize(raw):
maybe = unsmash_cpt_range(p)
tokens.append(maybe if maybe else p)
return tokens
def _is_revenue_only(tokens: List[str]) -> bool:
"""Check if all tokens are 4-digit revenue codes."""
return bool(tokens and all(QC_RE_REVENUE4.fullmatch(t) for t in tokens))
def _looks_code_like(t: str) -> bool:
"""Check if token looks like a valid code format."""
return bool(
QC_RE_CPT5.fullmatch(t)
or QC_RE_HCPCS.fullmatch(t)
or QC_RE_RANGE_CPT.fullmatch(t)
or QC_RE_RANGE_HCPCS.fullmatch(t)
or QC_RE_RANGE_HCPCS_ANY.fullmatch(t)
or QC_RE_REVENUE4.fullmatch(t)
or t.isalnum()
)
def find_invalid_tokens(cell) -> List[str]:
"""Return invalid tokens (not in adapter output but looks code-like)."""
kept = {normalize_token(k) for k in parse_cpt_adapter(cell)}
toks = [normalize_token(t) for t in _tokenize_for_flagging(cell)]
if not toks or _is_revenue_only(toks):
return []
invalid = [t for t in toks if t not in kept and _looks_code_like(t)]
return dedup_preserve_order(invalid)
def parse_revenue_list(raw) -> List[str]:
"""Return normalized 4-digit revenue codes and ranges, deduped."""
out = []
for base in safe_listify(raw):
for t in tokenize(base):
m = QC_RE_RANGE_REV.fullmatch(t)
if m:
a, b = m.group(1).zfill(4), m.group(2).zfill(4)
token = f"{a}-{b}"
if token not in out:
out.append(token)
elif t.isdigit():
code = t.zfill(4)
if QC_RE_REV4.fullmatch(code) and code not in out:
out.append(code)
return out
def parse_revenue_adapter(cell) -> List[str]:
"""Handle stringified list-style cells for revenue codes."""
bases = safe_listify(cell)
fallback = ",".join(bases)
return parse_revenue_list(fallback)
def find_invalid_revenue_tokens(cell) -> List[str]:
"""Find any token that is not a valid 4-digit revenue code."""
out, seen = [], set()
for t in tokenize(",".join(safe_listify(cell))):
t = t.upper()
if not QC_RE_REVENUE4.fullmatch(t) and t not in seen:
seen.add(t)
out.append(t)
return out
def has_length_n_tokens(value, valid_length: int = 2) -> List[str]:
"""Return only tokens of exact length valid_length."""
out = []
for base in safe_listify(value):
for t in tokenize(base):
if len(t) == valid_length:
out.append(t)
return out
def find_invalid_length_tokens(value, valid_length: int = 2) -> List[str]:
"""Return tokens not exactly valid_length."""
out = []
for base in safe_listify(value):
for t in tokenize(base):
if len(t) != valid_length:
out.append(t)
return dedup_preserve_order(out)
# ================================================================================================
# HALLUCINATION DETECTION
# ================================================================================================
# Compiled patterns
pattern_between_stars = re.compile(r"\*\*.*?\*\*")
pattern_json_like = re.compile(r"\{[^{}]*:[^{}]*\}")
def check_hallucination(df: pd.DataFrame) -> pd.DataFrame:
"""
Check DataFrame for AI hallucinations and artifacts.
Scans all text columns (except JSON columns) for:
- AI-like canned responses
- Generic placeholder text
- Markdown/HTML artifacts
- Phantom references
- JSON blobs in non-JSON fields
Args:
df: DataFrame to check
Returns:
DataFrame with hallucination flags added
"""
for col in df.columns:
if col in QC_JSON_OK_COLS:
continue
if col.startswith("Manual Review Flag"):
continue
if (
is_numeric_dtype(df[col])
or is_datetime64_any_dtype(df[col])
or is_bool_dtype(df[col])
):
continue
if not (
is_string_dtype(df[col])
or is_object_dtype(df[col])
or is_categorical_dtype(df[col])
):
continue
col_vals = df[col].astype(str)
# Various hallucination checks
mask_start = col_vals.str.lower().str.startswith(("it seems like"), na=False)
mask_pattern = col_vals.str.contains(pattern_between_stars, na=False, regex=True)
mask_json_blob = col_vals.str.contains(pattern_json_like, na=False, regex=True)
mask_canned = col_vals.str.contains(QC_CANNED_RGX, na=False, regex=True)
mask_generic = col_vals.str.contains(QC_GENERIC_RGX, na=False, regex=True)
mask_artifact = col_vals.str.contains(QC_ARTIFACTS_RGX, na=False, regex=True)
mask_phantom = col_vals.str.contains(QC_PHANTOM_RGX, na=False, regex=True)
# Combine
mask_hallucination = (
(mask_start & mask_pattern)
| mask_json_blob
| mask_canned
| mask_generic
| mask_artifact
| mask_phantom
)
df = append_flag(
df,
mask_hallucination,
"Manual Review Flag (Other)",
f"{col}- Hallucination",
)
return df