Merged in tin-npi-acc-testing (pull request #516)

Update testbed data sources and enhance metric calculations for blank fields

* Update testbed data sources and enhance metric calculations for blank fields

* Add order-independent provider analysis

* add tin-npi normalization and verbose

* Enhance provider field evaluation by excluding "UNKNOWN" and "NO_IDENTIFIERS_FOUND" values

* List-based method

* Enhance provider field evaluation by incorporating frequency-based metrics and excluding "UNKNOWN" values

* Fix splits

* fix headers

* fix comma splitting

* Merged main into tin-npi-acc-testing

* Merged main into tin-npi-acc-testing

* Refine provider metrics logging to exclude empty predicted values

* Merge branch 'tin-npi-acc-testing' of https://bitbucket.org/aarete/doczy.ai into tin-npi-acc-testing

* Merged main into tin-npi-acc-testing

* last commit before changing back to sets

* Refactor evaluation metrics calculation to use set operations (while still displaying counters for info)

* Merge branch 'tin-npi-acc-testing' of https://bitbucket.org/aarete/doczy.ai into tin-npi-acc-testing


Approved-by: Katon Minhas
This commit is contained in:
Alex Galarce
2025-05-12 18:23:50 +00:00
committed by Katon Minhas
parent 147f91eaa9
commit 6bc345ef9e
2 changed files with 328 additions and 20 deletions
+55 -19
View File
@@ -12,8 +12,8 @@ warnings.filterwarnings("ignore")
########################## READ AND PREPROCESS ##########################
# Read files
testbed = pd.read_excel("Doczy-Testbed.xlsx") # This is the testbed
results = pd.read_csv("inv-test-0430-RESULTS.csv") # This is the script results
testbed = pd.read_excel("v6_DoczyAI_all_test_bed_LIVE_5.6.25.xlsx") # This is the testbed
results = pd.read_csv("inv-test-0509-new-reimb-exhibit-handling-RESULTS.csv") # This is the doczy output
# Remove suffixes from the FILE_NAME values (like .txt, .pdf ONLY)
testbed['FILE_NAME'] = testbed['FILE_NAME'].apply(lambda x: '.'.join(x.split('.')[:-1]))
@@ -80,7 +80,7 @@ for field in one_to_one_fields:
raise
# Calculate metrics
precision, recall, accuracy = testbed_utils.calculate_field_metrics(testbed_one_to_one,
precision, recall, accuracy = testbed_utils.calculate_field_metrics_for_blanks(testbed_one_to_one,
results_one_to_one,
field)
metrics_df.loc[len(metrics_df)] = [field, precision, recall, accuracy]
@@ -100,24 +100,26 @@ print(metrics_df.to_string(
########################## One-to-N Analysis ##########################
print("\n*************** One-to-N Stats ****************")
one_to_n_fields = ["AARETE_DERIVED_PROV_TYPE",
"AARETE_DERIVED_PRODUCT",
"AARETE_DERIVED_LOB",
"AARETE_DERIVED_PROGRAM",
"AARETE_DERIVED_NETWORK",
"LESSER_OF_IND",
"GREATER_OF_IND",
"AARETE_DERIVED_REIMB_METHOD",
"UNIT_OF_MEASURE",
"REIMB_PCT_RATE",
"REIMB_FEE_RATE",
"NOT_TO_EXCEED_IND",
"DEFAULT_IND",
"AARETE_DERIVED_FEE_SCHEDULE",
"AARETE_DERIVED_FEE_SCHEDULE_VERSION",
one_to_n_fields = [
# "AARETE_DERIVED_PROV_TYPE",
# "AARETE_DERIVED_PRODUCT",
# "AARETE_DERIVED_LOB",
# "AARETE_DERIVED_PROGRAM",
# "AARETE_DERIVED_NETWORK",
# "LESSER_OF_IND",
# "GREATER_OF_IND",
# "AARETE_DERIVED_REIMB_METHOD",
# "UNIT_OF_MEASURE",
# "REIMB_PCT_RATE",
# "REIMB_FEE_RATE",
# "NOT_TO_EXCEED_IND",
# "DEFAULT_IND",
# "AARETE_DERIVED_FEE_SCHEDULE",
# "AARETE_DERIVED_FEE_SCHEDULE_VERSION",
"REIMB_PROV_TIN",
"REIMB_PROV_NPI",
"REIMB_PROV_NAME"] # comparison fields
"REIMB_PROV_NAME"
] # comparison fields
N = len(one_to_n_fields)
accuracies = []
@@ -146,6 +148,37 @@ print(overall.to_string(
col_space={'tp': 15, 'fp': 15, 'fn': 15, 'prec': 15, 'rec': 15, 'f1': 15}
))
########################## Order-Independent Provider Analysis ##########################
print("\n*************** Order-Independent Provider Analysis ****************")
# Use the specialized function for provider fields that:
# 1. puts GROUP and OTHER fields together
# 2. ignores the order of the values in the fields
provider_results = testbed_utils.evaluate_provider_fields_separately(
testbed,
results, verbose=True)
# Display the results
provider_metrics_df = pd.DataFrame([
{
'Field': f"Provider {field_type}",
'Precision': metrics['precision'],
'Recall': metrics['recall'],
'F1': metrics['f1'],
'TP': metrics['tp'],
'FP': metrics['fp'],
'FN': metrics['fn'],
'average_accuracy': metrics['average_accuracy'],
}
for field_type, metrics in provider_results.items()
])
print(provider_metrics_df.to_string(
index=False,
float_format=lambda x: '{:.2f}'.format(x) if pd.notnull(x) else 'Not found',
justify='left',
col_space={'Field': 15, 'Precision': 12, 'Recall': 12, 'F1': 12, 'TP': 8, 'FP': 8, 'FN': 8}
))
########################## Export to Excel ##########################
print("\nExporting results to Excel...")
@@ -175,6 +208,9 @@ with pd.ExcelWriter(output_file, engine='openpyxl') as writer:
precision_df.to_excel(writer, sheet_name='One-to-N Precision', index=False)
recall_df.to_excel(writer, sheet_name='One-to-N Recall', index=False)
overall.to_excel(writer, sheet_name='One-to-N Overall')
# Order-independent provider analysis
provider_metrics_df.to_excel(writer, sheet_name='Provider Fields', index=False)
# Auto-adjust column widths
for sheet_name in writer.sheets:
+273 -1
View File
@@ -3,6 +3,8 @@ import pandas as pd
import src.utils.string_utils as string_utils
import itertools
from collections import Counter
def convert_to_float(value):
try:
@@ -10,6 +12,22 @@ def convert_to_float(value):
except:
return value
def normalize_tin_npi(value):
"""Normalize TIN/NPI values by removing decimal points and trailing zeros."""
if string_utils.is_empty(value):
return ""
# Skip "UNKNOWN" and non-numeric values
if value == "UNKNOWN" or not value.replace('.', '', 1).isdigit():
return value
# Convert to integer-like string by removing decimal part
try:
# Remove decimal and any trailing zeros
return str(int(float(value)))
except:
return value
def testbed_preprocess(df):
df = df.applymap(convert_to_float)
df = df.applymap(lambda x: str(x).strip().upper())
@@ -17,9 +35,13 @@ def testbed_preprocess(df):
df = df.replace("NAN", "")
df = df.replace("N/N", "")
# Normalize TIN and NPI columns
tin_columns = [col for col in df.columns if 'TIN' in col]
for col in tin_columns:
npi_columns = [col for col in df.columns if 'NPI' in col]
for col in tin_columns + npi_columns:
df[col] = df[col].str.replace("-", "", regex=False)
df[col] = df[col].apply(normalize_tin_npi)
df[col] = df[col].str.replace(" ", "", regex=False)
df["AUTO_RENEWAL_TERM"] = df['AUTO_RENEWAL_TERM'].str.replace("ONE YEAR", "1 YEAR", regex=False)
@@ -156,9 +178,259 @@ def calculate_field_metrics(testbed_df, results_df, field):
return precision, recall, accuracy
def calculate_field_metrics_for_blanks(testbed_df, results_df, field):
"""Calculate metrics specially handling completely blank fields.
Args:
testbed_df (_type_): DataFrame containing ground truth
results_df (_type_): DataFrame containing predictions
field (_type_): Name of field to evaluate
Returns:
tuple: (precision, recall, accuracy)
"""
# Check if field is completely blank in both datasets
testbed_all_blank = string_utils.is_empty(testbed_df[field], pd_mask=True).all()
results_all_blank = string_utils.is_empty(results_df[field], pd_mask=True).all()
if testbed_all_blank and results_all_blank:
# Both are all blank - perfect agreement
return 1.0, 1.0, 1.0
# Regular calculation for non-blank fields
return calculate_field_metrics(testbed_df, results_df, field)
def evaluate_provider_fields_separately(testbed_df, results_df, verbose=False):
"""Evaluates TIN, NPI, and NAME fields separately for each provider."""
# Initialize metrics for each field type
metrics = {
"TIN": {'tp' : 0, 'fp': 0, 'fn': 0, 'accuracy_file': []},
"NPI": {'tp' : 0, 'fp': 0, 'fn': 0, 'accuracy_file': []},
"NAME": {'tp' : 0, 'fp': 0, 'fn': 0, 'accuracy_file': []}
}
if verbose: # instantiate provider_metrics.txt if verbose
csv_output_dicts = []
with open("provider_metrics.txt", "w") as f:
f.write("Provider Metrics\n")
for file in testbed_df['FILE_NAME'].unique():
# Get rows for this file
testbed_file = testbed_df[testbed_df['FILE_NAME'] == file]
results_file = results_df[results_df['FILE_NAME'] == file]
# Extract all TINs, NPIs, and NAMEs for ground truth
truth_tins = []
truth_npis = []
truth_names = []
# Add group provider info (with splitting)
for col_type, truth_list in [
("PROV_GROUP_TIN", truth_tins),
("PROV_GROUP_NPI", truth_npis),
("PROV_GROUP_NAME_FULL", truth_names)
]:
value = testbed_file[col_type].iloc[0]
values = []
pipe_split = value.split("|")
for item in pipe_split:
if col_type in ["PROV_GROUP_NPI", "PROV_GROUP_TIN"] and "," in item: # further split by comma, only if col_type is PROV_GROUP_NPI or PROV_GROUP_TIN
for comma_item in item.split(","):
comma_item = comma_item.strip()
if comma_item:
values.append(comma_item)
elif item.strip():
values.append(item.strip())
# Now append all values to the list
for v in values:
if v and v != "UNKNOWN":
truth_list.append(v)
# Add other provider info
for col_type, truth_list in [
("PROV_OTHER_TIN", truth_tins),
("PROV_OTHER_NPI", truth_npis),
("PROV_OTHER_NAME_FULL", truth_names)
]:
if not string_utils.is_empty(testbed_file[col_type].iloc[0]):
values = testbed_file[col_type].iloc[0].split("|")
if len(values) == 1: # possibly delimited by comma
values = testbed_file[col_type].iloc[0].split(",")
for value in values:
value = value.strip()
if value:
truth_list.append(value)
# Similarly, extract all TINs, NPIs, and NAMEs for predictions
pred_tins = []
pred_npis = []
pred_names = []
# Add group provider info
value = results_file['PROV_GROUP_TIN'].iloc[0]
pred_tins.append(value)
value = results_file['PROV_GROUP_NPI'].iloc[0]
pred_npis.append(value)
value = results_file['PROV_GROUP_NAME_FULL'].iloc[0]
pred_names.append(value)
# Add other provider info
for col_type, pred_list in [
("PROV_OTHER_TIN", pred_tins),
("PROV_OTHER_NPI", pred_npis),
("PROV_OTHER_NAME_FULL", pred_names)
]:
values = results_file[col_type].iloc[0].split("|")
if len(values) == 1: # possibly delimited by comma
values = results_file[col_type].iloc[0].split(",")
for value in values:
value = value.strip()
pred_list.append(value)
file_metrics = {} # Store file-specific metrics for verbose output
# Calculate metrics for each field type and print verbose output if requested
for field_type, truth_list, pred_list in [
("TIN", truth_tins, pred_tins),
("NPI", truth_npis, pred_npis),
("NAME", truth_names, pred_names)
]:
# Count occurrences in lists, excluding empty strings and "UNKNOWN"
truth_counter = Counter([x for x in truth_list if x and x != "UNKNOWN"])
pred_counter = Counter([x for x in pred_list if x and x != "UNKNOWN"])
# Filter lists to remove empty strings and "UNKNOWN"
filtered_truth_list = [x for x in truth_list if x and x != "UNKNOWN"]
filtered_pred_list = [x for x in pred_list if x and x != "UNKNOWN"]
# Keep counters for display/debugging purposes
truth_counter = Counter(filtered_truth_list)
pred_counter = Counter(filtered_pred_list)
# Convert to sets for set operations
truth_set = set(filtered_truth_list)
pred_set = set(filtered_pred_list)
# Compute metrics using set operations
file_tp = len(truth_set.intersection(pred_set)) # True Positives
file_fp = len(pred_set - truth_set) # False Positives
file_fn = len(truth_set - pred_set) # False Negatives
# Calculate accuracy for this file
file_accuracy = file_tp / (file_tp + file_fp + file_fn) if (file_tp + file_fp + file_fn) > 0 else 0
# Update overall metrics
metrics[field_type]['tp'] += file_tp
metrics[field_type]['fp'] += file_fp
metrics[field_type]['fn'] += file_fn
metrics[field_type]['accuracy_file'].append(file_accuracy)
# Store file-specific metrics for verbose output
file_metrics[field_type] = {
"truth_counter": dict(truth_counter),
"pred_counter": dict(pred_counter),
"file_tp": file_tp,
"file_fp": file_fp,
"file_fn": file_fn,
"file_accuracy": file_accuracy
}
if verbose:
# Write verbose output to file
# Initialize file if it doesn't exist
with open("provider_metrics.txt", "a") as f:
f.write(f"File: {file}\n")
f.write(f"Total TINs: {len(set([tin for tin in truth_tins if tin and tin != 'UNKNOWN']))}, Predicted TINs: {len(set([tin for tin in pred_tins if tin and tin != 'UNKNOWN']))}\n")
f.write(f"Total NPIs: {len(set([npi for npi in truth_npis if npi and npi != 'UNKNOWN']))}, Predicted NPIs: {len(set([npi for npi in pred_npis if npi and npi != 'UNKNOWN']))}\n")
f.write(f"Total Names: {len(set([name for name in truth_names if name and name != 'UNKNOWN']))}, Predicted Names: {len(set([name for name in pred_names if name and name != 'UNKNOWN']))}\n")
f.write("-" * 50 + "\n")
for field_type in metrics:
csv_output_dict = {}
csv_output_dict["Filename"] = file
csv_output_dict["Test Bed TINs"] = len(set([tin for tin in truth_tins if tin and tin != 'UNKNOWN']))
csv_output_dict["Test Bed NPIs"] = len(set([npi for npi in truth_npis if npi and npi != 'UNKNOWN']))
csv_output_dict["Test Bed Names"] = len(set([name for name in truth_names if name and name != 'UNKNOWN']))
csv_output_dict["Predicted TINs"] = len(set([tin for tin in pred_tins if tin and tin != 'UNKNOWN']))
csv_output_dict["Predicted NPIs"] = len(set([npi for npi in pred_npis if npi and npi != 'UNKNOWN']))
csv_output_dict["Predicted Names"] = len(set([name for name in pred_names if name and name != 'UNKNOWN']))
print_lines = [
f"Field Type: {field_type}",
f"Ground Truth: {dict(file_metrics[field_type]['truth_counter'])}",
f"Predictions: {dict(file_metrics[field_type]['pred_counter'])}",
f"True Positives: {file_metrics[field_type]['file_tp']}",
f"False Positives: {file_metrics[field_type]['file_fp']}",
f"False Negatives: {file_metrics[field_type]['file_fn']}",
f"File Accuracy: {file_metrics[field_type]['file_accuracy']}",
f"Precision: {file_metrics[field_type]['file_tp'] / (file_metrics[field_type]['file_tp'] + file_metrics[field_type]['file_fp']) if (file_metrics[field_type]['file_tp'] + file_metrics[field_type]['file_fp']) > 0 else 0}",
f"Recall: {file_metrics[field_type]['file_tp'] / (file_metrics[field_type]['file_tp'] + file_metrics[field_type]['file_fn']) if (file_metrics[field_type]['file_tp'] + file_metrics[field_type]['file_fn']) > 0 else 0}",
"-" * 50
]
csv_output_dict["field_type"] = field_type
csv_output_dict["Ground Truth"] = dict(file_metrics[field_type]['truth_counter']).copy()
csv_output_dict["Predictions"] = dict(file_metrics[field_type]['pred_counter']).copy()
csv_output_dict["True Positives"] = file_metrics[field_type]['file_tp']
csv_output_dict["False Positives"] = file_metrics[field_type]['file_fp']
csv_output_dict["False Negatives"] = file_metrics[field_type]['file_fn']
csv_output_dict["File Accuracy"] = file_metrics[field_type]['file_accuracy']
csv_output_dict["Precision"] = file_metrics[field_type]['file_tp'] / (file_metrics[field_type]['file_tp'] + file_metrics[field_type]['file_fp']) if (file_metrics[field_type]['file_tp'] + file_metrics[field_type]['file_fp']) > 0 else 0
csv_output_dict["Recall"] = file_metrics[field_type]['file_tp'] / (file_metrics[field_type]['file_tp'] + file_metrics[field_type]['file_fn']) if (file_metrics[field_type]['file_tp'] + file_metrics[field_type]['file_fn']) > 0 else 0
csv_output_dicts.append(csv_output_dict)
with open("provider_metrics.txt", "a") as f:
f.write("\n".join(print_lines) + "\n")
# Convert the list of dictionaries to a DataFrame
csv_output_df = pd.DataFrame(csv_output_dicts)
# Save the DataFrame to a CSV file
csv_output_df.to_csv("provider_metrics.csv", index=False)
# Calculate accuracy, precision, recall, and F1 for each field type
results = {}
for field_type, field_metrics in metrics.items():
precision = field_metrics['tp'] / (field_metrics['tp'] + field_metrics['fp']) if (field_metrics['tp'] + field_metrics['fp']) > 0 else 0
recall = field_metrics['tp'] / (field_metrics['tp'] + field_metrics['fn']) if (field_metrics['tp'] + field_metrics['fn']) > 0 else 0
f1 = (2 * precision * recall) / (precision + recall) if (precision + recall) > 0 else 0
results[field_type] = {
"precision": precision,
"recall": recall,
"f1": f1,
"tp": field_metrics['tp'],
"fp": field_metrics['fp'],
"fn": field_metrics['fn'],
"average_accuracy": sum(field_metrics['accuracy_file']) / len(field_metrics['accuracy_file']) if field_metrics['accuracy_file'] else 0
}
if verbose:
print("Verbose output written to provider_metrics.txt")
return results
def match_rows(fields, labels_df, predictions_df, N):
confusion_matrix = {field_name: {"tp": 0, "fp": 0} for field_name in fields}
# Check for completely blank fields and handle them separately
for field in fields:
labels_all_blank = string_utils.is_empty(labels_df[field], pd_mask=True).all()
preds_all_blank = string_utils.is_empty(predictions_df[field], pd_mask=True).all()
if labels_all_blank and preds_all_blank:
# Perfect agreement on blank field
confusion_matrix[field]["tp"] = min(len(labels_df), len(predictions_df))
confusion_matrix[field]["fp"] = 0
confusion_matrix[field]["fn"] = 0
# Skip this field in the regular matching process
fields = [f for f in fields if f != field]
# Only proceed with matching if there are fields less to process
if not fields:
return confusion_matrix
# Track which label rows have been matched
matched_label_indices = set()