From 5105b4048e03493b355b1e99b0f631eb60e6bc35 Mon Sep 17 00:00:00 2001 From: Katon Minhas Date: Wed, 14 May 2025 16:43:09 +0000 Subject: [PATCH] Merged in feature/testbed-row-check (pull request #528) Feature/testbed row check * Update for row count * Split one-to-n for different output by field type Approved-by: Alex Galarce --- fieldExtraction/src/testbed/test.py | 111 +++++++++---------- fieldExtraction/src/testbed/testbed_utils.py | 4 + 2 files changed, 57 insertions(+), 58 deletions(-) diff --git a/fieldExtraction/src/testbed/test.py b/fieldExtraction/src/testbed/test.py index da35c65..9c1509f 100644 --- a/fieldExtraction/src/testbed/test.py +++ b/fieldExtraction/src/testbed/test.py @@ -10,24 +10,21 @@ import src.testbed.testbed_utils as testbed_utils warnings.filterwarnings("ignore") + ########################## READ AND PREPROCESS ########################## # Read files -testbed = pd.read_excel("v6_DoczyAI_all_test_bed_LIVE_5.6.25.xlsx") # This is the testbed -results = pd.read_csv("inv-test-0509-new-reimb-exhibit-handling-RESULTS.csv") # This is the doczy output +testbed = pd.read_excel("Doczy-Testbed.xlsx") # This is the testbed +results = pd.read_csv("Doczy-Results.csv") # This is the doczy output -# Remove suffixes from the FILE_NAME values (like .txt, .pdf ONLY) -testbed['FILE_NAME'] = testbed['FILE_NAME'].apply(lambda x: '.'.join(x.split('.')[:-1])) -results['FILE_NAME'] = results['FILE_NAME'].apply(lambda x: '.'.join(x.split('.')[:-1])) +# Preprocess testbed and results data +testbed = testbed_utils.testbed_preprocess(testbed) +results = testbed_utils.testbed_preprocess(results) # Ensure FILE_NAME values match between testbed and results common_file_names = set(testbed['FILE_NAME']).intersection(set(results['FILE_NAME'])) testbed = testbed[testbed['FILE_NAME'].isin(common_file_names)] results = results[results['FILE_NAME'].isin(common_file_names)] -# Preprocess testbed and results data -testbed = testbed_utils.testbed_preprocess(testbed) -results = testbed_utils.testbed_preprocess(results) - ########################## Analyze Testbed ########################## print("*************** Test Bed Stats ****************") print(f"{len(testbed['FILE_NAME'].unique())} contracts") @@ -36,6 +33,14 @@ print(f"{len(testbed.columns)} fields") print(f"{(testbed == '').all().sum()} fields not populated:") # Count of completely empty fields print(f"{testbed.columns[(testbed == '').all()].tolist()}") # List of completely empty fields +########################## Row Comparison ########################## +row_comparison = pd.DataFrame(index=testbed['FILE_NAME'].unique()) +row_comparison['testbed'] = testbed.groupby('FILE_NAME').size() +row_comparison['results'] = results.groupby('FILE_NAME').size() + +print("\n*************** Row Count Comparison ****************") +print(f"Files with different row counts: {(row_comparison['testbed'] != row_comparison['results']).sum()}") + ########################## One-to-One Analysis ########################## print("\n*************** One-to-One Stats ****************") one_to_one_fields = FieldSet(file_path=config.FIELD_JSON_PATH).filter(relationship="one_to_one").list_fields() @@ -57,7 +62,6 @@ metrics_df = pd.DataFrame(columns=['Field', 'Precision', 'Recall', 'Accuracy']) comparison_df = pd.DataFrame(index=testbed_one_to_one['FILE_NAME'].unique()) for field in one_to_one_fields: - print(field) if field in testbed.columns and field in results.columns and field != 'FILE_NAME': # Merge and compare values (no need to drop_duplicates here since we already did) @@ -100,60 +104,44 @@ print(metrics_df.to_string( ########################## One-to-N Analysis ########################## print("\n*************** One-to-N Stats ****************") -one_to_n_fields = [ - # "AARETE_DERIVED_PROV_TYPE", - # "AARETE_DERIVED_PRODUCT", - # "AARETE_DERIVED_LOB", - # "AARETE_DERIVED_PROGRAM", - # "AARETE_DERIVED_NETWORK", - # "LESSER_OF_IND", - # "GREATER_OF_IND", - # "AARETE_DERIVED_REIMB_METHOD", - # "UNIT_OF_MEASURE", - # "REIMB_PCT_RATE", - # "REIMB_FEE_RATE", - # "NOT_TO_EXCEED_IND", - # "DEFAULT_IND", - # "AARETE_DERIVED_FEE_SCHEDULE", - # "AARETE_DERIVED_FEE_SCHEDULE_VERSION", - "REIMB_PROV_TIN", - "REIMB_PROV_NPI", - "REIMB_PROV_NAME" - ] # comparison fields -N = len(one_to_n_fields) -accuracies = [] +one_to_n_fields = {"methodology_breakout" : FieldSet(file_path=config.FIELD_JSON_PATH).filter(field_type="methodology_breakout").list_fields(), + "fee_schedule_breakout" : ["AARETE_DERIVED_FEE_SCHEDULE", "AARETE_DERIVED_FEE_SCHEDULE_VERSION"], + "dynamic_primary" : ["AARETE_DERIVED_LOB", "AARETE_DERIVED_PROGRAM", "AARETE_DERIVED_NETWORK", "AARETE_DERIVED_PRODUCT"], + "reimb_prov_info" : ["REIMB_PROV_TIN", "REIMB_PROV_NPI", "REIMB_PROV_NAME"] +} +one_to_n_metrics = {} +for field_type, fields in one_to_n_fields.items(): + N = len(fields) + accuracies = [] -for file in testbed['FILE_NAME'].unique(): - testbed_file = testbed[testbed['FILE_NAME'] == file] - results_file = results[results['FILE_NAME'] == file] + for file in testbed['FILE_NAME'].unique(): + testbed_file = testbed[testbed['FILE_NAME'] == file] + results_file = results[results['FILE_NAME'] == file] - # Ensure there are rows in both dataframes - if testbed_file.shape[0] == 0 or results_file.shape[0] == 0: - print(f"No rows in either labels or predictions for file: {file}") - continue - - confusion_matrix = testbed_utils.match_rows(one_to_n_fields, testbed_file, results_file, N) + # Ensure there are rows in both dataframes + if testbed_file.shape[0] == 0 or results_file.shape[0] == 0: + print(f"No rows in either labels or predictions for file: {file}") + continue + + confusion_matrix = testbed_utils.match_rows(fields, testbed_file, results_file, N) - confusion_matrix["Filename"] = file - accuracies.append(confusion_matrix) + confusion_matrix["Filename"] = file + accuracies.append(confusion_matrix) -precision_df, recall_df, overall = testbed_utils.calculate_precision_recall(accuracies) + precision_df, recall_df, overall = testbed_utils.calculate_precision_recall(accuracies) + one_to_n_metrics[field_type] = {'precision': precision_df, 'recall': recall_df, 'overall': overall} -print("\nOverall Metrics:") -print(overall.to_string( - index=True, - float_format=lambda x: '{:.2f}'.format(x) if pd.notnull(x) else 'Not found', - justify='left', - col_space={'tp': 15, 'fp': 15, 'fn': 15, 'prec': 15, 'rec': 15, 'f1': 15} -)) + print(f"\nOverall Metrics: {field_type}") + print(overall.to_string( + index=True, + float_format=lambda x: '{:.2f}'.format(x) if pd.notnull(x) else 'Not found', + justify='left', + col_space={'tp': 15, 'fp': 15, 'fn': 15, 'prec': 15, 'rec': 15, 'f1': 15} + )) ########################## Order-Independent Provider Analysis ########################## print("\n*************** Order-Independent Provider Analysis ****************") -# Use the specialized function for provider fields that: -# 1. puts GROUP and OTHER fields together -# 2. ignores the order of the values in the fields - provider_results = testbed_utils.evaluate_provider_fields_separately( testbed, results, verbose=True) @@ -198,6 +186,7 @@ with pd.ExcelWriter(output_file, engine='openpyxl') as writer: empty_fields_df = pd.DataFrame({'Empty Fields': testbed.columns[(testbed == '').all()].tolist()}) stats_df.to_excel(writer, sheet_name='Testbed Stats', index=False) + row_comparison.to_excel(writer, sheet_name='Row Counts') empty_fields_df.to_excel(writer, sheet_name='Empty Fields', index=False) # One-to-One Results @@ -205,9 +194,15 @@ with pd.ExcelWriter(output_file, engine='openpyxl') as writer: comparison_df.to_excel(writer, sheet_name='One-to-One Details') # Add new sheet # One-to-N Results - precision_df.to_excel(writer, sheet_name='One-to-N Precision', index=False) - recall_df.to_excel(writer, sheet_name='One-to-N Recall', index=False) - overall.to_excel(writer, sheet_name='One-to-N Overall') + for field_type, metrics in one_to_n_metrics.items(): + precision_df = metrics['precision'] + recall_df = metrics['recall'] + overall = metrics['overall'] + + # Add the field type to the sheet names + precision_df.to_excel(writer, sheet_name=f'{field_type} Precision', index=False) + recall_df.to_excel(writer, sheet_name=f'{field_type} Recall', index=False) + overall.to_excel(writer, sheet_name=f'{field_type} Overall') # Order-independent provider analysis provider_metrics_df.to_excel(writer, sheet_name='Provider Fields', index=False) diff --git a/fieldExtraction/src/testbed/testbed_utils.py b/fieldExtraction/src/testbed/testbed_utils.py index 19d995b..eb29bef 100644 --- a/fieldExtraction/src/testbed/testbed_utils.py +++ b/fieldExtraction/src/testbed/testbed_utils.py @@ -33,6 +33,10 @@ def normalize_tin_npi(value): return value def testbed_preprocess(df): + + # Capitalize and remove ".TXT" suffixes from FILE_NAME + df['FILE_NAME'] = df['FILE_NAME'].str.upper().str.replace(".TXT", "", regex=False) + # Apply numeric conversion to non-CPT/diag/rev fields only for col in df.columns: if "_CD" not in col and "CPT" not in col and "DIAG" not in col and "REV" not in col: