Merged in feature/test-bed-facilitation (pull request #404)

Feature/test bed facilitation for 1-1 fields

* start comparing output to label data

* Enhance load_test_files.py to merge and reorder labels and predictions, outputting to CSV

* Refactor load_test_files.py to streamline label and prediction loading, sorting, and CSV output for date and TIN/NPI fields

* rename to one_to_one_comparison

* Merged main into feature/test-bed-facilitation

* rename notebook_config

* Merge branch 'feature/test-bed-facilitation' of https://bitbucket.org/aarete/doczy.ai into feature/test-bed-facilitation

* Merged main into feature/test-bed-facilitation


Approved-by: Katon Minhas
This commit is contained in:
Alex Galarce
2025-02-20 16:31:29 +00:00
committed by Katon Minhas
parent 664397d713
commit dcde32e7af
@@ -0,0 +1,110 @@
import pandas as pd
from src import config
from src.prompts.investment_prompts import FieldSet
script_config = {"labels_location_dates" : "/home/arg/20250219_test_bed_facilitation/fieldExtraction/src/testbed/labeled_files/Document_Dates_2-13.xlsx",
"sheetname_labels_dates" : "doczy-ai-test-20240211",
"labels_location_tin_npi" : "/home/arg/20250219_test_bed_facilitation/fieldExtraction/src/testbed/labeled_files/TIN_NPI_Names_2-14.xlsx",
"sheetname_labels_tin_npi" : "doczy-ai-test-20240211",
"predictions_location" : "/home/arg/20250219_test_bed_facilitation/fieldExtraction/src/testbed/prediction_files/doczy-inv-consolidated-FEB12.csv",
}
def load_labels_file(filename: str, sheetname: str) -> pd.DataFrame:
df = pd.read_excel(filename, sheet_name=sheetname)
return df
def load_predictions(filename: str) -> pd.DataFrame:
df = pd.read_csv(filename)
return df
def consolidate_one_to_one_fields_in_predictions(predictions_df: pd.DataFrame) -> pd.DataFrame:
"""One-to-one fields need to be consolidated (by taking the first value) across CONTRACT_FILE_NAMEs.
Args:
predictions_df (pd.DataFrame): Predictions DataFrame
Returns:
pd.DataFrame: DataFrame with consolidated one-to-one fields
"""
one_to_one_fields = FieldSet(file_path=config.FIELD_JSON_PATH).filter(relationship="one_to_one").list_fields()
print(one_to_one_fields)
# Group by CONTRACT_FILE_NAME and take just the one-to-one fields. Since they'll be repeated
# across rows, we'll just take the first one
consolidated_df = predictions_df.groupby('CONTRACT_FILE_NAME').first().reset_index()
return consolidated_df
def sort_label_and_prediction_columns(merged_df: pd.DataFrame) -> pd.DataFrame:
"""Place prediction and label columns next to each other in the DataFrame.
Args:
merged_df (pd.DataFrame): Merged DataFrame of labels and predictions
Returns:
pd.DataFrame: Sorted DataFrame with labels and predictions next to each other. The first column is always CONTRACT_FILE_NAME.
"""
# List of columns in the original merged DataFrame
columns = merged_df.columns.tolist()
# Separate labels and predictions columns
labels_columns = [col for col in columns if col.endswith('_labels')]
predictions_columns = [col for col in columns if col.endswith('_predictions')]
# Reorder columns to put corresponding labels and predictions together
ordered_columns = []
for label_col in labels_columns:
# Find corresponding prediction column by removing '_labels' and adding '_predictions'
pred_col = label_col.replace('_labels', '_predictions')
if pred_col in predictions_columns:
ordered_columns.append(label_col)
ordered_columns.append(pred_col)
# Reindex the DataFrame with the new column order
merged_df = merged_df[["CONTRACT_FILE_NAME"] + ordered_columns]
return merged_df
if __name__ == "__main__":
# Load predictions
df_predictions = load_predictions(script_config["predictions_location"])
# Consolidate one-to-one fields in predictions
consolidated_one_to_one_predictions = consolidate_one_to_one_fields_in_predictions(df_predictions)
## Dates fields
# Load labels
filename_labels = script_config["labels_location_dates"]
sheetname_labels = script_config["sheetname_labels_dates"]
df_labels_dates = load_labels_file(filename_labels, sheetname_labels)
# Extract date fields from labels and predictions (also CONTRACT_FILE_NAME)
date_fields = df_labels_dates.columns.tolist()
print(date_fields)
predicted_date_fields = consolidated_one_to_one_predictions[date_fields]
# Join labels and predictions on CONTRACT_FILE_NAME
merged_df = pd.merge(df_labels_dates, predicted_date_fields, on="CONTRACT_FILE_NAME", suffixes=['_labels', '_predictions'])
sorted_df = sort_label_and_prediction_columns(merged_df)
sorted_df.to_csv("merged_predictions_labels_DATES.csv", index=False)
## TIN/NPI fields
# Load labels
filename_labels = script_config["labels_location_tin_npi"]
sheetname_labels = script_config["sheetname_labels_tin_npi"]
df_labels_tin_npi = load_labels_file(filename_labels, sheetname_labels)
# Extract TIN/NPI fields from labels and predictions (also CONTRACT_FILE_NAME)
tin_npi_fields = df_labels_tin_npi.columns.tolist()
predicted_tin_npi_fields = consolidated_one_to_one_predictions[tin_npi_fields]
# Join labels and predictions on CONTRACT_FILE_NAME
merged_df = pd.merge(df_labels_tin_npi, predicted_tin_npi_fields, on="CONTRACT_FILE_NAME", suffixes=['_labels', '_predictions'])
sorted_df = sort_label_and_prediction_columns(merged_df)
sorted_df.to_csv("merged_predictions_labels_TIN_NPI.csv", index=False)