Merged in bugfix/termGroupHotfix (pull request #324)

Bugfix/termGroupHotfix

* Minor term group fix

* Added xlsb to gitignore, updated apply_cnc_hotfix for Filename prefix, added xlsb read functionality to utils

* Date bugfix

* Fixed effective date bug

* Effective date bug fixes

* Fix agreement name field name bug

* Further bug fixes

* update poetry for pyxlsb

* Write to excel after each file is completed

* Fix for restarts


Approved-by: Alex Galarce
This commit is contained in:
Katon Minhas
2024-12-09 21:40:53 +00:00
committed by Alex Galarce
parent d76c24211a
commit d62e477436
6 changed files with 252 additions and 52 deletions
+1 -1
View File
@@ -84,4 +84,4 @@ humana-output/
output_consolidated/ output_consolidated/
*.zip *.zip
*.xlsx *.xlsx
*.xlsb
+12 -1
View File
@@ -2339,6 +2339,17 @@ files = [
{file = "pywinpty-2.0.14.tar.gz", hash = "sha256:18bd9529e4a5daf2d9719aa17788ba6013e594ae94c5a0c27e83df3278b0660e"}, {file = "pywinpty-2.0.14.tar.gz", hash = "sha256:18bd9529e4a5daf2d9719aa17788ba6013e594ae94c5a0c27e83df3278b0660e"},
] ]
[[package]]
name = "pyxlsb"
version = "1.0.10"
description = "Excel 2007-2010 Binary Workbook (xlsb) parser"
optional = false
python-versions = "*"
files = [
{file = "pyxlsb-1.0.10-py2.py3-none-any.whl", hash = "sha256:87c122a9a622e35ca5e741d2e541201d28af00fb46bec492cfa9586890b120b4"},
{file = "pyxlsb-1.0.10.tar.gz", hash = "sha256:8062d1ea8626d3f1980e8b1cfe91a4483747449242ecb61013bc2df85435f685"},
]
[[package]] [[package]]
name = "pyyaml" name = "pyyaml"
version = "6.0.2" version = "6.0.2"
@@ -3238,4 +3249,4 @@ files = [
[metadata] [metadata]
lock-version = "2.0" lock-version = "2.0"
python-versions = "^3.12" python-versions = "^3.12"
content-hash = "0e35b66a1f66f50246b5659fd9caccabf943ff29e5685e07e17a18d870b40def" content-hash = "ddbf73cc29bce29753fb812a03ecac342c22cde0a4622e8a53bd67862b329604"
+1
View File
@@ -14,6 +14,7 @@ anthropic = "^0.36.0"
python-dotenv = "^1.0.1" python-dotenv = "^1.0.1"
psutil = "^6.1.0" psutil = "^6.1.0"
rapidfuzz = "^3.10.1" rapidfuzz = "^3.10.1"
pyxlsb = "^1.0.10"
[tool.poetry.group.dev.dependencies] [tool.poetry.group.dev.dependencies]
black = "^24.10.0" black = "^24.10.0"
+47 -20
View File
@@ -8,10 +8,12 @@ import valid
import pandas as pd import pandas as pd
import os
from rapidfuzz import fuzz, process from rapidfuzz import fuzz, process
import concurrent.futures import concurrent.futures
ABC_PATH = 'UPDATE_PATH_HERE' clean_file_name = config.get_arg_value('clean_file', '')
ABC_PATH = f'clean_output/{clean_file_name}'
def get_highest_similarity(target_str, str_list): def get_highest_similarity(target_str, str_list):
if target_str in str_list: if target_str in str_list:
@@ -95,34 +97,51 @@ def process_hotfix(abc, contract_text):
def main(): def main():
# Create Excel
file_path = f'{config.BATCH_ID}-Hotfix-Final.xlsx'
if not os.path.exists(file_path):
already_ran = None
with pd.ExcelWriter(file_path, engine='openpyxl') as writer:
pd.DataFrame(columns=cnc_hotfix.HOTFIX_ORDER).to_excel(writer, sheet_name='Hotfix_Results', index=False)
else:
starting_df = pd.read_excel(file_path)
already_ran = set(starting_df['Contract Name'].unique())
# Read clean data # Read clean data
print("Reading clean data...") print("Reading clean data...")
if '.xlsx' in ABC_PATH: if '.xlsx' in ABC_PATH:
abc_clean = pd.read_excel(ABC_PATH) abc_clean = pd.read_excel(ABC_PATH)
else: elif '.csv' in ABC_PATH:
abc_clean = pd.read_csv(ABC_PATH) abc_clean = pd.read_csv(ABC_PATH)
elif '.xlsb' in ABC_PATH:
abc_clean = utils.read_xlsb(ABC_PATH)
# print(list(abc_clean.columns))
unique_output_files = abc_clean['Contract Name'].unique() unique_output_files = abc_clean['Contract Name'].unique()
print("ABC Clean: ", abc_clean.shape) print("ABC Clean: ", abc_clean.shape)
print(len(unique_output_files), 'files in clean output...') print(len(unique_output_files), 'files in clean output...')
# print(list(abc_clean.columns)) # print(list(abc_clean.columns))
abc_clean.rename(columns=valid.HOTFIX_MAPPING, inplace=True) abc_clean.rename(columns=valid.HOTFIX_MAPPING, inplace=True)
abc_clean['Contract Name'] = 'Filename: ' + abc_clean['Contract Name'] abc_clean['Contract Name'] = abc_clean['Contract Name'].apply(lambda x: ('Filename: ' + str(x) if not str(x).startswith('Filename: ') else str(x)) + ('.txt' if not str(x).endswith('.txt') else ''))
print(list(abc_clean.columns)) print(list(abc_clean.columns))
# Read input dict # Read input dict
print("Reading input .txt files...") print("Reading input .txt files...")
input_dict = utils.read_input() input_dict = utils.read_input()
input_dict = {'Filename: ' + filename.strip() : contract_text for filename, contract_text in input_dict.items()} input_dict = {'Filename: ' + filename.strip() : contract_text for filename, contract_text in input_dict.items()}
if starting_df:
input_dict = {filename : contract_text for filename, contract_text in input_dict.items() if filename not in already_ran}
print("Input Dict: ", len(input_dict)) print("Input Dict: ", len(input_dict))
input_keys = list(input_dict.keys()) # input_keys = list(input_dict.keys())
abc_clean['Contract Name'] = abc_clean['Contract Name'].apply( # abc_clean['Contract Name'] = abc_clean['Contract Name'].apply(
lambda x: get_highest_similarity(x, input_keys) # lambda x: get_highest_similarity(x, input_keys)
) # )
unique_output_files = abc_clean['Contract Name'].unique()
unique_output_files = abc_clean['Contract Name'].unique()
print(unique_output_files[0:2])
print(list(input_dict.keys())[0:2])
print(f"{len([file for file in input_dict.keys() if file not in unique_output_files])} files in full s3, not in output") print(f"{len([file for file in input_dict.keys() if file not in unique_output_files])} files in full s3, not in output")
print(f"{len([file for file in unique_output_files if file not in input_dict.keys()])} files in output, not in full s3") print(f"{len([file for file in unique_output_files if file not in input_dict.keys()])} files in output, not in full s3")
@@ -131,19 +150,27 @@ def main():
print(f"{len(abc_clean_list)} - files in input") print(f"{len(abc_clean_list)} - files in input")
print(f"{len([f for f in abc_clean_list if f[1]])} - valid files in input") print(f"{len([f for f in abc_clean_list if f[1]])} - valid files in input")
df_list = []
print("Processing hotfixes...")
for abc_df, contract_text in abc_clean_list:
new_df = process_hotfix(abc_df, contract_text)
df_list.append(new_df)
abc_final = pd.concat(df_list, ignore_index=True)
print(abc_final.shape) def process_pair(abc_df, contract_text):
print(list(abc_final.columns)) return process_hotfix(abc_df, contract_text)
with concurrent.futures.ThreadPoolExecutor(max_workers=config.MAX_WORKERS) as executor:
print("Writing to excel...") futures = [executor.submit(process_pair, abc_df, contract_text) for abc_df, contract_text in abc_clean_list]
abc_final.to_excel(f'{config.BATCH_ID}-Hotfix-Final.xlsx') for future in concurrent.futures.as_completed(futures):
try:
result = future.result()
if result is not None:
# Standardize columns by reindexing with all known columns, filling missing columns with NaN
standardized_result = result.reindex(columns=cnc_hotfix.HOTFIX_ORDER, fill_value=pd.NA)
# Append the standardized DataFrame to the Excel file on the same sheet
with pd.ExcelWriter(file_path, engine='openpyxl', mode='a', if_sheet_exists='overlay') as writer:
# Determine the start row for new data
start_row = writer.sheets['Hotfix_Results'].max_row if 'Hotfix_Results' in writer.sheets else 0
standardized_result.to_excel(writer, sheet_name='Hotfix_Results', startrow=start_row, header=start_row == 0, index=False)
except Exception as e:
# Log the exception or handle it otherwise
print(f"An error occurred: {e}")
print("Completed writing to Excel.")
if __name__ == "__main__": if __name__ == "__main__":
+179 -29
View File
@@ -131,33 +131,34 @@ def populate_default_term(df, text_dict):
extracted_text = '' extracted_text = ''
for page in range(int(round(start_page)), int(round(end_page))): for page in range(int(round(start_page)), int(round(end_page))):
str_page = f'{page:.0f}'
exhibit_text = text_dict[f'{page:.0f}'] if str_page in text_dict.keys():
exhibit_text = text_dict[str_page]
def_rate = file_df.loc[file_df['Page_Num'] == start_page, 'Default Rate Corrected Step 2'] def_rate = file_df.loc[file_df['Page_Num'] == start_page, 'Default Rate Corrected Step 2']
if isinstance(def_rate, pd.Series): if isinstance(def_rate, pd.Series):
def_rate = def_rate.min() def_rate = def_rate.min()
elif not isinstance(def_rate, str): elif not isinstance(def_rate, str):
def_rate = str(def_rate) def_rate = str(def_rate)
sentences = exhibit_text.split('.')
pct_pattern = r'\d+\%'
sentences = exhibit_text.split('.') reimb_pattern = r'(medicare|allowed charge|allowable charge|billed charge|medicaid|average wholesale price|fee schedule)'
pct_pattern = r'\d+\%' if (def_rate == def_rate) and (def_rate != '') and (re.search(pct_pattern, def_rate)) and (re.search(reimb_pattern, def_rate.lower())):
def_pct = re.search(pct_pattern, def_rate).group(0)
reimb_pattern = r'(medicare|allowed charge|allowable charge|billed charge|medicaid|average wholesale price|fee schedule)' def_reimb = re.search(reimb_pattern, def_rate.lower()).group(0)
for sentence in sentences:
if (def_rate == def_rate) and (def_rate != '') and (re.search(pct_pattern, def_rate)) and (re.search(reimb_pattern, def_rate.lower())): sentence = sentence.replace('\n', ' ')
def_pct = re.search(pct_pattern, def_rate).group(0) if (def_pct in sentence) and (def_reimb in sentence.lower()):
def_reimb = re.search(reimb_pattern, def_rate.lower()).group(0) extracted_text = sentence
for sentence in sentences: break
sentence = sentence.replace('\n', ' ')
if (def_pct in sentence) and (def_reimb in sentence.lower()): if extracted_text != '':
extracted_text = sentence
break break
if extracted_text != '':
break
file_df.loc[file_df['Attachment/Exhibit'] == exhibit, 'Default Term Corrected Step 3'] = extracted_text.strip() file_df.loc[file_df['Attachment/Exhibit'] == exhibit, 'Default Term Corrected Step 3'] = extracted_text.strip()
file_df = file_df[['Attachment/Exhibit', 'Default Term Corrected Step 3']] file_df = file_df[['Attachment/Exhibit', 'Default Term Corrected Step 3']]
@@ -548,9 +549,10 @@ def clean_contract_effective_date(
answer,d,page_list,claude_answer_raw = "N/A",{},[],"" answer,d,page_list,claude_answer_raw = "N/A",{},[],""
else: else:
answer = contract_effective_date answer = contract_effective_date
d,page_list,claude_answer_raw = {},[],""
try: try:
corrected_date = utils.convert_to_us_date_format(answer) # apply date formatting fix corrected_date = utils.convert_to_us_date_format(str(answer)) # apply date formatting fix
except utils.InvalidDateException as e: except utils.InvalidDateException as e:
corrected_date = "N/A" corrected_date = "N/A"
# print(f"Effective Date for {file_name} --> {corrected_date}") # print(f"Effective Date for {file_name} --> {corrected_date}")
@@ -622,8 +624,11 @@ def clean_term_group(df: pd.DataFrame,
field_group_answer_dict = ac_funcs.json_parsing_search(field_group_answer_raw, fields) field_group_answer_dict = ac_funcs.json_parsing_search(field_group_answer_raw, fields)
for field in fields: for field in fields:
field_answer = field_group_answer_dict[field] if field in field_group_answer_dict:
ac_answers_dict[field] = field_answer field_answer = field_group_answer_dict[field]
ac_answers_dict[field] = field_answer
else:
ac_answers_dict[field] = field_answer
ac_df = pd.DataFrame([ac_answers_dict]) ac_df = pd.DataFrame([ac_answers_dict])
ac_df = postprocessing_funcs.clean_term_clause(ac_df) ac_df = postprocessing_funcs.clean_term_clause(ac_df)
@@ -647,8 +652,8 @@ def clean_agreement_name(df: pd.DataFrame,
If agreement name is not found, it runs on the context selected based on keywords. If agreement name is not found, it runs on the context selected based on keywords.
""" """
df["Agreement Name (Contract Title)_corrected"] = df['Agreement Name (Contract Title)'] df["Agreement_Name (Contract Title)_corrected"] = df['Agreement_Name (Contract Title)']
agreement_df = df[(df['Agreement Name (Contract Title)'].isna())|(df['Agreement Name (Contract Title)'].astype(str).str.len() < 5)] agreement_df = df[(df['Agreement_Name (Contract Title)'].isna())|(df['Agreement_Name (Contract Title)'].astype(str).str.len() < 5)]
agreement_df['Filename'] = agreement_df['Contract Name'] agreement_df['Filename'] = agreement_df['Contract Name']
contract_list = agreement_df['Filename'].to_list() contract_list = agreement_df['Filename'].to_list()
contract_list = list(set(contract_list)) contract_list = list(set(contract_list))
@@ -678,8 +683,153 @@ def clean_agreement_name(df: pd.DataFrame,
prompt_answer_raw = claude_funcs.invoke_claude( prompt_answer_raw = claude_funcs.invoke_claude(
prompt, config.MODEL_ID_CLAUDE35_SONNET, contract_name, 8192) prompt, config.MODEL_ID_CLAUDE35_SONNET, contract_name, 8192)
df.loc[df['Contract Name'] == contract_name, 'Agreement Name (Contract Title)_corrected'] = prompt_answer_raw df.loc[df['Contract Name'] == contract_name, 'Agreement_Name (Contract Title)_corrected'] = prompt_answer_raw
return df return df
HOTFIX_ORDER = [
"Contract Name",
"Agreement_Name (Contract Title)",
"PAYER NAME",
"Health Plan State",
"Affiliate (Y/N)",
"Credentialing Application Indicator",
"Term Clause",
"Contract Auto-Renewal Indicator",
"Termination Date",
"Termination Upon Notice - Days",
"Termination With Cause - Days",
"Non-Renewal Language",
"Non-Renewal - Days",
"Amend Contract Upon Notice Flag (Y/N)",
"Timeframe to Object - Days",
"Assignments Clause (Y/N)",
"Contract Effective Date",
"IRS #",
"IRS Name",
"NPI (10-digits)",
"NPI Name",
"PROV_GROUP_TIN_SIGNATORY",
"PROV_TIN_OTHER",
"PROV_NPI_OTHER",
"Notice to Provider Name",
"Notice to Provider Address",
"Sequestration Language",
"Sequestration Reductions (Y/N)",
"Parent Agreement Code",
"Pages",
"Page_Num",
"Attachment/Exhibit",
"Line of Business",
"Provider Type",
"Provider Type - Level 2",
"IP/OP",
"Service Type",
"Plan Type",
"Lesser of Logic language, included (Y/N)",
"Lesser of Rate",
"Reimb. Methodology",
"Reimb. Methodology_Short",
"If rate is % of Payor or MCR [STANDARD]",
"If Rate is % of Payor or MCR [Standard] Short",
"FLAT FEE",
"Default Term",
"Default Rate",
'Inclusion of Essential RBRVS "Fee Source" Language (Y/N)',
"CDM Neutralization Language, Included (Y/N)",
"CONTRACT_CHARGEMASTER_PROTECTION_LANGUAGE",
"IP - DSH/IME/UC, Included (Y/N)",
"IP - Stoploss Catastrophic Threshold",
"Exclusions",
"Not to Exceed",
"Escalator or COLA (Y/N)",
"Escalator I, Eff. Date",
"Delegated Function Indicator",
"Delegated Terms",
"ECM",
"National Agreement Indicator",
"Cost Settlement (Y/N)",
"Cost Settlement (Language)",
"Late Paid Claims (Y/N)",
"Late Paid Claims (Language)",
"Deemer Amendment",
"Regulatory Requirements",
"Recovery Rights",
"Arbitration and Disputes",
"Exclusivity Requirement (Y/N)",
"Exclusivity Requirement (Language)",
"Payor",
"Participation in Products",
"Clean Claim",
"Independent Review (Y/N)",
"Independent Review (Language)",
"Indemnification",
"Access to Medical Records",
"Member Confinement Days Language (Y/N)",
"Member Confinement Days Language (Language)",
"Network Access Fees (Y/N)",
"Network Access Fees (Language)",
"Payment in Advance of Claims Submission Language (Y/N)",
"Payment in Advance of Claims Submission Language",
"Eligibility Verification",
"Preauthorization",
"Policies and Procedures",
"Insurance Requirement",
"Carve-Out Vendors",
"Conflicts Between Certain Documents (Y/N)",
"Conflicts Between Certain Documents (Language)",
"Relationship of Parties (Y/N)",
"Relationship of Parties (Language)",
"Nonstandard Appeals Process (Y/N)",
"Nonstandard Appeals Process (Language)",
"Product Removal",
"Disparagement Prohibition (Y/N)",
"Disparagement Prohibition (Language)",
"Claims Editing Language (Y/N)",
"Claims Editing Language (Language)",
"Guarantee of Provider Yield (Y/N)",
"Guarantee of Provider Yield (Language)",
"HCBS Services",
"Add On Reimbursement (Y/N)",
"Add On Reimbursement (Language)",
"PMPM",
"Single Code Multiple Rates (Y/N)",
"Single Code Multiple Rates (Language)",
"Invoice Pricing (Y/N)",
"Invoice Pricing (Language)",
"Medical Necessity Language (Y/N)",
"Medical Necessity Language (Language)",
"Template",
"Provider-Based Billing Exclusion (Y/N)",
"Provider-Based Billing Exclusion (Language)",
"Agreement Name (Contract Title)_corrected",
"Health Plan State_corrected",
"Term Clause_corrected",
"Contract Auto-Renewal Indicator_corrected",
"Termination Date_corrected",
"Contract Effective Date_corrected",
"IRS_corrected",
"PROV_TIN_OTHER_corrected",
"PROV_TIN_GROUP_SIGNATORY_corrected",
"NPI_corrected",
"NPI_other_corrected",
"LOB_Correct",
"PROV_TYPE_LEVEL_2_Corrected",
"Lesser_Flag",
"Service-Methodology",
"Default Term Corrected",
"Default Rate Corrected",
"FULL_METHODOLOGY_fixed",
"LESSER_fixed",
"RATE_STANDARD_fixed",
"RATE_SHORT_fixed",
"FLAT_FEE_STANDARD_fixed",
"LESSER_RATE_fixed",
"NOT_TO_EXCEED_fixed",
"SHORT_METHODOLOGY_fixed",
"_merge",
"key1_fixed",
"key2_fixed",
"key3_fixed"]
+12 -1
View File
@@ -9,6 +9,7 @@ from collections import defaultdict
import config import config
from botocore.exceptions import ClientError from botocore.exceptions import ClientError
from io import StringIO from io import StringIO
from pyxlsb import open_workbook
class InvalidDateException(Exception): class InvalidDateException(Exception):
@@ -454,4 +455,14 @@ def remove_unnamed_columns(df):
def remove_txt_extension(filename): def remove_txt_extension(filename):
if isinstance(filename, str) and filename.lower().endswith(".txt"): if isinstance(filename, str) and filename.lower().endswith(".txt"):
return filename[:-4] return filename[:-4]
return filename return filename
def read_xlsb(path):
with open_workbook(path) as wb:
with wb.get_sheet(1) as sheet:
rows = []
for row in sheet.rows():
rows.append([item.v for item in row]) # Extracting cell values
df = pd.DataFrame(rows[1:], columns=rows[0]) # Converting to DataFrame, assuming first row is header
return df