Merged in cnc_fl_main (pull request #258)

Cnc fl main

* prov type moved

* comment out invalid prov2 bc it's overly restrictive

* Update prompt template to ensure N/A rather than blanks

* Update AC multi-prompt template. Update prompts to ensure all keys are present and N/As are filled

* Update AC postprocessing

* Update clean_tin() to add N/A for values other than 10 digits

* update tin postprocessing

* Update npi postprocessing

* Add 'Filename:' to the start of all filenames

* Update group TIN/NPI postprocessing

* Update parent agreement code

* fix health plan state mapping

* Update B postprocessing

* Updated indicators, re-ordering

* Update postprocessing - all columns included and reordered

* Ensure Pages is included in AC-only output

* Remove excess print statements

* remove example_test to pass pipeline

* Fix parent agreement code for AC

* Updated postprocessing for NPI/TIN other

* Update prompts for N explicitely on AC prompt-based indicators

* Updated main and utils for more intuitive filtering

* Updated file_counting to match new main process

* update test

* Add example_test


Approved-by: Michael McGuinness
This commit is contained in:
Katon Minhas
2024-11-04 22:54:31 +00:00
parent de41567da0
commit bb279b45fd
16 changed files with 935 additions and 381 deletions
+6 -1
View File
@@ -71,12 +71,17 @@ streamlit/results.csv
# env
streamlit/venv
*.pdf
*.PDF
*.TXT
*.txt
*.tfplan
*.pem
textfiles/
texts/
*.csv
*.xlsx
*.zip
*.json
myenv/
*.env
+1 -1
View File
@@ -35,7 +35,7 @@ def run_conditional(combined_results, text_dict, filename):
rbrvs_answer = claude_funcs.invoke_claude(
prompt, config.MODEL_ID_CLAUDE2, filename, max_tokens=16
)
# print(rbrvs_answer)
d['Inclusion of essential RBRVS "Fee Source" Language (Y/N)'] = (
rbrvs_answer.strip()
)
+3 -3
View File
@@ -110,9 +110,9 @@ TABLE_ANALYSIS_NAME = f"{BATCH_ID}-Table-Analysis.csv"
# Bedrock
if RUN_MODE == "local":
AWS_ACCESS_KEY_ID = "ASIA6GBMBVWOKQQI6OCB"
AWS_SECRET_ACCESS_KEY = "hUt0Y7BbSx31giz6cSD7pblxiHq7IsIaLupccqok"
AWS_SESSION_TOKEN = "IQoJb3JpZ2luX2VjEBcaCXVzLWVhc3QtMiJHMEUCIQD0FlMPo1DTqKa39N0aoMtu4jzT9J5BJaIT96mMXCi/3QIgA3vDTUNHDLvTFZ5KUpqC38GHHt5kP785CbKgD/q7SNYqjgMIkP//////////ARAAGgw5NzUwNDk5NjA4NjAiDDW0B0VQj9y1pYY7uCriAiNJcvJ6LSyPmJdkFai72ait8YWjQPUFgvEe6Em92rIQ2BQu5jyWFQjCRCfcp7lU/PFYFsDG1JKla/octKPVbAu2/RllGp76Ii4G+K/hWyNlstPVoM3mPXRlucUnZq2GwtEd3qCLTYCci2NPEqTfmizvc2Ragpi97pL7Q1sYPBOZ92K8IU3ff9iSduMsu+9w6jVuYQiLa+Qn5hZeLsyjBg7ya07x8WdwsaMCre/TYocajjnT1eDDsQ/eu7kcxmyivlhAvTFUqzJw/uGzv28jSrzTLMbU4/LOWeYEfh365BZaSe8BPpCexGyiKwYRUZd+nLhSS3mfSHsYg+a9Y2XoXOpb3lToA0QFcNKauvwyLkwLdlbusvm6beaGS9dhmV3YTuHqYSwyQJArnpPRv1qPPTuMb8DEl4uyLj15BMN1s5+gl+driZUK2ElxcnFNB+GXTaQPMBJbv+ahmNvcbuUUREVfZTCKrI65BjqmAUxxjX0uHTuugmUKmB5/NfYGQRrwS+/gthlbI1DP1Tm5qP86QXyvyC+Yi3erASb7I30lt47qA7cYzHBo/GOnCkx093qZwZTIcyMDnH7JJ7K6K1f7QqfKpME6Bn9DhRJiiYANvVI5ruU6wMlSu257HKamVH61ZjxjtKy9z7OjsKeYTTSFOvyw73x6BeEdz4vTiZOu/W51fydhE1mChe+nocVhApm9O9A="
AWS_ACCESS_KEY_ID="ASIA6GBMBVWONKTNLEQK"
AWS_SECRET_ACCESS_KEY="OJ/mV2gfHAXHA62dNpRP7HBF0AmwPhTSx9Lx7GKW"
AWS_SESSION_TOKEN="IQoJb3JpZ2luX2VjEOv//////////wEaCXVzLWVhc3QtMiJGMEQCIB52FvtJu1yJRF4m7p4Gl9APRpI4xCx8h/y+yshFMXbnAiAPktTJSCX0XF2bXAVQUftZD3VGXsZqlOedaKMNNNXHASqJAwhkEAAaDDk3NTA0OTk2MDg2MCIMI11ogWEEpCUL9KjOKuYCIyD27hNGF+jBMlWnEEn34k12WZd0/ZvYyGUqg7EYGj9DiznQ1D90+ynSHQjTiqEuTbnmhM58pHnhn5rSiB7LVVZldKHYZm2s9LXyw5lnPyytR7xI0JgiExeJGBePd12YtqCf3iDDh81hT5IPcW3kQCZ7IM/LQC6kBlT6j+dvD8DeRRYG/uH1ngkLNHj/1BUVYZohA5mwkUmle21IeVoLIArtgx2k5wWNV0QT2GM3tuDeCCCGqmZOomau5ffMJIWm+fgIJam9ckUl2IFyX0Vn/AX906tz3FZJIUdzxFcX+8wn1/yfvT1jpXXWSFekS7d/BaM6zRwAcRA2FMK4iiBITAmZm1FnxWDC0b4atdbst7KvYtuIUJODDeh7bzAotFAe6faZvIwnAYxkadMJ8rMpdamldVb+/HfwX7gYbyAs9qHjAbam2L+ZWVycMzDL68TjL2RF/gRO+Ts4uTzcihyis1bGxS6wDDDX14S5BjqnAdIuu67CAQs0ynGkmI+7Q39o6F2lg1/l0638QhjY7VugNcqhxNN2zdu8Njc0kZLwa72QsekjhUZ59cBRzV7z1zY4UdT6eGQJbR1ovNd0PsxV/BQ+PH60G+aKJuYAx1rpdz81r4cX3tal5gzktYNN9XujCA4Dts9inRl2pka6BrmuH+GyO5K9Ba9Kjrom3+bVz0QdGEsEvnQWt+AA5NMEUEXCblO9SiZc"
config = Config(read_timeout=2000)
BEDROCK_RUNTIME = boto3.client(
service_name="bedrock-runtime",
@@ -22,6 +22,7 @@ def consolidate_output():
except:
pass
ac_final_df = pd.concat(ac_dfs, ignore_index=True)
ac_final_df.to_csv(
os.path.join(
config.CONSOLIDATED_OUTPUT_DIRECTORY, f"{config.BATCH_ID}-AC.csv"
@@ -40,6 +41,7 @@ def consolidate_output():
except:
pass
b_final_df = pd.concat(b_dfs, ignore_index=True)
b_final_df.to_csv(
os.path.join(
config.CONSOLIDATED_OUTPUT_DIRECTORY, f"{config.BATCH_ID}-B.csv"
@@ -48,6 +50,8 @@ def consolidate_output():
# If ABC
if "a" in config.FIELDS and "c" in config.FIELDS and "b" in config.FIELDS:
b_final_df.drop(['Pages', 'Parent Agreement Code'], axis=1, inplace=True) # Drop Pages so only one in result
abc_path = os.path.join(
config.CONSOLIDATED_OUTPUT_DIRECTORY, f"{config.BATCH_ID}-ABC.csv"
)
@@ -57,6 +61,8 @@ def consolidate_output():
abc_final_df = abc_final_df[
[col for col in valid.ABC_COLUMNS if col in abc_final_df]
]
print("Columns not found: ", [f for f in valid.ABC_COLUMNS if f not in abc_final_df.columns])
abc_final_df.to_csv(abc_path)
print(f"Consolidation complete. Results written to {abc_path}")
+14 -9
View File
@@ -7,20 +7,25 @@ print(f"Input Dir: {config.LOCAL_PATH}")
print(f"S3 Bucket: {config.S3_BUCKET}")
print(f"S3 Prefix: {config.S3_PREFIX}")
input_dict = utils.read_input()
input_dict = (
utils.read_input()
) # keys are contract names, values are full contract text
total_files = len(input_dict)
print(f"Input Files : {len(input_dict)}")
print(f"Total Input Files : {len(input_dict)}")
# Filter no reimbursement
input_dict = {
k: v for k, v in input_dict.items() if utils.contains_reimbursement(str(v))
}
# Filter B
input_dict_b = {
k: v
for k, v in input_dict.items()
if utils.contains_reimbursement(str(v))
} # Filter out non-contracts
print(
f"B Input Files after reimbursement filter: {len(input_dict)} | {total_files-len(input_dict)} files removed"
f"B Input Files after reimbursement filter: {len(input_dict_b)} | {total_files-len(input_dict_b)} files removed"
)
# Filter already processed
if config.FILTER_ALREADY_PROCESSED:
input_dict = utils.filter_already_processed(input_dict)
print(f"Input Files left to be processed : {len(input_dict)}")
input_dict_ac, input_dict_b = utils.filter_already_processed(input_dict, input_dict_b)
print(f"AC Input Files left to be processed : {len(input_dict_ac)}")
print(f"B Input Files left to be processed : {len(input_dict_b)}")
+8 -89
View File
@@ -21,22 +21,11 @@ import claude_funcs
def run_ac_prompts(file_object):
# Batch 4 Only
chunked_and_full_fields = [
"RELATIONSHIP_OF_PARTIES_LANGUAGE",
"PROV_GROUP_TIN",
"EXCLUSIVITY_REQUIREMENT_LANGUAGE",
"CONTRACT_EFFECTIVE_DT",
"TERMINATION_UPON_NOTICE",
"NOTICE_PROVIDER_NAME",
"NOTICE_PROVIDER_ADDRESS",
]
################## INITIATE PROCESSING ##################
filename, contract_text = file_object
print(f"Processing AC for {filename}...")
print(f"Total contract word count: {len(contract_text.split())}")
################## PREPROCESS ##################
text_dict, exhibit_pages, num_pages, ac_chunks = preprocess.preprocess(
contract_text, filename, fields="ac"
@@ -45,25 +34,18 @@ def run_ac_prompts(file_object):
################## RUN FULL CONTEXT PROMPTS ##################
ac_dict = {}
log_data = []
# Process 'full_context' fields together
full_context_fields = [
field for field in prompts.AC_DICT if field not in keywords.KEYWORD_MAPPINGS
]
full_context_fields += chunked_and_full_fields # 4 Only
full_context_questions = {
field: prompts.AC_DICT.get(field) for field in full_context_fields
}
if full_context_questions:
full_context_prompt = ac_funcs.create_prompt(
contract_text[0 : min(100000, len(contract_text) - 1)],
question=full_context_questions,
)
# print(f"High accuracy prompt word count: {len(high_accuracy_prompt.split())}")
full_context_prompt = prompts.AC_MULTI_FIELD_TEMPLATE(contract_text[0 : min(100000, len(contract_text) - 1)], full_context_questions)
try:
full_context_answers = ac_funcs.get_ac_answer(
full_context_prompt,
@@ -72,32 +54,8 @@ def run_ac_prompts(file_object):
)
ac_dict.update(full_context_answers)
for field, answer in full_context_answers.items():
log_data.append(
{
"Field": field,
"Prompt": f"Question: {full_context_questions[field]}\n\nContext: {contract_text}", # Include full context
"Context Type": "Full Context (High Accuracy)",
"Response": answer,
}
)
except Exception as e:
print(f"Error processing high accuracy fields: {str(e)}")
for field in full_context_questions.keys():
ac_dict[field] = f"Error: {str(e)}"
log_data.append(
{
"Field": field,
"Prompt": f"Question: {full_context_questions[field]}\n\nContext: {contract_text}", # Include full context
"Context Type": "Full Context (High Accuracy)",
"Response": f"Error: {str(e)}",
}
)
# Add suffix (4 Only)
for field in list(ac_dict.keys()).copy():
if field in chunked_and_full_fields:
ac_dict[field + "_FULL"] = ac_dict.pop(field)
print(f'Exception in AC Full Context : {e}')
################## RUN CHUNKED PROMPTS ##################
all_fields = set(prompts.AC_DICT.keys())
@@ -125,35 +83,13 @@ def run_ac_prompts(file_object):
prompt = prompts.AC_SINGLE_FIELD_TEMPLATE(context, question)
print(
f"Field {field} prompt word count: {len(prompt.split())} ({context_type})"
)
try:
field_answer = claude_funcs.invoke_claude(
prompt, config.MODEL_ID_CLAUDE35_SONNET, filename, 8192
)
ac_dict[field] = field_answer
log_data.append(
{
"Field": field,
"Prompt": f"Question: {question}\n\nContext: {context}", # Include full context
"Context Type": context_type,
"Response": field_answer,
}
)
except Exception as e:
print(f"Error processing field {field}: {str(e)}")
ac_dict[field] = f"Error: {str(e)}"
log_data.append(
{
"Field": field,
"Prompt": f"Question: {question}\n\nContext: {context}", # Include full context
"Context Type": context_type,
"Response": f"Error: {str(e)}",
}
)
################## AC CONDITIONAL PROMPTS ##################
# Non-Renewal - Days
@@ -184,15 +120,6 @@ def run_ac_prompts(file_object):
for field in all_fields:
if field not in ac_dict:
print(f"Field not found in results. {field}")
# ac_dict[field] = "N/A"
log_data.append(
{
"Field": field,
"Prompt": "N/A",
"Context Type": "N/A",
"Response": "N/A",
}
)
################## CREATE OUTPUT DIRECTORIES ##################
base_filename = os.path.splitext(filename)[0].strip()
@@ -200,23 +127,15 @@ def run_ac_prompts(file_object):
os.makedirs(output_dir, exist_ok=True)
################## POSTPROCESS ##################
ac_dict["Contract Name"] = filename
for key in ac_dict.keys():
print(key, ac_dict[key])
ac_df = pd.DataFrame([ac_dict])
ac_df = postprocess.ac_postprocess(ac_df, num_pages)
ac_df = postprocess.ac_postprocess(ac_df, filename, num_pages)
################## WRITE TO OUTPUT ##################
ac_df.to_csv(os.path.join(output_dir, config.AC_RESULTS_NAME), index=False)
# ################## WRITE LOG TO CSV ##################
# log_file_path = os.path.join(output_dir, f"logs/{base_filename}_prompt_log.csv")
# with open(log_file_path, 'w', newline='', encoding='utf-8') as log_file:
# fieldnames = ['Field', 'Prompt', 'Context Type', 'Response']
# writer = csv.DictWriter(log_file, fieldnames=fieldnames)
# writer.writeheader()
# writer.writerows(log_data)
# print(f"Prompt log written to {log_file_path}")
return ac_dict
+53 -62
View File
@@ -46,77 +46,68 @@ def process_ac(item):
def main():
if config.TEST:
print(
claude_funcs.invoke_claude(
"Write 'test', nothing more.",
model_id=config.MODEL_ID_CLAUDE2,
filename="test",
max_tokens=10,
)
)
else:
input_dict = (
utils.read_input()
) # keys are contract names, values are full contract text
total_files = len(input_dict)
input_dict = (
utils.read_input()
) # keys are contract names, values are full contract text
total_files = len(input_dict)
print(f"Input Files : {len(input_dict)}")
print(f"Total Input Files : {len(input_dict)}")
# Filter already processed
if config.FILTER_ALREADY_PROCESSED:
input_dict = utils.filter_already_processed(input_dict)
print(f"Input Files left to be processed : {len(input_dict)}")
# Filter B
input_dict_b = {
k: v
for k, v in input_dict.items()
if utils.contains_reimbursement(str(v))
} # Filter out non-contracts
print(
f"B Input Files after reimbursement filter: {len(input_dict_b)} | {total_files-len(input_dict_b)} files removed"
)
# Filter no reimbursement
# Filter already processed
if config.FILTER_ALREADY_PROCESSED:
input_dict_ac, input_dict_b = utils.filter_already_processed(input_dict, input_dict_b)
print(f"AC Input Files left to be processed : {len(input_dict_ac)}")
print(f"B Input Files left to be processed : {len(input_dict_b)}")
batch_results = []
processed_count = 0
with concurrent.futures.ThreadPoolExecutor(
max_workers=config.MAX_WORKERS
) as executor:
if "b" in config.FIELDS:
input_dict_b = {
k: v
for k, v in input_dict.items()
if utils.contains_reimbursement(str(v))
} # Filter out non-contracts
print(
f"Input Files after reimbursement filter: {len(input_dict_b)} | {total_files-len(input_dict_b)} files removed"
)
futures = [
executor.submit(process_b, item) for item in input_dict_b.items()
]
if "a" in config.FIELDS and "c" in config.FIELDS:
futures = [
executor.submit(process_ac, item) for item in input_dict_ac.items()
]
batch_results = []
processed_count = 0
for future in concurrent.futures.as_completed(futures):
try:
result = future.result()
if result:
batch_results.append(result)
processed_count += 1
with concurrent.futures.ThreadPoolExecutor(
max_workers=config.MAX_WORKERS
) as executor:
if "b" in config.FIELDS:
futures = [
executor.submit(process_b, item) for item in input_dict_b.items()
]
if "a" in config.FIELDS and "c" in config.FIELDS:
futures = [
executor.submit(process_ac, item) for item in input_dict.items()
]
if processed_count % 5 == 0:
tracking.write_batch_results(batch_results)
batch_results = []
except Exception as e:
print(f"Error in future: {e}")
traceback.print_exc()
for future in concurrent.futures.as_completed(futures):
try:
result = future.result()
if result:
batch_results.append(result)
processed_count += 1
if batch_results:
tracking.write_batch_results(batch_results)
if processed_count % 5 == 0:
tracking.write_batch_results(batch_results)
batch_results = []
except Exception as e:
print(f"Error in future: {e}")
traceback.print_exc()
tracking.write_stats_to_csv()
print("\nIndividual Processing Complete")
if batch_results:
tracking.write_batch_results(batch_results)
tracking.write_stats_to_csv()
print("\nIndividual Processing Complete")
print("Consolidation starting...")
consolidate_output.consolidate_output()
print("Consolidation complete...")
print("Consolidation starting...")
consolidate_output.consolidate_output()
print("Consolidation complete...")
if __name__ == "__main__":
+50 -8
View File
@@ -12,9 +12,12 @@ import utils
def b_postprocess(filename, df, pages):
if df.shape[0] > 0:
# Metadata fields
df["Contract Name"] = filename
df.drop('Filename', axis=1, inplace=True)
df["Contract Name"] = 'Filename: ' + filename
df["Parent Agreement Code"] = postprocessing_funcs.get_parent_agreement_code(
filename
)
@@ -41,28 +44,67 @@ def b_postprocess(filename, df, pages):
# Clean LOB
df = postprocessing_funcs.clean_lob(df, filename)
# Rename and reorder
# Rename and reorder
df.rename(columns=valid.B_MAPPING, inplace=True)
column_order = [col for col in valid.B_MAPPING.values() if col in df.columns]
final_df = df[column_order]
return final_df
else:
return df
def ac_postprocess(df, pages):
def ac_postprocess(df, filename, pages):
if df.shape[0] > 0:
if "b" not in config.FIELDS:
df["Pages"] = pages
df["Contract Name"] = 'Filename: ' + filename
df = postprocessing_funcs.clean_ac_fields(df)
df["Pages"] = pages
df["Parent Agreement Code"] = postprocessing_funcs.get_parent_agreement_code(
filename
)
df = df.fillna("")
df = postprocessing_funcs.clean_term_clause(df)
df = postprocessing_funcs.clean_auto_renewal_ind(df)
df = postprocessing_funcs.get_tin_from_filename(df)
df = postprocessing_funcs.clean_tin(df)
df = postprocessing_funcs.clean_npi(df)
df = postprocessing_funcs.clean_network_access_fees(df)
df = postprocessing_funcs.clean_health_plan_state(df)
df = postprocessing_funcs.clean_notice_provider_name_and_address(df)
df = postprocessing_funcs.clean_policies_and_procedures(df)
df = postprocessing_funcs.clean_tin_npi_other(df)
df = df.apply(lambda x: x.map(postprocessing_funcs.replace_null_terms))
df = df.apply(lambda x: x.map(postprocessing_funcs.replace_quotes))
# Derive Indicators
df = df.apply(postprocessing_funcs.derive_indicators, axis=1)
# Rename and Reorder
df = df.rename(columns=valid.AC_MAPPING)
column_order = [col for col in valid.ABC_COLUMNS if col in df.columns] + [
col for col in df.columns if col not in valid.ABC_COLUMNS
]
final_df = df[column_order]
return final_df
else:
return df
+91 -31
View File
@@ -135,10 +135,14 @@ def clean_td(td):
def get_parent_agreement_code(filename):
try:
filename = filename.split(".txt")[0]
filename = re.sub(r'\([^)]*\)', '', filename)
match = re.search(r"([^\sa-zA-Z]+)(?=\.\w+$|$)", filename)
end = [i for i in match.group(1).split("_") if i]
return end[0]
except:
except Exception as e:
print(f"Error: {e}")
return "N/A"
@@ -177,18 +181,28 @@ def clean_msr_lesser(df, ls=valid.VALID_MSR):
return df
def clean_default_term(df, ls=valid.INVALID_DEFAULT):
# Create a case-insensitive regex pattern that matches any of the values in ls
pattern = "|".join(re.escape(term) for term in ls)
# Find rows where the DEFAULT_TERM column contains any of the terms from ls
def clean_default_term(df):
# Remove any invalid Defaults
pattern = "|".join(re.escape(term) for term in valid.INVALID_DEFAULT)
df["DEFAULT_TERM"] = df["DEFAULT_TERM"].astype(str)
mask = df["DEFAULT_TERM"].str.contains(pattern, case=False, na=False)
# Replace these values with 'N/A'
df.loc[mask, "DEFAULT_TERM"] = "N/A"
df.loc[mask, "DEFAULT_RATE"] = "N/A"
# Replace RATE with acronyms (AC, BC, etc)
df['DEFAULT_RATE'] = df['DEFAULT_RATE'].replace(valid.RATE_REPLACEMENTS, regex=True)
# Ensure Outpatient, Inpatient, etc match up
def replace_terms(row):
if 'outpatient' in row['DEFAULT_TERM'].lower() and row['IP_OP'] != 'OP':
row['DEFAULT_TERM'] = 'N/A'
row['DEFAULT_RATE'] = 'N/A'
elif 'inpatient' in row['DEFAULT_TERM'].lower() and row['IP_OP'] != 'IP':
row['DEFAULT_TERM'] = 'N/A'
row['DEFAULT_RATE'] = 'N/A'
return row
df = df.apply(replace_terms, axis=1)
return df
@@ -274,17 +288,11 @@ def clean_prov_2(df):
logging.warning(f" FULL_SERVICE: {row['FULL_SERVICE']}")
logging.warning(f" EXHIBIT: {row['EXHIBIT']}")
df.loc[invalid_assignments.index, "PROV_TYPE_LEVEL_2"] = ""
# df.loc[invalid_assignments.index, "PROV_TYPE_LEVEL_2"] = ""
return df
def clean_ac_fields(final_df):
# additional post-processing
final_df = final_df.fillna("")
# final_df.loc[final_df['CREDENTIALING_APP_IND'] != 'Yes', 'CREDENTIALING_APP_IND'] = "No"
# final_df.loc[final_df['AFFILIATION_CLAUSE_IND'] != 'Yes', 'AFFILIATION_CLAUSE_IND'] = "No"
# final_df.loc[final_df['ASSIGNMENTS_CLAUSE_IND'] != 'Yes', 'ASSIGNMENTS_CLAUSE_IND'] = "No"
def clean_term_clause(final_df):
if "TERM_CLAUSE" in final_df:
final_df.loc[
final_df["TERM_CLAUSE"].str.startswith("IL-4 Termination"), "TERM_CLAUSE"
@@ -300,8 +308,10 @@ def clean_ac_fields(final_df):
final_df["TERM_CLAUSE"].str.contains("No term or termination"),
"TERM_CLAUSE",
] = "N/A"
# if 'CONTRACT_SIGNATORY_IND' in final_df and 'PROV_PARTICIPATION_STATUS' in final_df:
# final_df.loc[final_df['CONTRACT_SIGNATORY_IND'] == 'Yes', 'PROV_PARTICIPATION_STATUS'] = "Yes"
return final_df
def clean_auto_renewal_ind(final_df):
if (
"CONTRACT_AUTO_RENEWAL_IND" in final_df
and "CONTRACT_TERMINATION_DT" in final_df
@@ -309,30 +319,47 @@ def clean_ac_fields(final_df):
final_df.loc[
final_df["CONTRACT_AUTO_RENEWAL_IND"] == "Yes", "CONTRACT_TERMINATION_DT"
] = np.nan
return final_df
# check if NPI has 10 digits
if "PROV_GROUP_NPI" in final_df:
final_df["PROV_GROUP_NPI"] = final_df["PROV_GROUP_NPI"].map(
lambda x: x if sum(c.isdigit() for c in str(x) + " ") == 10 else ""
)
# NETWORK_ACCESS_FEES_IND
def clean_npi(final_df):
if "PROV_GROUP_NPI" in final_df:
# Replace strings with a digit count not equal to 10 with "N/A"
final_df['PROV_GROUP_NPI'] = final_df['PROV_GROUP_NPI'].apply(
lambda x: f'N/A - (model detected: {x})' if (len(''.join(filter(str.isdigit, x))) != 10 and x != 'N/A') else x)
return final_df
def clean_network_access_fees(final_df):
if "NETWORK_ACCESS_FEES_IND" in final_df:
final_df.loc[
~final_df["NETWORK_ACCESS_FEES_IND"].isin(["N/A", "No", "", " "]),
not is_empty(final_df["NETWORK_ACCESS_FEES_IND"]),
"NETWORK_ACCESS_FEES_IND",
] = "Yes"
return final_df
def clean_health_plan_state(final_df):
if "PAYER_NAME" in final_df and "HEALTH_PLAN_STATE" in final_df:
final_df.loc[
final_df["PAYER_NAME"].str.startswith("Illini"), "HEALTH_PLAN_STATE"
] = "Illinois"
if "HEALTH_PLAN_STATE" in final_df:
final_df["HEALTH_PLAN_STATE"] = final_df["HEALTH_PLAN_STATE"].str.upper()
final_df["HEALTH_PLAN_STATE"] = (
final_df["HEALTH_PLAN_STATE"]
.map(valid.STATE_MAP)
.fillna(final_df["HEALTH_PLAN_STATE"])
)
final_df["HEALTH_PLAN_STATE"] = final_df["HEALTH_PLAN_STATE"].str.title()
return final_df
def clean_notice_provider_name_and_address(final_df):
if "NOTICE_PROVIDER_NAME" in final_df and "NOTICE_PROVIDER_ADDRESS" in final_df:
final_df.loc[
final_df["NOTICE_PROVIDER_NAME"].str.contains(
@@ -346,7 +373,10 @@ def clean_ac_fields(final_df):
),
"NOTICE_PROVIDER_ADDRESS",
] = np.nan
return final_df
def get_tin_from_filename(final_df):
if "Contract Name" in final_df.columns:
final_df["temp_filename"] = (
final_df["Contract Name"].str[:10].str.replace("-", "")
@@ -358,6 +388,7 @@ def clean_ac_fields(final_df):
"PROV_GROUP_TIN",
] = final_df["temp_filename"]
final_df.drop(columns=["temp_filename"], inplace=True)
elif "Filename" in final_df.columns:
final_df["temp_filename"] = final_df["Filename"].str[:10].str.replace("-", "")
if "PROV_GROUP_TIN" in final_df and "Filename" in final_df:
@@ -367,8 +398,11 @@ def clean_ac_fields(final_df):
"PROV_GROUP_TIN",
] = final_df["temp_filename"]
final_df.drop(columns=["temp_filename"], inplace=True)
return final_df
# For POLICIES_AND_PROCEDURES, filter out anything without either "policies" or "procedures".
def clean_policies_and_procedures(final_df):
if "POLICIES_AND_PROCEDURES" in final_df:
final_df.loc[
~final_df["POLICIES_AND_PROCEDURES"].str.contains(
@@ -379,13 +413,42 @@ def clean_ac_fields(final_df):
),
"POLICIES_AND_PROCEDURES",
] = "N/A"
return final_df
final_df = final_df.apply(lambda x: x.map(replace_null_terms))
final_df = final_df.apply(lambda x: x.map(replace_quotes))
def clean_tin(final_df):
if "PROV_GROUP_TIN" in final_df:
# Replace SSN-like strings with "N/A"
final_df['PROV_GROUP_TIN'] = final_df['PROV_GROUP_TIN'].replace(
r'\b\d{3}-\d{2}-\d{4}\b', 'N/A', regex=True)
# Function to format and validate TIN numbers
def format_tin(x):
digits = ''.join(filter(str.isdigit, x))
if len(digits) == 9:
# Format as '##-#######'
return f'{digits[:2]}-{digits[2:]}'
else:
# Retain prior handling for invalid entries
if x != 'N/A':
return f'N/A - (model detected: {x})'
else:
return 'N/A'
# Apply the formatting function to the 'PROV_GROUP_TIN' column
final_df['PROV_GROUP_TIN'] = final_df['PROV_GROUP_TIN'].apply(format_tin)
return final_df
def clean_tin_npi_other(final_df):
if "PROV_TIN_OTHER" in final_df:
final_df["PROV_TIN_OTHER"] = final_df["PROV_TIN_OTHER"].str.replace("[", "", regex=False).str.replace("]", "", regex=False)
if "PROV_NPI_OTHER" in final_df:
final_df["PROV_NPI_OTHER"] = final_df["PROV_NPI_OTHER"].str.replace("[", "", regex=False).str.replace("]", "", regex=False)
return final_df
def replace_quotes(value):
try:
return str(value).replace(r"\"", '"')
@@ -490,13 +553,10 @@ def clean_lob(df, filename):
def derive_indicators(results):
for field in DERIVED_INDICATOR_FIELDS:
indicator_field = field + "_IND"
if field in results and results[field] and not is_empty(results[field]):
print(f"Y - {indicator_field}")
results[indicator_field] = "Y"
else:
print(f"N - {indicator_field}")
results[indicator_field] = "N"
return results
+9 -2
View File
@@ -2,7 +2,7 @@ import preprocessing_funcs
import top_down_funcs
import table_funcs
import config
import csv
def preprocess(contract_text, filename, fields=config.FIELDS):
@@ -12,7 +12,8 @@ def preprocess(contract_text, filename, fields=config.FIELDS):
text_dict = preprocessing_funcs.split_text(
contract_text
) # return a dictionary with keys - page_num (str), values as the page_text
#text_dict = {k: v for k, v in text_dict.items() if k.isdigit() and 42 <= int(k) <= 42}
num_pages = len(text_dict.keys())
text_dict = preprocessing_funcs.filter_quick_review(text_dict)
@@ -22,6 +23,12 @@ def preprocess(contract_text, filename, fields=config.FIELDS):
) # All pages with exhibit headers
text_dict = table_funcs.align_and_format_tables(text_dict, filename)
text_dict = preprocessing_funcs.chunk_consecutive(text_dict, exhibit_pages)
#write text_dict to a csv file:
with open("text_dict.csv", "w") as f:
writer = csv.writer(f)
for key, value in text_dict.items():
writer.writerow([key, value])
ac_chunks, full_context = "", ""
if fields == "ac":
exhibit_pages = []
+69 -76
View File
@@ -20,7 +20,10 @@ Return at least one json object for each combination of attributes seen. It is p
Here are the attributes to be included in each dictionary, and instructions on how to correctly answer:
'FULL_SERVICE' : What is the Service that is being reimbursed? Include the full Service, including any distinguishing context. Include any codes that are part of the service, including revenue codes and procedure codes. Use only exact text from the document. Do not leave this field N/A.
'SUBHEADER' : What subheader or subsection name does the FULL_SERVICE belong to? This may be include crucial additional detail about the Service. Note that this is not a primary header like that of a compensation schedule or exhibit. If there is no subheader, write 'N/A'.
'FULL_METHODOLOGY' : Write the full sentence or paragraph in the text describing the reimbursement methodology. Ensure any relevant detail is included, including rates in tables if applicable. Include any 'Lesser of' statement that applies to the reimbursement. The 'Lesser of' statement might not be found in immediate proximity to the reimbursement term and may instead be found in a paragraph above. If the methodology is presented in a table, concatenate any relevant lesser of statement that applies to the table with the portion of the methodology found in the table.
'FULL_METHODOLOGY' : Write the full sentence or paragraph in the text describing the reimbursement methodology.
'PROV_TYPE' : Write the Provider Type of the service. Choose ONLY from the following: {valid.VALID_PROV_TYPES}. Note that Facility may also be referred to as Hospital, Clinic, Institutional or similar. Professional may also be referred to as Physician, Physician Services, Practicioner, Provider or similar. Make sure to break out/differentiate between Professional, Facility, and Ancillary. Do not mix the two. Do NOT write 'N/A' for this field.
Ensure any relevant detail is included, including rates in tables if applicable. Include any 'Lesser of' statement that applies to the reimbursement. The 'Lesser of' statement might not be found in immediate proximity to the reimbursement term and may instead be found in a paragraph above. If the methodology is presented in a table, concatenate any relevant lesser of statement that applies to the table with the portion of the methodology found in the table.
Here are some examples of language with additional context to be included in the FULL_SERVICE:
Text: 'Covered Services that are Medicare Covered Services and are not Medicaid Covered Services',
@@ -31,13 +34,13 @@ Text: 'Where Payor is the Payor for both Medicare Covered Services and Medicaid
Text: 'Outpatient Services'
Here is an example of text with a Lesser of statement and a table that should be included as a FULL_METHODOLOGY answer:
Text: 'Covered Services is the lesser of: (i) Allowable Charges; or (ii) the "Contracted Rate" percentage found in Table 1. Table 1 - OB and Anesthesia Services : 100% of the Payor's Oregon Health Plan DMAP fee schedule. Radiology Services : 110% of Medicare Fee Schedule'
Answer: "[{{'FULL_SERVICE' : 'OB and Anesthesia Services', 'SUBHEADER' : 'N/A', 'FULL_METHODOLOGY' : 'lesser of (i) Allowable Charges; or (ii) 100% of the Payor's Oregon Health Plan DMAP fee schedule'}},
{{'FULL_SERVICE' : 'Radiology Services', 'SUBHEADER' : 'N/A', 'FULL_METHODOLOGY' : 'lesser of (i) Allowable Charges; or (ii) 110% of Medicare Fee Schedule'}}]"
Text: 'Professional Covered Services is the lesser of: (i) Allowable Charges; or (ii) the "Contracted Rate" percentage found in Table 1. Table 1 - OB and Anesthesia Services : 100% of the Payor's Oregon Health Plan DMAP fee schedule. Radiology Services : 110% of Medicare Fee Schedule'
Answer: "[{{'FULL_SERVICE' : 'OB and Anesthesia Services', 'SUBHEADER' : 'N/A', 'FULL_METHODOLOGY' : 'lesser of (i) Allowable Charges; or (ii) 100% of the Payor's Oregon Health Plan DMAP fee schedule', 'PROV_TYPE' :'Professional'}},
{{'FULL_SERVICE' : 'Radiology Services', 'SUBHEADER' : 'N/A', 'FULL_METHODOLOGY' : 'lesser of (i) Allowable Charges; or (ii) 110% of Medicare Fee Schedule', 'PROV_TYPE' :'Professional'}}]"
If there is no clear FULL_SERVICE, then fill in 'Covered Services' for the FULL_SERVICE value.
Ensure that you do not forget to extract reimbursement items. MAKE SURE THAT YOU DON'T MISS ANY REIMBURSEMENT ITEMS AND RETURN ALL NECESSARY AND RELEVANT JSON OBJECTS.
Ensure that you do not forget to extract reimbursement items. MAKE SURE THAT YOU DON'T MISS ANY REIMBURSEMENT ITEMS AND RETURN ALL NECESSARY AND RELEVANT JSON OBJECTS.
Only return the list, with no other commentary or explanation. Ensure you abide by proper JSON formatting.
"""
@@ -110,17 +113,6 @@ Only return the dictionary, with no other commentary or explanation. Ensure you
"""
def BOTTOM_UP_100_6(d):
return f"""Analyze the service and full methodology text listed below:
Service: {d['FULL_SERVICE']}
Methodology: {d['FULL_METHODOLOGY']}
If the context indicates that the Service is a drug paid at 100% ASP + 6%, return 'Y'. Otherwise, return 'N'.
Return only the answer, without using complete sentences.
"""
#####################################################################################
##################################### TOP DOWN ######################################
#####################################################################################
@@ -135,7 +127,7 @@ Here are the attributes to be included in each dictionary, and instructions on h
'EXHIBIT' : List any exhibit, attachment, or amendment names found on the page. Write the full name of the exhibit, including the exhibit number or letter, as well as any other subtitles describing the contents of the exhibit. Do NOT write the page number. If no Exhibit, Attachment, or Amendment is found, write 'N/A'.
'CONTRACT_LOB' : List the Line of Business mentioned on the page. Choose ONLY from the following: {valid.VALID_LOBS}. If there are multiple on the page, write them in a comma-separated list. If no LOB is found, write 'N/A'.
'CONTRACT_PROGRAM' : List any Programs or Plans mentioned on the page. {valid.VALID_PROGRAMS}. If no program is found, write 'N/A'.
'PROV_TYPE' : Write the Provider Type of the page. It will likely be found in the header. Choose ONLY from the following: {valid.VALID_PROV_TYPES}. Note that Facility may also be referred to as Hospital, Clinic, or similar. Professional may also be referred to as Physician, Physician Services, Practicioner, Provider or similar. Do NOT write 'N/A' for this field.
'DEFAULT_TERM' : If there is a section labelled 'Default' or similar, write that entire section here. It may also be a term that describes the payment rate if there is no established payment amount. If there is no such term, write 'N/A'.
'DEFAULT_RATE' : Write the rate from the Default Term in the form 'X% of Y'. If there is no Default Term, or no rate in the Default Term, write 'N/A'.
'CDM_IND' : Is there language on the page regarding CDM Neutralization? Write 'Y' if yes, 'N' if no.
@@ -154,7 +146,6 @@ TD_PRIMARY_FIELDS = [
"EXHIBIT",
"CONTRACT_LOB",
"CONTRACT_PROGRAM",
"PROV_TYPE",
"DEFAULT_TERM",
"DEFAULT_RATE",
"CDM_IND",
@@ -245,12 +236,14 @@ def CONDITIONAL_PROV_2(d, page):
elif "FACILITY" in d["PROV_TYPE"].upper():
valid_list = valid.VALID_FAC
else:
valid_list = valid.VALID_PROF + ", " + valid.VALID_ANC + ", " + valid.VALID_FAC
valid_list = valid.VALID_PROF + valid.VALID_ANC + valid.VALID_FAC
return f"""### PAGE START ### {page} ### PAGE END
The above text contains a number of reimbursement terms. For the rest of this request, focus only on the specific term listed below:
Service: {d['FULL_SERVICE']}
Methodology: {d['FULL_METHODOLOGY']}
Exhibit Name: {d['EXHIBIT']}
Based on the context given, what is the Provider Type applicable to the term above?
@@ -388,102 +381,102 @@ def AC_MULTI_FIELD_TEMPLATE(context, questions):
{context}
## END CONTRACT TEXT ##
Answer the following questions, using the above contract text given as context:
Answer the following questions, using the above contract text given as context.
For each question, provide ONLY the answer, with no additional commentary or explanation. Answer in JSON dictionary format, with every key included and no missing answers.
If for any question, no correct answer is found, return 'N/A'.
{questions}.
Do NOT add any additional commentary explaining the answer. Return ONLY the exact text from the contract, or N/A.
"""
AC_DICT = {
"CONTRACT_EFFECTIVE_DT": "Extract the contract effective date. This may be mentioned in one of the following locations: the signatory section, the preamble of the agreement, or the start of the amendment. Look for phrases such as /'This amendment is effective/'. Return this date in YYYY-MM-DD format.",
"CONTRACT_TITLE": "What is the title of the contract? It generally contains the word Agreement and is found at the very beginning of the document before a paragraph. This will be filled out if there is an amendment or numbered amendment. It typically contains the words agreement, amendment, or contract. An amendment typically has a number attached to it. The typical structure is the provider name, followed by the phrase with agreement, followed potentially by a number signifying a version of the contract. Only return the title, do not return any context.",
"CONTRACT_EFFECTIVE_DT": "Extract the contract effective date. This may be mentioned in one of the following locations: the signatory section, the preamble of the agreement, or the start of the amendment. Look for phrases such as /'This amendment is effective/'. Return this date in YYYY-MM-DD format.",
"CONTRACT_TITLE": "What is the title of the contract? It generally contains the word Agreement and is found at the very beginning of the document before a paragraph. This will be filled out if there is an amendment or numbered amendment. It typically contains the words agreement, amendment, or contract. An amendment typically has a number attached to it. The typical structure is the provider name, followed by the phrase with agreement, followed potentially by a number signifying a version of the contract.",
"PAYER_NAME": "What is the name of the payer that is a party to the contract as stated in the Preamble?",
"PROV_GROUP_NAME": "What is the name of the Group provider that is a party to the contract? This is generally found near keyword provider. Only return the name of the Group provider along with dba name, do not provide any context.",
"PROV_GROUP_NAME": "What is the name of the Group provider that is a party to the contract? This is generally found near keyword provider. Only return the name of the Group provider along with dba name.",
"PROV_GROUP_NPI": "What is the group provider's national provider identifier number mentioned in the contract? This may be found on the signature page or on the roster. It consists of 10 digits. Only return these 10 digits.",
"PROV_GROUP_TIN_SIGNATORY": "What is the Group provider's taxpayer identification number stated near signature in the contract? This is a 9 digit long number typically with a hyphen after the first 2 digits. Return only this 9 digit number with a hyphen after the first 2 digits.",
"PROV_GROUP_TIN": "What is the Group provider's taxpayer identification number stated in the contract? This is a 9 digit long number typically with a hyphen after the first 2 digits in it. Return only this 9 digit number with a hyphen after the first 2 digits. As an example, it has the format of XX-XXXXXXX Do not elaborate or add context. If not found, then only return your answer with N/A. Do not pull any dates, do not pull any numbers that deviate from earlier described pattern. It must always be 9 digits. Either return the number found, or return N/A if not found. Strictly follow this instructions. Do not return any other information.",
"PROV_GROUP_TIN": "What is the Group provider's taxpayer identification number stated in the contract? This is a 9 digit long number typically with a hyphen after the first 2 digits in it. Return only this 9 digit number with a hyphen after the first 2 digits. As an example, it has the format of XX-XXXXXXX Do not elaborate or add context. Do not pull any dates, do not pull any numbers that deviate from earlier described pattern. It must always be 9 digits. Strictly follow this instructions. Do not return any other information.",
"PROV_NPI_OTHER": 'What are all the other provider national provider identifier numbers (NPI) associated with this agreement? They consist of 10 digits and do not contain "-". Provide the list of all the other provider identifier numbers including those that are present in the table or attachment.',
"PROV_TIN_OTHER": "What are the provider taxpayer identification numbers (TIN) for each of the other providers associated with this agreement? They consist of 9 digits and may contain a hyphen after the first 2 digits. This is usually listed in the preample of the contract or on a roster. Provide only the 9 digit TIN.",
"PROV_TIN_OTHER": "What are the provider taxpayer identification numbers (TIN) for each of the other providers associated with this agreement? They consist of 9 digits and may contain a hyphen after the first 2 digits. This is usually listed in the preample of the contract or on a roster. Provide only the 9 digit TINs.",
"TERMINATION_UPON_NOTICE": 'Extract the number of days can the contract be terminated by either party by giving notice. The answer may be found near the words "written notice of such termination" or "prior to the expiration". Only extract a value if one is present. Only return the number of days and nothing else.',
"TERMINATION_UPON_CAUSE": 'After how many days can the contract be terminated by either party upon breach of material term? The answer is usually found in section 7.2.2 or 10.2 but might be present in other sections as well. The answer is present after the Term language section. The answer may be found near the words "is in breach of any material term or condition". Here is an example of possible input text and the correct response. EXAMPLE INPUT TEXT: "By either party upon ninety (90) days prior written notice if the other party is in material breach of this Agreement, except that such termination shall not take place if the breach is cured within sixty (60) days following the written notice". EXAMPLE RESPONSE: 90. Only return the answer, with no other commentary or explanation.',
"TIME_TO_OBJECT": "How many days does the provider have to object to the amendment? The answer may be present in the Amendment section. Only return the number of days nothing else.",
"ASSIGNMENTS_CLAUSE_IND": 'Is assignment clause present in contract language? The answer is usually found in section 8.3 but might not always be present there. It can be found in the section with title containing "Assignment". Answer with a Yes or No, do not provide any other context.',
"NOTICE_PROVIDER_NAME": 'What is the notice provider name present usually near keywords like "Attn:". Return the full name, do not return any context. If it only says President/CEO and not the name itself then return N/A. Ensure that you do not pull the Payer\'s (EG. Superior Health Plan) info instead. Only pull the provider name. If it is not present, answer with N/A',
"NOTICE_PROVIDER_ADDRESS": "What is the notice provider address mentioned after the name of the provider. Only return the address of the provider, do not return any context. New Info --> Ensure Payer address (Eg. Superior Health Plan) isn't pulled. The payer address usually starts with 2100 South... Ensure you do not pull this. Always pull the provider address only.Ensure that you pull the full address with the city and state and zipcode. MAKE SURE YOU ALWAYS PULL THE CITY AND STATE AND ZIPCODE AT THE END. If it is not present in the text file, then do your best to atleast return the state. If it is not present, answer with N/A.",
"NPI_NAME": "What is the group provider's name whose national provider identifier number is mentioned in the contract? This may be found on the signature page or on the roster. Only return the name, do not provide any context.",
"TERMINATION_UPON_CAUSE": 'After how many days can the contract be terminated by either party upon breach of material term? The answer is usually found in section 7.2.2 or 10.2 but might be present in other sections as well. The answer is present after the Term language section. The answer may be found near the words "is in breach of any material term or condition". Here is an example of possible input text and the correct response. EXAMPLE INPUT TEXT: "By either party upon ninety (90) days prior written notice if the other party is in material breach of this Agreement, except that such termination shall not take place if the breach is cured within sixty (60) days following the written notice". EXAMPLE RESPONSE: 90.',
"TIME_TO_OBJECT": "How many days does the provider have to object to the amendment? The answer may be present in the Amendment section.",
"ASSIGNMENTS_CLAUSE_IND": 'Is assignment clause present in contract language? The answer is usually found in section 8.3 but might not always be present there. It can be found in the section with title containing "Assignment". Answer ONLY with Y or N. Do NOT write N/A.',
"NOTICE_PROVIDER_NAME": 'What is the notice provider name present usually near keywords like "Attn:". Return the full name, do not return any context. If it only says President/CEO and not the name itself then return N/A. Ensure that you do not pull the Payer\'s (EG. Superior Health Plan) info instead. Only pull the provider name.',
"NOTICE_PROVIDER_ADDRESS": "What is the notice provider address mentioned after the name of the provider. Only return the address of the provider, do not return any context. New Info --> Ensure Payer address (Eg. Superior Health Plan) isn't pulled. The payer address usually starts with 2100 South... Ensure you do not pull this. Always pull the provider address only. Ensure that you pull the full address with the city and state and zipcode. MAKE SURE YOU ALWAYS PULL THE CITY AND STATE AND ZIPCODE AT THE END. If it is not present in the text file, then do your best to at least return the state.",
"NPI_NAME": "What is the group provider's name whose national provider identifier number is mentioned in the contract? This may be found on the signature page or on the roster.",
"HEALTH_PLAN_STATE": "Which state is this health plan for?",
"SEQUESTRATION_LANGUAGE": "Is sequestration language present in the contract? It may be after reimbursement or compensation section. It is only present in case of medicare and medicare advantage. If Yes, extract the complete paragraph.",
# "SEQUESTRATION_REDUCTIONS_IND":"Are sequestration and reduction percentage present in the contract? Answer Yes if both are present otherwise answer No.",
"CONTRACT_AUTO_RENEWAL_IND": "Does the contract automatically renew? Answer with a Yes or No, do not provide any other context. If the information is not present, answer N/A.",
"AMEND_CONTRACT_NOTICE_IND": 'Is amendment clause present in contract language? The answer is usually found in section 8.7 but might not always be present there. It may be found in the section with title containing "Amendment". Answer may be found near keywords "Agreement may be amended". Answer with a Yes or No, do not leave the answer as blank. Do not provide any other context. Here is an example of possible input text and the correct response. EXAMPLE INPUT TEXT: "may amend this Agreement by giving Provider written notice". EXAMPLE RESPONSE: Yes. If the agreement does not explicitly indicate that it can be amended by giving written notice, the answer will be No.',
"AFFILIATION_CLAUSE_IND": 'Is affiliation clause present in contract language? Answer may be found near keywords like "affiliate means a person or entity directly". Answer with a Yes or No, do not provide any other context.',
"CONTRACT_AUTO_RENEWAL_IND": "Does the contract automatically renew? Answer ONLY with Y or N. Do NOT write N/A.",
"AMEND_CONTRACT_NOTICE_IND": 'Is amendment clause present in contract language? The answer is usually found in section 8.7 but might not always be present there. It may be found in the section with title containing "Amendment". Answer may be found near keywords "Agreement may be amended". Answer ONLY with Y or N. Do NOT write N/A.. Here is an example of possible input text and the correct response. EXAMPLE INPUT TEXT: "may amend this Agreement by giving Provider written notice". EXAMPLE RESPONSE: "Y". If the agreement does not explicitly indicate that it can be amended by giving written notice, the answer will be "N".',
"AFFILIATION_CLAUSE_IND": 'Is affiliation clause present in contract language? Answer may be found near keywords like "affiliate means a person or entity directly". Answer ONLY with Y or N. Do NOT write N/A.',
# "TERM_CLAUSE": """Extract the subsection or paragraph(s) labeled Term. It may also be labeled Termination. In instances there may be 2 bodies of text that you will need to pull. For example, first paragraph may be 'MCO must follow the procedures outlines in Section...' and the second one may be 'Not later than thirty (30) days following receipt of the termination notice...'. In this case, ensure that YOU ALWAYS PULL the whole text of the two paragraphs. If there are no such sections, extract the paragraph that most closely resembles this label. If there is still no correct answer, simply return 'N/A'.""",
"TERM_CLAUSE": """Extract the term or termination sections (or subsections or paragraphs), which are always present in the contract. They contain information related to term and termination of the contract. The biggest hint of a term or termination section will be the heading 'Term and Termination'. A secondary hint is that the first paragraph may be 'MCO must follow the procedures outlined in Section...' followed by 'Not later than XX days following receipt of the termination notice...'. In this case, ensure that YOU ALWAYS PULL the whole text of all relevant paragraphs.""",
# "TIMELY_FILING": "Extract all of the clauses of text and section paragraphs along with their section number that are pertaining to the Timely Filing clause. This clause usually details the conditions and days for when a provider must submit a claim by. Look for section heading titled /'Payment of Clean Claims/'. Look for language such as /'all provider claims shall be processed within 30 days from the date of claim reciept by the MCO/'. We want to extract all following langauage such as MCO shall not pay any claim submitted by provider..., MCO must adjudicate all appealed claims, till the whole section that starts with 'MCO may deny a claim for failure to file'. If it is not present, answer N/A.",
"CREDENTIALING_APP_IND": "Is credentialing defined in section 2.4 or any other subsection of the contract? Answer with a Yes or No, do not provide any other context.",
"CONTRACT_TERMINATION_DT": "What is the contract termination date? This might be listed within the Terms or Terms of Agreement section of the contract, or can be calculated by adding contract duration to the contract effective date. Only return the date, do not return any context.",
"TIME_TO_OBJECT": "How many days does the provider have to object to the amendment? The answer may be present in the Amendment section. Only return the number of days nothing else.",
# "TIMELY_FILING": "Extract all of the clauses of text and section paragraphs along with their section number that are pertaining to the Timely Filing clause. This clause usually details the conditions and days for when a provider must submit a claim by. Look for section heading titled /'Payment of Clean Claims/'. Look for language such as /'all provider claims shall be processed within 30 days from the date of claim reciept by the MCO/'. We want to extract all following langauage such as MCO shall not pay any claim submitted by provider..., MCO must adjudicate all appealed claims, till the whole section that starts with 'MCO may deny a claim for failure to file'.",
"CREDENTIALING_APP_IND": "Is credentialing defined in section 2.4 or any other subsection of the contract? Answer ONLY with Y or N. Do NOT write N/A.",
"CONTRACT_TERMINATION_DT": "What is the contract termination date? This might be listed within the Terms or Terms of Agreement section of the contract, or can be calculated by adding contract duration to the contract effective date. Only return the date.",
"TIME_TO_OBJECT": "How many days does the provider have to object to the amendment? The answer may be present in the Amendment section. Only return the number of days, as an integer.",
"NON_RENEWAL_LANGUAGE": "Extract the section of the contract that describes non-renewal obligations, especially the number of days notice that must be given. Write the full relevant sentence or paragraph. Ensure that the exact term 'non-renewal' is present in the answer. Do NOT return an answer that does not have 'non-renewal'.",
"NON_RENEWAL_DAYS": "How many days of notice does the section indicate must be given to terminate the agreement? Examples are: 30, 60, 90, 180. Only return the number of days.",
"ACCESS_TO_MEDICAL_RECORDS": "Access clause contains information regarding access to medical records. It may be found in section 4.2 of the contract. Extract the Access clause present in the contract. If it is not present, answer N/A.",
"ACCESS_TO_MEDICAL_RECORDS": "Access clause contains information regarding access to medical records. It may be found in section 4.2 of the contract. Extract the Access clause present in the contract.",
# "ADD_ON_REIMBURSEMENT_IND":"Add On Reimbursement is usually present in Compensation section of the contract. Is Add On Reimbursement present in the Compensation section of the contract? Answer with a Yes or No, do not provide any other context.",
# "ADD_ON_REIMBURSEMENT_LANGUAGE":"Add On Reimbursement language is usually present in Compensation section of the contract. Extract the Add On Reimbursement language present in the contract. If it is not present, answer N/A.",
"CARVEOUT_VENDORS": "Carve-Out Vendors clause may be found in section 2.9 of the contract. Extract the Carve-Out Vendors clause present in the contract. If it is not present, answer N/A.",
"CLAIMS_EDITING_LANGUAGE": "Claims Editing Language is typically found in the claims section in the beginning parts of the contract. Extract Claims Editing Language present in the contract. If it is not present, answer N/A.",
"CARVEOUT_VENDORS": "Carve-Out Vendors clause may be found in section 2.9 of the contract. Extract the Carve-Out Vendors clause present in the contract.",
"CLAIMS_EDITING_LANGUAGE": "Claims Editing Language is typically found in the claims section in the beginning parts of the contract. Extract Claims Editing Language present in the contract.",
# "CLAIMS_EDITING_LANGUAGE_IND":"Claims Editing Language is typically found in the claims section in the beginning parts of the contract. Is Claims Editing Language present in the contract? Answer with a Yes or No, do not provide any other context.",
"CLEAN_CLAIM": 'Extract the "Clean Claim" subsection present in definition section of the contract. If it is not present, answer N/A.',
"CLEAN_CLAIM": 'Extract the "Clean Claim" subsection present in definition section of the contract.',
# "CONFLICTS_BETWEEN_CERTAIN_DOCUMENTS_IND":"Is Conflicts Between Certain Documents clause present in the contract? Answer with a Yes or No, do not provide any other context.",
"CONFLICTS_BETWEEN_CERTAIN_DOCUMENTS_LANGUAGE": "Extract the Conflicts Between Certain Documents clause present in the contract. If it is not present, answer N/A.",
"CONFLICTS_BETWEEN_CERTAIN_DOCUMENTS_LANGUAGE": "Extract the Conflicts Between Certain Documents clause present in the contract.",
# "COST_SETTLEMENT_IND":"Is cost settlement language present in the contract? Answer with a Yes or No, do not provide any other context.",
"COST_SETTLEMENT_LANGUAGE": "Cost settlement language may be present in disputes and arbitration subsection of the contract. It usually contains keywords like settlementor cost settlement. Extract the cost settlement language present in the contract. If it is not present, answer N/A.",
"DEEMER_AMENDMENT": "Deemer Amendment clause may be found in section 8.7.2 of the contract. Extract the Deemer Amendment clause present in the contract. If it is not present, answer N/A.",
"COST_SETTLEMENT_LANGUAGE": "Cost settlement language may be present in disputes and arbitration subsection of the contract. It usually contains keywords like settlementor cost settlement. Extract the cost settlement language present in the contract.",
"DEEMER_AMENDMENT": "Deemer Amendment clause may be found in section 8.7.2 of the contract. Extract the Deemer Amendment clause present in the contract.",
# "DELEGATED_FUNCTION_IND":"Is Delegated Function clause present in the contract? Answer with a Yes or No, do not provide any other context.",
"DELEGATED_TERMS": "Extract Delegated Terms present in the contract. If it is not present, answer N/A.",
"ECM": 'What is the "ECM" number mentioned in the contract? Only Answer the ECM number, do not provide any context. Do not provide ICM number. If ECM number is not present, answer N/A.',
"DELEGATED_TERMS": "Extract Delegated Terms present in the contract.",
"ECM": 'What is the "ECM" number mentioned in the contract? Only Answer the ECM number, do not provide any context. Do not provide ICM number.',
# "EXCLUSIVITY_REQUIREMENT_IND":"Is Exclusivity requirement clause present in the contract? Answer with a Yes or No, do not provide any other context.",
"EXCLUSIVITY_REQUIREMENT_LANGUAGE": "Extract Exclusivity requirement clause present in the contract. If it is not present, answer N/A.",
"EXCLUSIVITY_REQUIREMENT_LANGUAGE": "Extract Exclusivity requirement clause present in the contract.",
# "GUARANTEE_OF_PROVIDER_YIELD_IND":"Is Guarantee of Provider Yield Language present in the contract? Answer with a Yes or No, do not provide any other context.",
"GUARANTEE_OF_PROVIDER_YIELD_LANGUAGE": "Extract Guarantee of Provider Yield Language present in the contract. If it is not present, answer N/A.",
"HCBS_SERVICES": 'Extract HCBS services clause present either in the compensation or exhibit section of the contract. It may contain keywords like "for covered HCBS services, plan shall pay". Do not include definitions of HCBS. If it is not present, answer N/A.',
"INDEMNIFICATION": "Indemnification clause may be found in section 5.2 of the contract. Extract the Indemnification clause present in the contract. If it is not present, answer N/A.",
"GUARANTEE_OF_PROVIDER_YIELD_LANGUAGE": "Extract Guarantee of Provider Yield Language present in the contract.",
"HCBS_SERVICES": 'Extract HCBS services clause present either in the compensation or exhibit section of the contract. It may contain keywords like "for covered HCBS services, plan shall pay". Do not include definitions of HCBS.',
"INDEMNIFICATION": "Indemnification clause may be found in section 5.2 of the contract. Extract the Indemnification clause present in the contract. ",
# "INDEPENDENT_REVIEW_IND":"Is Independent Review language present in the contract? Answer with a Yes or No, do not provide any other context.",
"INDEPENDENT_REVIEW_LANGUAGE": "Extract Independent Review language present in the contract. If it is not present, answer N/A.",
"INDEPENDENT_REVIEW_LANGUAGE": "Extract Independent Review language present in the contract.",
# "INVOICE_PRICING_IND":"Is table with heading \"INVOICED SERVICES\" present in compensation schedule of the contract?",
"INVOICE_PRICING_LANGUAGE": 'Extract table with heading "INVOICED SERVICES" if it is present in compensation schedule of the contract. If it is not present, answer N/A',
"INVOICE_PRICING_LANGUAGE": 'Extract table with heading "INVOICED SERVICES" if it is present in compensation schedule of the contract.',
# "LATE_PAID_CLAIMS_IND":"Is Late Paid Claims language present in the contract? Answer with a Yes or No, do not provide any other context.",
"LATE_PAID_CLAIMS_LANGUAGE": "Extract Late Paid Claims language present in the contract. This will be a sentence or paragraph describing late paid claims, interest payments or similar. Here is an example of a correct response: 'Any Clean Claim, as defined in 42 C.F.R. § 422.500, shall be paid within thirty (30) days of receipt by Plan at such address as may be designated by Plan, and Plan shall pay interest on any Clean Claim not paid within thirty (30) days of such receipt by Plan at the rate of interest required by law, or as otherwise set forth in the Provider Manual.'. Here is an example of an incorrect response: 'Any Clean Claim, as defined in 42 C.F.R. 422.500, shall be paid within thirty (30) days of receipt by Health Plan, Payor or (if Provider contracts with Downstream Entities) Provider, as applicable, as designated by Provider or such Downstream Entity, as applicable.'. If no valid answer is present, answer N/A.",
"LATE_PAID_CLAIMS_LANGUAGE": "Extract Late Paid Claims language present in the contract. This will be a sentence or paragraph describing late paid claims, interest payments or similar. Here is an example of a correct response: 'Any Clean Claim, as defined in 42 C.F.R. § 422.500, shall be paid within thirty (30) days of receipt by Plan at such address as may be designated by Plan, and Plan shall pay interest on any Clean Claim not paid within thirty (30) days of such receipt by Plan at the rate of interest required by law, or as otherwise set forth in the Provider Manual.'. Here is an example of an incorrect response: 'Any Clean Claim, as defined in 42 C.F.R. 422.500, shall be paid within thirty (30) days of receipt by Health Plan, Payor or (if Provider contracts with Downstream Entities) Provider, as applicable, as designated by Provider or such Downstream Entity, as applicable.'.",
# "MEDICAL_NECESSITY_LANGUAGE":"Extract the \"Medically Necessary\" or \"Medical Necessity\" care (or services) verbiage mentioned in the compensation schedule or section of the contract. Do not mention what \"Medically Necessary\" means. If it is not present in compensation schedule or section, answer N/A.",
# "MEDICAL_NECESSITY_LANGUAGE_IND":"Is \"Medically Necessary\" or \"Medical Necessity\" care (or services) verbiage mentioned in the compensation schedule or section of the contract? Answer with a Yes or No, do not provide any other context.",
"MEMBER_CONFINEMENT_DAYS_LANGUAGE": "Extract Member Confinement Days Language present in the contract. If it is not present, answer N/A.",
"MEMBER_CONFINEMENT_DAYS_LANGUAGE": "Extract Member Confinement Days Language present in the contract.",
# "MEMBER_CONFINEMENT_DAYS_LANGUAGE_IND":"Is Member Confinement Days Language present in the contract? Answer with a Yes or No, do not provide any other context.",
"NATIONAL_AGREEMENT_IND": "Are more than one states covered in the contract? Answer with a Yes or No, do not provide any other context.",
"NATIONAL_AGREEMENT_IND": "Are more than one states covered in the contract? Answer ONLY with Y or N. Do NOT write N/A.",
# "NETWORK_ACCESS_FEES_IND":"Is \"Network Access Fee\" present in the contract? If yes what is the fee mentioned?",
"NETWORK_ACCESS_FEES_LANGUAGE": 'Extract "Network Access Fee" verbiage present in the contract. If it is not present, answer N/A.',
"NETWORK_ACCESS_FEES_LANGUAGE": 'Extract "Network Access Fee" verbiage present in the contract.',
# "NONSTANDARD_APPEALS_PROCESS_IND":"Is Nonstandard Appeals Process present in the contract? Answer with a Yes or No, do not provide any other context.",
"NONSTANDARD_APPEALS_PROCESS_LANGUAGE": "Extract Nonstandard Appeals Process language present in the contract. If it is not present, answer N/A.",
"PAYMENT_IN_ADVANCE_OF_CLAIMS_SUBMISSION_LANGUAGE": "Extract Payment in Advance of Claims Submission language present in the contract. If it is not present, answer N/A.",
"NONSTANDARD_APPEALS_PROCESS_LANGUAGE": "Extract Nonstandard Appeals Process language present in the contract.",
"PAYMENT_IN_ADVANCE_OF_CLAIMS_SUBMISSION_LANGUAGE": "Extract Payment in Advance of Claims Submission language present in the contract.",
# "PAYMENT_IN_ADVANCE_OF_CLAIMS_SUBMISSION_LANGUAGE_IND":"Is Payment in Advance of Claims Submission language present in the contract? Answer with a Yes or No, do not provide any other context.",
"PAYOR": '"Payor" is usually found in section 1.12 of the contract. Extract the "Payor" subsection present in definition section of the contract. If it is not present, answer N/A.',
"PMPM": "What is the PMPM rate mentioned in the contract? Only return the rate. Do not provide any context. If it is not present, answer N/A.",
"PREAUTHORIZATION": "Preauthorization clause may be found in section 2.7 of the contract. Extract the Preauthorization clause present in the contract. If it is not present, answer N/A.",
"PRODUCT_REMOVAL": "Product removal may be found in the products attachment, stating a product is being removed. Extract product removal language present in the contract. If it is not present, answer N/A.",
"PAYOR": '"Payor" is usually found in section 1.12 of the contract. Extract the "Payor" subsection present in definition section of the contract.',
"PMPM": "What is the PMPM rate mentioned in the contract? Only return the rate. Do not provide any context.",
"PREAUTHORIZATION": "Preauthorization clause may be found in section 2.7 of the contract. Extract the Preauthorization clause present in the contract.",
"PRODUCT_REMOVAL": "Product removal may be found in the products attachment, stating a product is being removed. Extract product removal language present in the contract.",
# "PROVIDER_BASED_BILLING_EXCLUSION_IND":"Provider-based Billing Exclusion is often listed under the Additional Provisions section. Is Provider-based Billing Exclusion mentioned the contract? Answer with a Yes or No, do not provide any other context.",
"PROVIDER_BASED_BILLING_EXCLUSION_LANGUAGE": "Provider-based Billing Exclusion is often listed under the Additional Provisions section. Extract the Provider-based Billing Exclusion language mentioned in the contract. If it is not present, answer N/A.",
"RECOVERY_RIGHTS": "Recovery Rights clause may be found in section 3.5 of the contract. Extract the entire Recovery Rights sentence or paragraph present in the contract. If it is not present, answer N/A.",
"PROVIDER_BASED_BILLING_EXCLUSION_LANGUAGE": "Provider-based Billing Exclusion is often listed under the Additional Provisions section. Extract the Provider-based Billing Exclusion language mentioned in the contract.",
"RECOVERY_RIGHTS": "Recovery Rights clause may be found in section 3.5 of the contract. Extract the entire Recovery Rights sentence or paragraph present in the contract.",
# "SINGLE_CODE_MULTIPLE_RATES_IND":"Is Single Code Multiple Rates present in the compensation or exhibit section of the contract? Answer with a Yes or No, do not provide any other context.",
# "SINGLE_CODE_MULTIPLE_RATES_LANGUAGE":"Extract Single Code Multiple Rates present in the compensation or exhibit section of the contract. If it is not present, answer N/A.",
"TEMPLATE": "Is this contract in standard Centene template containing definition, terms clauses and other standard sections. Answer with a Yes or No, do not provide any other context.",
"ARBITRATION_AND_DISPUTES": "Dispute Resolution clause is usually found in section 6.1 of the contract. Arbitration clause is usually found in section 6.2 of the contract. Extract both the Dispute Resolution and the arbitration clause present in the contract. In case one of them is present, extract that one. If none of them are present, answer N/A.",
"TEMPLATE": "Is this contract in standard Centene template containing definition, terms clauses and other standard sections. Answer with 'Y' or 'N'.",
"ARBITRATION_AND_DISPUTES": "Dispute Resolution clause is usually found in section 6.1 of the contract. Arbitration clause is usually found in section 6.2 of the contract. Extract both the Dispute Resolution and the arbitration clause present in the contract. In case one of them is present, extract that one.",
# "DISPARAGEMENT_PROHIBITION_IND":"Is Disparagement Prohibition clause present in the contract? It may be found in section 2.1 of the contract. Answer with a Yes or No, do not provide any other context.",
"DISPARAGEMENT_PROHIBITION_LANGUAGE": "Disparagement Prohibition clause may be found in section 2.1 of the contract. Extract the Disparagement Prohibition clause present in the contract. If it is not present, answer N/A.",
"ELIGIBILITY_VERIFICATION": "Eligibility Verification or Determination clause may be found in section 2.6 of the contract. Extract the Eligibility Verification or Determination clause present in the contract. If it is not present, answer N/A.",
"INSURANCE_REQUIREMENT": "Extract Insurance requirement mentioned in the contract. If it is not present, answer N/A.",
"PARTICIPATION_IN_PRODUCTS": "Participation in Products clause may be found in section 2.2.2 of the contract. Extract the Participation in Products clause present in the contract. If it is not present, answer N/A.",
"POLICIES_AND_PROCEDURES": "Extract the entire Policies and Procedures sentence or paragraph present in the contract. Do not extract language about conflicts and construction. Do not extract language about PP (Preferred Provider) nor PPG (Preferred Provider Group). If it is not present, answer N/A.",
"REGULATORY_REQUIREMENTS": 'Extract the entire Regulatory Requirements paragraph present in the contract. It usually contains the keyword "Regulatory Requirements". If it is not present, answer N/A.',
"DISPARAGEMENT_PROHIBITION_LANGUAGE": "Disparagement Prohibition clause may be found in section 2.1 of the contract. Extract the Disparagement Prohibition clause present in the contract.",
"ELIGIBILITY_VERIFICATION": "Eligibility Verification or Determination clause may be found in section 2.6 of the contract. Extract the Eligibility Verification or Determination clause present in the contract.",
"INSURANCE_REQUIREMENT": "Extract Insurance requirement mentioned in the contract.",
"PARTICIPATION_IN_PRODUCTS": "Participation in Products clause may be found in section 2.2.2 of the contract. Extract the Participation in Products clause present in the contract.",
"POLICIES_AND_PROCEDURES": "Extract the entire Policies and Procedures sentence or paragraph present in the contract. Do not extract language about conflicts and construction. Do not extract language about PP (Preferred Provider) nor PPG (Preferred Provider Group).",
"REGULATORY_REQUIREMENTS": 'Extract the entire Regulatory Requirements paragraph present in the contract. It usually contains the keyword "Regulatory Requirements".',
# "RELATIONSHIP_OF_PARTIES_IND":"Is Relationship of Parties clause present in the contract? Answer with a Yes or No, do not provide any other context.",
"RELATIONSHIP_OF_PARTIES_LANGUAGE": "Extract the entire Relationship of Parties passage present in the contract. If it is not present, answer N/A.",
"RELATIONSHIP_OF_PARTIES_LANGUAGE": "Extract the entire Relationship of Parties passage present in the contract.",
}
+51
View File
@@ -0,0 +1,51 @@
import pandas as pd
import os
import re
import postprocessing_funcs
import consolidate_output
import utils
import valid
import postprocess
# output_dir = 'output_individual/CNC-5A-Output'
# b_filenames, ac_filenames = [], []
# for subdir in os.listdir(output_dir):
# for file in os.listdir(os.path.join(output_dir, subdir)):
# if file == 'CNC-5A-B.csv' or file == 'CNC-5A-only-remaining-B-B.csv':
# b_filenames.append(subdir)
# elif file == 'CNC-5A-AC.csv' or file == 'CNC-5A-onlyAC-AC.csv':
# ac_filenames.append(subdir)
# print("B Files Complete: " , len(set(b_filenames)))
# print("AC Files Complete: ", len(set(ac_filenames)))
output_dirs = ['../../complex-cnc/doczy.ai/5b-outputs', '../../complex-cnc/doczy.ai/ALL-5B-OUTPUTS']
filename_dict = {}
for output_dir in output_dirs:
for subdir in os.listdir(output_dir):
for file in os.listdir(os.path.join(output_dir, subdir)):
if file in filename_dict:
filename_dict[file] += 1
else:
filename_dict[file] = 0
print(filename_dict)
b_filenames, ac_filenames = [], []
for output_dir in output_dirs:
for subdir in os.listdir(output_dir):
for file in os.listdir(os.path.join(output_dir, subdir)):
if file == '5brun-B.csv' or file == '5brun-remaining-B.csv':
b_filenames.append(subdir)
elif file == '5brun-withallac-AC.csv' or file == '5brun-remaining-AC.csv':
ac_filenames.append(subdir)
print("B Files Complete: " , len(set(b_filenames)))
print("AC Files Complete: ", len(set(ac_filenames)))
+78 -62
View File
@@ -5,13 +5,79 @@ import config
import dict_operations
import valid
import postprocessing_funcs
import csv
import json
import re
def get_exhibit_range(text_dict, exhibit_pages, bu_page):
lower_bound = max([int(page) for page in exhibit_pages if int(page) <= int(bu_page)], default=None)
upper_bound = min([int(page) for page in exhibit_pages if int(page) > int(bu_page)], default=None)
if lower_bound is not None and upper_bound is not None:
pages_between = [page for page in text_dict.keys() if int(lower_bound) <= int(page) <= int(upper_bound)]
elif lower_bound is not None: # No upper_bound found
pages_between = [page for page in text_dict.keys() if int(lower_bound) <= int(page)]
elif upper_bound is not None: # No lower_bound found
pages_between = [page for page in text_dict.keys() if int(page) <= int(upper_bound)]
else:
pages_between = []
return pages_between
def run_top_down(filename, text_dict, bu_results, exhibit_pages):
bu_pages = {bu_dict['page_num'] for bu_dict in bu_results}
td_page_dict = {}
td_answer_dict = {}
for bu_page in bu_pages:
td_pages = get_exhibit_range(text_dict, exhibit_pages, bu_page)
if td_pages not in td_page_dict.values():
answer_dict = run_top_down_primary(text_dict, td_pages, filename)
td_page_dict[bu_page] = td_pages
td_answer_dict[bu_page] = answer_dict
else:
td_page_dict[bu_page] = td_pages
for key, value in td_page_dict.items():
if value == td_pages:
answer_dict = td_answer_dict[key]
td_answer_dict[bu_page] = answer_dict
break
merged_dicts = []
for bu_dict in bu_results:
answer_dict = td_answer_dict[bu_dict['page_num']]
bu_dict.update(answer_dict)
merged_dicts.append(bu_dict)
# Run and Merge Exclusions
exclusion_dict = get_td_dict(
text_dict, filename, prompts.TOP_DOWN_EXCLUSIONS, valid.EXCLUSION_TERMS
)
final_dicts = add_exclusions(merged_dicts, exclusion_dict)
# Run and Merge Medically Necessary
medically_necessary_dict = get_td_dict(
text_dict,
filename,
prompts.TOP_DOWN_MEDICALLY_NECESSARY,
valid.VALID_MEDICALLY_NECESSARY
)
medically_necessary_dict = postprocessing_funcs.filter_dict(
medically_necessary_dict,
pattern=re.compile(
"|".join(
re.escape(keyword) for keyword in valid.INVALID_MEDICALLY_NECESSARY
),
re.IGNORECASE,
),
)
final_dicts = add_medically_necessary(final_dicts, medically_necessary_dict)
return final_dicts
def run_top_down_primary(text_dict, td_pages, filename):
page_text = '\n'.join([text_dict[page_num] for page_num in td_pages])
def run_top_down_primary(text_dict, page_num, filename):
page_text = text_dict[page_num]
prompt = prompts.TOP_DOWN_PRIMARY(page_text)
answer = claude_funcs.invoke_claude(
prompt, config.MODEL_ID_CLAUDE35_SONNET, filename, max_tokens=4096
@@ -35,59 +101,6 @@ def run_top_down_primary(text_dict, page_num, filename):
answer_dict["PROV_TYPE"] = config.PROV_TYPE
return answer_dict
def run_top_down(filename, text_dict, bu_results, exhibit_pages):
## Run TD Primary, Merge with BU
merged_dicts = []
td_dicts = {}
for bu_dict in bu_results:
page = bu_dict["page_num"]
if page in exhibit_pages:
td_page = page
else:
try:
td_page = str(
max([int(i) for i in exhibit_pages if int(i) < int(page)])
)
except:
td_page = page
if td_page not in td_dicts.keys():
answer_dict = run_top_down_primary(text_dict, td_page, filename)
td_dicts[td_page] = answer_dict
else:
answer_dict = td_dicts[td_page]
bu_dict.update(answer_dict)
merged_dicts.append(bu_dict)
# Run and Merge Exclusions
exclusion_dict = get_td_dict(
text_dict, filename, prompts.TOP_DOWN_EXCLUSIONS, valid.EXCLUSION_TERMS
)
final_dicts = add_exclusions(merged_dicts, exclusion_dict)
# Run and Merge Medically Necessary
medically_necessary_dict = get_td_dict(
text_dict,
filename,
prompts.TOP_DOWN_MEDICALLY_NECESSARY,
valid.VALID_MEDICALLY_NECESSARY,
)
medically_necessary_dict = postprocessing_funcs.filter_dict(
medically_necessary_dict,
pattern=re.compile(
"|".join(
re.escape(keyword) for keyword in valid.INVALID_MEDICALLY_NECESSARY
),
re.IGNORECASE,
),
)
final_dicts = add_medically_necessary(final_dicts, medically_necessary_dict)
return final_dicts
def get_td_dict(text_dict, filename, PROMPT_FUNC, VALID_VALUES):
td_dict = {}
for page_num, page_text in text_dict.items():
@@ -106,8 +119,8 @@ def get_td_dict(text_dict, filename, PROMPT_FUNC, VALID_VALUES):
def add_medically_necessary(merged_dicts, medically_necessary_dict):
if not medically_necessary_dict:
for d in merged_dicts:
d["Medical Necessity Language (Language)"] = "N/A"
d["Medical Necessity Language (Y/N)"] = "N"
d["MEDICAL_NECESSITY_LANGUAGE"] = "N/A"
d["MEDICAL_NECESSITY_LANGUAGE_IND"] = 'N'
return merged_dicts
sorted_keys = sorted(int(k) for k in medically_necessary_dict.keys())
@@ -127,13 +140,16 @@ def add_medically_necessary(merged_dicts, medically_necessary_dict):
current_page_num = int(d["page_num"])
# Find the closest preceding medically necessary page
closest_page = find_closest_page(current_page_num)
d["Medical Necessity Language (Language)"] = medically_necessary_dict[
d["MEDICAL_NECESSITY_LANGUAGE"] = medically_necessary_dict[
str(closest_page)
]
d["Medical Necessity Language (Y/N)"] = "Y"
if not utils.is_empty(d["MEDICAL_NECESSITY_LANGUAGE"]):
d["MEDICAL_NECESSITY_LANGUAGE_IND"] = 'Y'
else:
d["MEDICAL_NECESSITY_LANGUAGE_IND"] = 'N'
except:
d["Medical Necessity Language (Language)"] = "N/A"
d["Medical Necessity Language (Y/N)"] = "N"
d["MEDICAL_NECESSITY_LANGUAGE"] = "N/A"
d["MEDICAL_NECESSITY_LANGUAGE_IND"] = "N"
return merged_dicts
+17 -27
View File
@@ -182,39 +182,29 @@ def contains_keyword(text, keywords):
return
def filter_already_processed(input_dict, fields=config.FIELDS):
already_processed = []
def filter_already_processed(input_dict, input_dict_b):
already_processed_ac, already_processed_b = [], []
for folder_name in os.listdir(config.OUTPUT_DIRECTORY):
# ABC
if (
fields == "abc"
and config.AC_RESULTS_NAME
in os.listdir(os.path.join(config.OUTPUT_DIRECTORY, folder_name))
and config.B_RESULTS_NAME
in os.listdir(os.path.join(config.OUTPUT_DIRECTORY, folder_name))
# AC
if (config.AC_RESULTS_NAME in os.listdir(os.path.join(config.OUTPUT_DIRECTORY, folder_name))
):
already_processed.append(folder_name + ".txt")
# Fill AC Here
if (
config.AC_RESULTS_NAME
in os.listdir(os.path.join(config.OUTPUT_DIRECTORY, folder_name))
and config.FIELDS == "ac"
already_processed_ac.append(folder_name + ".txt")
# B
if (config.B_RESULTS_NAME in os.listdir(os.path.join(config.OUTPUT_DIRECTORY, folder_name))
):
already_processed.append(folder_name + ".txt")
# Fill B Here
if (
config.B_RESULTS_NAME
in os.listdir(os.path.join(config.OUTPUT_DIRECTORY, folder_name))
and config.FIELDS == "b"
):
already_processed.append(folder_name + ".txt")
input_dict = {
already_processed_b.append(folder_name + ".txt")
input_dict_ac = {
key: input_dict[key]
for key in input_dict.keys()
if key not in already_processed
if key not in already_processed_ac
}
return input_dict
input_dict_b = {
key: input_dict_b[key]
for key in input_dict_b.keys()
if key not in already_processed_b
}
return input_dict_ac, input_dict_b
def is_empty(value):
+20 -10
View File
@@ -233,11 +233,21 @@ def select_valid_prov_2(provider_type):
valid_list = VALID_PROF + VALID_ANC + VALID_FAC
return valid_list
RATE_REPLACEMENTS = {
"Medicare": "MCR",
"Medicaid": "MCD",
"Allowable Charges": "AC",
"Allowed Amount": "AA",
"Billed Charges": "BC",
"Average Sales Price": "ASP",
"Average Wholesale Price": "AWP",
"Amount Payable by Medicare": "MCR",
"Amount Payable by Medicaid": "MCD"
}
# List of fields that will have derived indicators
DERIVED_INDICATOR_FIELDS = [
"PROVIDER_BASED_BILLING_EXCLUSION_LANGUAGE",
"MEDICAL_NECESSITY_LANGUAGE",
"INVOICE_PRICING_LANGUAGE",
"GUARANTEE_OF_PROVIDER_YIELD_LANGUAGE",
"CLAIMS_EDITING_LANGUAGE",
@@ -253,6 +263,7 @@ DERIVED_INDICATOR_FIELDS = [
"LATE_PAID_CLAIMS_LANGUAGE",
"COST_SETTLEMENT_LANGUAGE",
"DELEGATED_TERMS",
"SEQUESTRATION_LANGUAGE"
]
B_MAPPING = {
@@ -276,8 +287,8 @@ B_MAPPING = {
"FLAT_FEE_STANDARD": "FLAT FEE",
"DEFAULT_TERM": "Default Term",
"DEFAULT_RATE": "Default Rate",
"Medical Necessity Language (Language)": "Medical Necessity Language (Language)",
"Medical Necessity Language (Y/N)": "Medical Necessity Language (Y/N)",
"MEDICAL_NECESSITY_LANGUAGE": "Medical Necessity Language (Language)",
"MEDICAL_NECESSITY_LANGUAGE_IND": "Medical Necessity Language (Y/N)",
'Inclusion of essential RBRVS "Fee Source" Language (Y/N)': 'Inclusion of essential RBRVS "Fee Source" Language (Y/N)',
"CDM_IND": "CDM Neutralization Language, included (Y/N)",
"CHARGEMASTER": "CONTRACT_CHARGEMASTER_PROTECTION_LANGUAGE",
@@ -295,6 +306,8 @@ B_MAPPING = {
AC_MAPPING = {
"Filename": "Contract Name",
"Pages" : "Pages",
"Parent Agreement Code" : "Parent Agreement Code",
"CONTRACT_TITLE": "Agreement_Name (Contract Title)",
"PAYER_NAME": "PAYER NAME",
"AFFILIATION_CLAUSE_IND": "Affiliate (Y/N)",
@@ -318,7 +331,7 @@ AC_MAPPING = {
"HEALTH_PLAN_STATE": "Health Plan State",
"CREDENTIALING_APP_IND": "Credentialing Application Indicator",
"SEQUESTRATION_LANGUAGE": "Sequestration Language",
"SEQUESTRATION_REDUCTIONS_IND": "Sequestration Reductions (Y/N)",
"SEQUESTRATION_LANGUAGE_IND": "Sequestration Reductions (Y/N)",
"NPI_NAME": "NPI Name",
"DELEGATED_TERMS_IND": "Delegated Function Indicator",
"DELEGATED_TERMS": "Delegated Terms",
@@ -373,8 +386,6 @@ AC_MAPPING = {
"SINGLE_CODE_MULTIPLE_RATES_LANGUAGE": "Single Code Multiple Rates (Language)",
"INVOICE_PRICING_LANGUAGE_IND": "Invoice Pricing (Y/N)",
"INVOICE_PRICING_LANGUAGE": "Invoice Pricing (Language)",
"MEDICAL_NECESSITY_LANGUAGE_IND": "Medical Necessity Language (Y/N)",
"MEDICAL_NECESSITY_LANGUAGE": "Medical Necessity Language (Language)",
"TEMPLATE": "Template",
"PROVIDER_BASED_BILLING_EXCLUSION_LANGUAGE_IND": "Provider-based Billing Exclusion (Y/N)",
"PROVIDER_BASED_BILLING_EXCLUSION_LANGUAGE": "Provider-based Billing Exclusion (Language)",
@@ -396,6 +407,8 @@ ABC_COLUMNS = [
"Termination Date",
"Termination Upon Notice - Days",
"Termination With Cause - Days",
"Non-Renewal Language",
"Non-Renewal - Days",
"Amend Contract Upon notice Flag (Y/N)",
"Timeframe to Object - Days",
"Assignments Clause (Y/N)",
@@ -496,10 +509,7 @@ ABC_COLUMNS = [
"Medical Necessity Language (Language)",
"Template",
"Provider-based Billing Exclusion (Y/N)",
"Provider-based Billing Exclusion (Language)",
"Non-Renewal Language",
"Timely Filing",
"Non-Renewal - Days",
"Provider-based Billing Exclusion (Language)"
]
+459
View File
@@ -0,0 +1,459 @@
import pandas as pd
import numpy as np
import re
import difflib
import config
import prompts
import valid
from valid import DERIVED_INDICATOR_FIELDS
import claude_funcs
from utils import is_empty
import logging
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
def sanitize_value(value):
try:
if isinstance(value, list):
return ', '.join(str(v) for v in value)
elif isinstance(value, str):
value = value.strip('[]')
return ', '.join([item.strip(" '") for item in value.split(',')])
elif pd.isna(value):
return "N/A"
except:
return value
def exact_match(val, valid_values):
val = val.strip().upper()
for valid_val in valid_values:
if val == valid_val.upper():
return valid_val
return None
def get_closest_match(val, valid_values, similarity_threshold=0.7):
if pd.isna(val):
return None
val = val.strip().upper()
matches = difflib.get_close_matches(val, [v.upper() for v in valid_values], n=1, cutoff=similarity_threshold)
return matches[0] if matches else None
def correct_misplaced_values(df, columns, valid_values_dict):
for index, row in df.iterrows():
for col in columns:
if pd.notna(row[col]):
terms = row[col].split(',')
for term in terms:
term = term.strip()
for target_col, valid_values in valid_values_dict.items():
if target_col != col:
match = exact_match(term, valid_values)
if match:
if pd.isna(row[target_col]) or not row[target_col].strip():
df.at[index, target_col] = match
df.at[index, col] = None
else:
current_value = row[target_col].strip()
if get_closest_match(match, [current_value], config.FUZZY_MATCH_THRESHOLD) is None:
df.at[index, 'Corrected_' + target_col] = f"Found {term} in {col} cell"
df.at[index, col] = None
return df
def filter_service_column(answer_dicts):
filtered_list = []
for d in answer_dicts:
if 'FULL_SERVICE' not in d:
continue # Skip this dictionary if it doesn't have FULL_SERVICE
clean_dict = True
for keyword in valid.SERVICE_FILTER:
if keyword.upper() in d['FULL_SERVICE'].upper() or d['FULL_SERVICE'].upper() in keyword.upper():
clean_dict = False
break
if clean_dict:
filtered_list.append(d)
return filtered_list
def filter_methodology_column(answer_dicts):
filtered_list = []
for d in answer_dicts:
if 'FULL_METHODOLOGY' not in d:
continue # Skip this dictionary if it doesn't have FULL_METHODOLOGY
clean_dict = True
for keyword in valid.METHODOLOGY_FILTER:
if keyword.upper() in d['FULL_METHODOLOGY'].upper() or d['FULL_METHODOLOGY'].upper() in keyword.upper():
clean_dict = False
break
if clean_dict:
filtered_list.append(d)
return filtered_list
def clean_td(td):
td_clean = []
for d in td:
new_d = {}
for k, v in d.items():
if 'DATE' in k:
new_d[k] = v if isinstance(v, list) else [v]
elif k not in ['page_num', 'Filename']:
if isinstance(v, str) and ',' in v:
new_d[k] = [item.strip() for item in v.split(',')]
elif v == 'N/A':
new_d[k] = []
else:
new_d[k] = [v] if isinstance(v, str) else v
else:
new_d[k] = v
td_clean.append(new_d)
return td_clean
def get_parent_agreement_code(filename):
try:
filename = filename.split('.txt')[0]
match = re.search(r'([^\sa-zA-Z]+)(?=\.\w+$|$)', filename)
end = [i for i in match.group(1).split('_') if i]
return end[0]
except:
return 'N/A'
def consolidate_subheader(dict_list):
modified_list = []
for d in dict_list:
if 'FULL_SERVICE' in d and 'SUBHEADER' in d:
if d['SUBHEADER'] != 'N/A':
d['FULL_SERVICE'] = d['SUBHEADER'] + ' - ' + d['FULL_SERVICE']
# Remove the 'SUBHEADER' key
del d['SUBHEADER']
modified_list.append(d)
return modified_list
def clean_msr_lesser(df, ls=valid.VALID_MSR):
pattern = '|'.join(re.escape(item) for item in ls)
mask = df['FULL_SERVICE'].str.contains(pattern, case=False, na=False)
target_rows = df[mask]
# Iterate over these rows
for index, row in target_rows.iterrows():
# Get all rows with the same 'EXHIBIT' value
exhibit_rows = df[df['EXHIBIT'] == row['EXHIBIT']]
# Check the 'LESSER' values of these rows
if (exhibit_rows['LESSER'] == 'Y').any():
# If any row has 'LESSER' == 'Y', set the 'LESSER' value of the original row to 'Y'
df.at[index, 'LESSER'] = 'Y'
# Move first LESSER_RATE where LESSER==Y to MSR row
lesser_rate = exhibit_rows[exhibit_rows['LESSER'] == 'Y']['LESSER_RATE'].iloc[0]
df.at[index, 'LESSER_RATE'] = lesser_rate
return df
def clean_default_term(df, ls=valid.INVALID_DEFAULT):
# Create a case-insensitive regex pattern that matches any of the values in ls
pattern = '|'.join(re.escape(term) for term in ls)
# Find rows where the DEFAULT_TERM column contains any of the terms from ls
mask = df['DEFAULT_TERM'].str.contains(pattern, case=False, na=False)
# Replace these values with 'N/A'
df.loc[mask, 'DEFAULT_TERM'] = 'N/A'
df.loc[mask, 'DEFAULT_RATE'] = 'N/A'
return df
def clean_lesser_rate(df):
for index, row in df.iterrows():
if 'Y' in row['LESSER'] and is_empty(row['LESSER_RATE']) and not is_empty(row['RATE_STANDARD']):
df.at[index, 'LESSER_RATE'] = row['RATE_STANDARD']
df.at[index, 'RATE_STANDARD'] = 'N/A'
return df
def clean_prov_2(df):
valid_types = valid.select_valid_prov_2(df['PROV_TYPE'])
target_rows = df[df['PROV_TYPE_LEVEL_2'].apply(is_empty)]
def find_exact_match(text):
if pd.isna(text) or text == '':
return None
words = re.findall(r'\b[\w/]+(?:[-\s][\w/]+)*\b', text)
for i in range(len(words)):
for j in range(i+1, len(words)+1):
phrase = ' '.join(words[i:j])
if phrase in valid_types: # Case-sensitive matching
return phrase
return None
for index, row in target_rows.iterrows():
match = None
if not is_empty(row["FULL_SERVICE"]):
match = find_exact_match(str(row["FULL_SERVICE"]))
if match:
logging.info(f"Row {index}: Matched in FULL_SERVICE: {match}")
df.at[index, 'PROV_TYPE_LEVEL_2'] = match
continue
if not is_empty(row["EXHIBIT"]):
match = find_exact_match(str(row["EXHIBIT"]))
if match:
logging.info(f"Row {index}: Matched in EXHIBIT: {match}")
df.at[index, 'PROV_TYPE_LEVEL_2'] = match
continue
exhibit_rows = df[df['EXHIBIT'] == row['EXHIBIT']]
if not exhibit_rows.empty:
for _, exhibit_row in exhibit_rows.iterrows():
if not is_empty(exhibit_row['PROV_TYPE_LEVEL_2']):
match = find_exact_match(str(exhibit_row['PROV_TYPE_LEVEL_2']))
if match:
logging.info(f"Row {index}: Matched in other row's PROV_TYPE_LEVEL_2: {match}")
df.at[index, 'PROV_TYPE_LEVEL_2'] = match
break
elif not is_empty(exhibit_row['FULL_SERVICE']):
match = find_exact_match(str(exhibit_row['FULL_SERVICE']))
if match:
logging.info(f"Row {index}: Matched in other row's FULL_SERVICE: {match}")
df.at[index, 'PROV_TYPE_LEVEL_2'] = match
break
if match is None:
logging.warning(f"Row {index}: No valid match found")
# Final check to ensure no invalid values were assigned
invalid_assignments = df[
(~df['PROV_TYPE_LEVEL_2'].isin(valid_types)) &
(~df['PROV_TYPE_LEVEL_2'].apply(is_empty))
]
if not invalid_assignments.empty:
df.loc[invalid_assignments.index, 'PROV_TYPE_LEVEL_2'] = ''
return df
def clean_ac_fields(final_df):
# additional post-processing
final_df = final_df.fillna('')
# final_df.loc[final_df['CREDENTIALING_APP_IND'] != 'Yes', 'CREDENTIALING_APP_IND'] = "No"
# final_df.loc[final_df['AFFILIATION_CLAUSE_IND'] != 'Yes', 'AFFILIATION_CLAUSE_IND'] = "No"
# final_df.loc[final_df['ASSIGNMENTS_CLAUSE_IND'] != 'Yes', 'ASSIGNMENTS_CLAUSE_IND'] = "No"
if 'TERM_CLAUSE' in final_df:
final_df.loc[final_df['TERM_CLAUSE'].str.startswith('IL-4 Termination'), 'TERM_CLAUSE'] = "N/A"
final_df.loc[final_df['TERM_CLAUSE'].str.startswith('bonus payment shall be effective'), 'TERM_CLAUSE'] = "N/A"
final_df.loc[final_df['TERM_CLAUSE'].str.contains('does not contain'), 'TERM_CLAUSE'] = "N/A"
final_df.loc[final_df['TERM_CLAUSE'].str.contains('No term or termination'), 'TERM_CLAUSE'] = "N/A"
# if 'CONTRACT_SIGNATORY_IND' in final_df and 'PROV_PARTICIPATION_STATUS' in final_df:
# final_df.loc[final_df['CONTRACT_SIGNATORY_IND'] == 'Yes', 'PROV_PARTICIPATION_STATUS'] = "Yes"
if 'CONTRACT_AUTO_RENEWAL_IND' in final_df and 'CONTRACT_TERMINATION_DT' in final_df:
final_df.loc[final_df['CONTRACT_AUTO_RENEWAL_IND'] == 'Yes', 'CONTRACT_TERMINATION_DT'] = np.nan
# check if NPI has 10 digits
if 'PROV_GROUP_NPI' in final_df:
final_df['PROV_GROUP_NPI'] = final_df['PROV_GROUP_NPI'].map(lambda x: x if sum(c.isdigit() for c in x+' ') == 10 else '')
# NETWORK_ACCESS_FEES_IND
if 'NETWORK_ACCESS_FEES_IND' in final_df:
final_df.loc[~final_df['NETWORK_ACCESS_FEES_IND'].isin(['N/A', 'No', '', ' ']), 'NETWORK_ACCESS_FEES_IND'] = "Yes"
if 'PAYER_NAME' in final_df and 'HEALTH_PLAN_STATE' in final_df:
final_df.loc[final_df['PAYER_NAME'].str.startswith('Illini'), 'HEALTH_PLAN_STATE'] = "Illinois"
if 'HEALTH_PLAN_STATE' in final_df:
final_df['HEALTH_PLAN_STATE'] = final_df['HEALTH_PLAN_STATE'].map(valid.STATE_MAP).fillna(final_df['HEALTH_PLAN_STATE'])
if 'NOTICE_PROVIDER_NAME' in final_df and 'NOTICE_PROVIDER_ADDRESS' in final_df:
final_df.loc[final_df['NOTICE_PROVIDER_NAME'].str.contains('Superior HealthPlan', na=False), 'NOTICE_PROVIDER_NAME'] = np.nan
final_df.loc[final_df['NOTICE_PROVIDER_NAME'].str.contains('Superior HealthPlan', na=False), 'NOTICE_PROVIDER_ADDRESS'] = np.nan
final_df['temp_filename'] = final_df['Contract Name'].str[:10].str.replace('-','')
if 'PROV_GROUP_TIN' in final_df and 'Contract Name' in final_df:
final_df.loc[(final_df['PROV_GROUP_TIN'].isin([' ', '']))&(final_df['temp_filename'].str.isnumeric()), 'PROV_GROUP_TIN'] = final_df['temp_filename']
final_df.drop(columns=['temp_filename'], inplace=True)
# For POLICIES_AND_PROCEDURES, filter out anything without either "policies" or "procedures".
if 'POLICIES_AND_PROCEDURES' in final_df:
final_df.loc[~final_df['POLICIES_AND_PROCEDURES'].str.contains('policies', flags=re.IGNORECASE) &
~final_df['POLICIES_AND_PROCEDURES'].str.contains('procedures', flags=re.IGNORECASE), 'POLICIES_AND_PROCEDURES'] = "N/A"
final_df = final_df.apply(lambda x: x.map(replace_null_terms))
final_df = final_df.apply(lambda x: x.map(replace_quotes))
return final_df
def replace_quotes(value):
try:
return str(value).replace(r'\"', '"')
except:
return value
def replace_null_terms(value):
# Convert value to string and check if any term from NULL_ANSWER_TERMS is in the value
if any(term.lower() in str(value).lower() for term in valid.NULL_ANSWER_TERMS):
return 'N/A'
return value
def filter_dict(d, pattern):
final_dict = {}
for page_num, answer in d.items():
if not pattern.search(answer):
final_dict[page_num] = answer
return final_dict
def filter_add_ons(df):
pattern = r"(in no event).+(includ.?)|(forward).+(payments)"
mask = df['ADD_ON_REIMBURSEMENT_LANGUAGE'].str.contains(pattern, case=False, na=False, regex=True)
df.loc[mask, 'ADD_ON_REIMBURSEMENT_LANGUAGE'] = 'N/A'
df.loc[mask, "ADD_ON_REIMBURSEMENT_IND"] = 'N'
def check_add_on(group):
# Check if any 'SERVICE_TYPE' contains 'Add-on' or 'Add On'
if group['FULL_SERVICE'].str.contains('Add-on|Add On', regex=True, case=False, na=False).any():
group['ADD_ON_REIMBURSEMENT_LANGUAGE'] = 'N/A' # Set 'ADD_ON' to 'N/A' for the whole group
group["ADD_ON_REIMBURSEMENT_IND"] = 'N'
return group
# Group by 'Contract Name' and 'EXHIBIT', then apply the check_add_on function
df = df.groupby(['Contract Name', 'EXHIBIT']).apply(check_add_on)
# TODO : If original add on value is actually an Exclusion, and the Exclusion value for the row is invalid, then move the Add On value to the Exclusions column
return df
import pandas as pd
import re
def clean_process_AWP_FLATFEE_cols(df):
# Function to extract dollar amount after "NET INVOICE PRICE" or "FLAT FEE"
def extract_dollar_amount(text, pattern):
if pd.isna(text):
return None
match = re.search(pattern, str(text), re.IGNORECASE)
if match:
return match.group(1).replace(',', '')
return None
# Apply changes based on conditions
def apply_changes(row):
if 'FULL_METHODOLOGY' in row and isinstance(row['FULL_METHODOLOGY'], str) and 'AWP' in row['FULL_METHODOLOGY'].upper():
full_methodology = row['FULL_METHODOLOGY'].upper()
dollar_amount_found = False
# Check FLAT_FEE_STANDARD first
if 'FLAT_FEE_STANDARD' in row and row['FLAT_FEE_STANDARD'] != 'N/A' and pd.notna(row['FLAT_FEE_STANDARD']) and row['FLAT_FEE_STANDARD'] != '':
row['SHORT_METHODOLOGY'] = 'Flat Fee'
dollar_amount_found = True
else:
# Then check for FLAT FEE in FULL_METHODOLOGY
flat_fee = extract_dollar_amount(full_methodology, r'FLAT FEE\s*:\s*\$?([\d,]+(?:\.\d{2})?)')
if flat_fee:
row['FLAT_FEE_STANDARD'] = flat_fee
row['SHORT_METHODOLOGY'] = 'Flat Fee'
dollar_amount_found = True
else:
# Finally, check for NET INVOICE PRICE
net_invoice_price = extract_dollar_amount(full_methodology, r'NET INVOICE PRICE\s*:\s*\$?([\d,]+(?:\.\d{2})?)')
if net_invoice_price:
row['FLAT_FEE_STANDARD'] = net_invoice_price
row['SHORT_METHODOLOGY'] = 'Flat Fee'
dollar_amount_found = True
# Set RATE_STANDARD and RATE_SHORT to 'N/A' if a dollar amount was found
if dollar_amount_found:
row['RATE_STANDARD'] = 'N/A'
row['RATE_SHORT'] = 'N/A'
return row
# Apply the changes to the DataFrame
df = df.apply(apply_changes, axis=1)
return df
def add_scmr(df):
def count_dollar_values(s):
return str(s).count('$')
df['dollar_count'] = df['FLAT_FEE_STANDARD'].apply(count_dollar_values)
df['Single Code Multiple Rates (Language)'] = df.apply(
lambda row: row['FLAT_FEE_STANDARD'] if row['dollar_count'] > 1 else 'N/A', axis=1)
df['Single Code Multiple Rates (Y/N)'] = df['dollar_count'].apply(
lambda x: 'Y' if x > 1 else 'N')
df.drop('dollar_count', axis=1, inplace=True)
return df
def clean_lob(df, filename):
def update_contract_lob(row):
# Check if more than one valid lob is present in the 'FULL_SERVICE' column
if sum(val.lower() in row['FULL_SERVICE'].lower() for val in valid.VALID_LOBS) > 1:
prompt = prompts.LOB_SWEEPER(row['EXHIBIT'])
answer = claude_funcs.invoke_claude(prompt, config.MODEL_ID_CLAUDE35_SONNET, filename, 256)
return answer # Return the value from 'EXHIBIT' if condition is met
return row['CONTRACT_LOB']
df['CONTRACT_LOB'] = df.apply(update_contract_lob, axis=1)
return df
def derive_indicators(results):
def is_populated(value):
if not value: # Handles None and empty string
return False
# Convert to string and lowercase for comparison
str_value = str(value).lower().strip()
# Check if the value is any variation of N/A
na_values = ['n/a', 'na', 'not applicable', '']
if str_value in na_values:
return False
return True
# Process standard derived indicator fields
for field in DERIVED_INDICATOR_FIELDS:
root_field = field
indicator_field = field.rsplit('_', 1)[0] + '_IND'
if root_field in results and is_populated(results[root_field]):
results[indicator_field] = 'Y'
else:
results[indicator_field] = 'N'
# Handle special cases
special_mappings = {
'DELEGATED_TERMS': 'DELEGATED_FUNCTION_IND',
'SEQUESTRATION_REDUCTIONS': 'SEQUESTRATION_REDUCTIONS_IND'
}
for root_field, indicator_field in special_mappings.items():
if root_field in results and is_populated(results[root_field]):
results[indicator_field] = 'Y'
else:
results[indicator_field] = 'N'
return results