Merged in feature/pc (pull request #773)

Feature/pc

* file read update

* removed unused code

* replace pring with logging.info

* input args for docczy output

* input args for docczy output

* added json for loading mappings

* exception for bool keys

* exception for bool keys

* added init.py

* write to s3/local functionality

* write to s3/local functionality

* logging errors and main functionality fix

* pc functionality from main

* pc functionality from main

* function signature fix

* fixed path returns

* Updated names for varaibles; Modified termination date cleaning function

* Removed commented out function

* Renamed client column names to SAAS column names

* Merge remote-tracking branch 'origin/main' into feature/pc

* removed pc utils

* black formatting

* rename function

* rename func

* additional file removed


Approved-by: Katon Minhas
This commit is contained in:
Karan Desai
2025-11-17 14:16:39 +00:00
committed by Katon Minhas
parent ff400f4b1c
commit 5887cf651e
9 changed files with 1632 additions and 17 deletions
@@ -0,0 +1,41 @@
{
"True": [
"exhibit",
"amendment",
"addendum",
"expirits",
"instructions",
"advance of payment",
"rate agreement",
"assignment of agreement",
"hedis",
"attachment",
"request",
"disclosure",
"exhibits",
"non-disclosure agreement",
"assumption agreement",
"rate letter",
"insurance",
"letter",
"letter intent",
"application",
"certification",
"notification",
"amd",
"explanation",
"meeting",
"minutes",
"agenda",
"summary notes",
"tracker",
"settlement"
],
"False": [
"exhibit",
"amendment",
"addendum",
"attachment"
]
}
@@ -0,0 +1,59 @@
{
"ordinal_regex_pattern": "\\b(?:first|second|third|fourth|fifth|sixth|seventh|eighth|ninth|tenth|eleventh|twelfth|thirteenth|fourteenth|fifteenth|sixteenth|seventeenth|eighteenth|nineteenth|twentieth|twenty-first|twenty-second|twenty-third|twenty-fourth|twenty-fifth|\\d{1,2}(?:st|nd|rd|th))\\b",
"ordinal_word_to_number": {
"first": 1,
"second": 2,
"third": 3,
"fourth": 4,
"fifth": 5,
"sixth": 6,
"seventh": 7,
"eighth": 8,
"ninth": 9,
"tenth": 10,
"eleventh": 11,
"twelfth": 12,
"thirteenth": 13,
"fourteenth": 14,
"fifteenth": 15,
"sixteenth": 16,
"seventeenth": 17,
"eighteenth": 18,
"nineteenth": 19,
"twentieth": 20,
"twenty-first": 21,
"twenty-second": 22,
"twenty-third": 23,
"twenty-fourth": 24,
"twenty-fifth": 25
},
"cardinal_word_to_number": {
"one": 1,
"two": 2,
"three": 3,
"four": 4,
"five": 5,
"six": 6,
"seven": 7,
"eight": 8,
"nine": 9,
"ten": 10,
"eleven": 11,
"twelve": 12,
"thirteen": 13,
"fourteen": 14,
"fifteen": 15,
"sixteen": 16,
"seventeen": 17,
"eighteen": 18,
"nineteen": 19,
"twenty": 20,
"twenty one": 21,
"twenty two": 22,
"twenty three": 23,
"twenty four": 24,
"twenty five": 25
}
}
+5 -1
View File
@@ -311,10 +311,14 @@ DOCZY_PDF_FILES_BUCKET_S3_URL = f"https://{DOCZY_PDF_FILES_BUCKET_NAME}.s3.us-ea
############## VISION API SETTINGS ##############
ENABLE_VISION = get_arg_value("enable_vision", "False") == "True" # False by default
############## PARENT CHILD CONFIG ##############
PERFORM_PARENT_CHILD_MAPPING = get_arg_value("perform_pc", "False") == "True"
DOCZY_OUTPUT_FOR_PC = get_arg_value("doczy_output_for_pc", "") # Pass either S3 URI path, or local path
WRITE_PC_TO_S3 = get_arg_value("write_pc_to_s3", "False") == "True"
############## DOCUMENT TYPE CLASSIFICATION SETTINGS ##############
PERFORM_DTC = get_arg_value("perform_dtc", "False") == "True" # Enable document type classification pre-filter
DTC_PROMPTS_JSON_PATH = "src/prompts/document_classification_prompts.json"
DTC_MAX_PAGES_TO_CHECK = 5 # Number of pages to check per document for contract detection
DTC_OUTPUT_FILE = f"{BATCH_ID}_Document_Classification_Report.csv" # Output CSV file path
DTC_JSON_OUTPUT_FOLDER = "dtc_json_results" # Folder for individual JSON results
DTC_JSON_OUTPUT_FOLDER = "dtc_json_results" # Folder for individual JSON results
+13 -9
View File
@@ -17,6 +17,7 @@ import src.utils.logging_utils as logging_utils
import src.utils.usage_tracking as usage_tracking
from constants.constants import Constants
from src import config
from src.parent_child import main as parent_child_main
from src.document_classification import main as dtc_main
def safe_process_file(item, constants, run_timestamp):
@@ -173,9 +174,9 @@ def main(testing=False, test_params={}):
)
if config.WRITE_TO_S3:
io_utils.write_s3(FINAL_RESULT_DF, "", run_timestamp, "final")
doczy_output_for_pc = io_utils.write_s3(FINAL_RESULT_DF, "", run_timestamp, "final")
if not ERROR_RESULT_DF.empty:
io_utils.write_s3(ERROR_RESULT_DF, "", run_timestamp, "error")
doczy_output_for_pc = io_utils.write_s3(ERROR_RESULT_DF, "", run_timestamp, "error")
# Export usage and cost tracking data to S3 only if both dataframes have data
if not per_file_usage_df.empty and not batch_summary_df.empty:
@@ -183,15 +184,18 @@ def main(testing=False, test_params={}):
io_utils.write_s3(per_file_usage_df, "", run_timestamp, "usage")
io_utils.write_s3(batch_summary_df, "", run_timestamp, "usage_summary")
else:
io_utils.write_local(FINAL_RESULT_DF, "", run_timestamp, "final")
doczy_output_for_pc = io_utils.write_local(FINAL_RESULT_DF, "", run_timestamp, "final")
if not ERROR_RESULT_DF.empty:
io_utils.write_local(ERROR_RESULT_DF, "", run_timestamp, "error")
doczy_output_for_pc = io_utils.write_local(ERROR_RESULT_DF, "", run_timestamp, "error")
if config.PERFORM_PARENT_CHILD_MAPPING:
parent_child_main.main(doczy_output_for_pc)
# Export usage and cost tracking data locally only if both dataframes have data
if not per_file_usage_df.empty and not batch_summary_df.empty:
logging.info("Exporting usage and cost tracking data locally...")
io_utils.write_local(per_file_usage_df, "", run_timestamp, "usage")
io_utils.write_local(batch_summary_df, "", run_timestamp, "usage_summary")
# Export usage and cost tracking data locally only if both dataframes have data
if not per_file_usage_df.empty and not batch_summary_df.empty:
logging.info("Exporting usage and cost tracking data locally...")
io_utils.write_local(per_file_usage_df, "", run_timestamp, "usage")
io_utils.write_local(batch_summary_df, "", run_timestamp, "usage_summary")
else:
return FINAL_RESULT_DF, ERROR_RESULT_DF
+56
View File
@@ -0,0 +1,56 @@
import src.config as config
# from src.parent_child.file_reading_utils import read_file_from_path
from src.parent_child.parent_child_preprocessing import parent_child_preprocessing
from src.parent_child.parent_child_mapping import parent_child_mapping
import os
from src.utils import io_utils
import logging
from datetime import datetime
def main(doczy_output_for_pc):
# Generate timestamp for this run
run_timestamp = f"run_{datetime.now().strftime('%Y%m%d_%H-%M')}_{config.BATCH_ID}"
try:
df_read = io_utils.read_DataFrame(doczy_output_for_pc)
logging.info(f"original shape: {df_read.shape}")
cleaned_df = parent_child_preprocessing(df_read)
logging.info(
f"Unique contracts in the output: {cleaned_df["FILE_NAME"].nunique()}"
)
mapped_df = parent_child_mapping(cleaned_df)
# Save detailed results using same pattern as investment pipeline
if config.WRITE_PC_TO_S3:
io_utils.write_s3(
mapped_df, "Parent_Child_Mapping_", run_timestamp, "parent_child"
)
logging.info(
f"PC results uploaded to S3: {config.BATCH_ID}/{run_timestamp}/{config.BATCH_ID}-PC.csv"
)
else:
io_utils.write_local(
mapped_df, "Parent_Child_Mapping_", run_timestamp, "parent_child"
)
logging.info(
f"PC results saved locally: {config.CONSOLIDATED_OUTPUT_DIRECTORY}/{run_timestamp}/{config.BATCH_ID}-PC.csv"
)
return mapped_df.shape[0]
except Exception as e:
logging.info(f"Parent child mapping failed due to {e}")
if __name__ == "__main__":
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
)
# Generate timestamp for this run
run_timestamp = f"run_{datetime.now().strftime('%Y%m%d_%H-%M')}_{config.BATCH_ID}"
if config.DOCZY_OUTPUT_FOR_PC:
results = main(config.DOCZY_OUTPUT_FOR_PC)
logging.info(f"PC Mapping complete. Processed {results} files.")
@@ -0,0 +1,890 @@
import warnings
warnings.filterwarnings("ignore")
import pandas as pd
import re
from src.parent_child.parent_child_preprocessing import (
clean_effective_date,
clean_termination_date,
)
import numpy as np
import logging
from src.utils import io_utils
def create_df_per_tin_presence(df, is_tin_present):
"""
Filters the DataFrame based on the presence or absence of TIN.
Parameters:
----------
df : input DataFrame containing TIN and ICMP-related columns.
is_tin_present : bool
If True, returns rows where 'TIN' is present (not NaN).
If False, returns rows where 'TIN' is missing and 'icmp_part' is present.
Returns:
-------
pandas.DataFrame
Filtered DataFrame according to the TIN presence logic.
"""
if is_tin_present:
return df[~df["TIN"].isna()]
else:
return df[df["TIN"].isna() & df["icmp_part"].notna()]
def filter_on_exclude_keywords(df, is_tin_present):
"""
Flags rows as 'parent' based on exclusion keywords.
Parameters:
----------
df : input DataFrame containing contract information.
is_tin_present : bool
Flag used to select the appropriate keyword list from `exclude_keywords_dict`.
Returns:
-------
updated DataFrame with the 'parent' column set to False
if either 'contract_title_cleaned' or 'fixed_contract_name'
contains any of the exclusion keywords (case-insensitive).
"""
df["parent"] = True
exclude_keywords_dict = io_utils.convert_json_to_dict(
"constants/mappings/exclude_keywords_for_pc.json"
)
exclude_keywords_list = exclude_keywords_dict[is_tin_present]
exclude_pattern = (
r"\b(?:" + "|".join(re.escape(kw) for kw in exclude_keywords_list) + r")\b"
)
df["parent"] = ~(
df["contract_title_cleaned"].str.contains(exclude_pattern, case=False, na=False)
| df["contract_name_cleaned"].str.contains(
exclude_pattern, case=False, na=False
)
)
return df
def filter_on_missing_title(df):
"""
Updates 'parent' flag based on missing or empty contract titles and keywords in FILE_NAME.
Rules:
1. If 'contract_title_cleaned' is NaN or empty, default 'parent' to False.
2. Exception: if 'contract_title_cleaned' is null and 'agmt' in 'contract_name', set parent = True.
3. Exception to exception: if 'contract_title_cleaned' is null and both 'agmt' and 'letter' in 'contract_name', set parent = False.
"""
df = df.copy()
# Normalize text columns
contract_name_lower = df["contract_name_cleaned"].astype(str).str.lower()
title_missing = df["contract_title_cleaned"].isna() | df[
"contract_title_cleaned"
].astype(str).str.strip().isin(["", "nan"])
# Base: set parent False where title is missing
df["parent"] = df["parent"].where(~title_missing, False)
# Compute final parent mask in one step
# Parent = True if missing title & 'agmt' in name, except if also 'letter' in name
df.loc[
title_missing
& contract_name_lower.str.contains("agmt")
& ~contract_name_lower.str.contains("letter"),
"parent",
] = True
return df
def flag_duplicates_within_group(df, grouper_col):
"""
Flags duplicates within each group for rows where parent == True,
based on an exact match of:
- 'contract_title_cleaned'
- 'fixed_effective_date'
- 'payer_name_cleaned'
NaN values in 'payer_name_cleaned' are treated as equal when both are NaN,
and as different when only one is NaN.
Non-parent rows are returned unchanged.
"""
df = df.copy()
df["is_duplicate"] = False # default for all rows
# Work only on parent rows
mask = df["parent"] == True
df_parent = df.loc[mask].copy()
# Handle NaNs consistently
df_parent["_payer"] = df_parent["payer_name_cleaned"].fillna("__NULL__")
# Build duplicate key
df_parent["_dup_key"] = (
df_parent["contract_title_cleaned"].astype(str)
+ "||"
+ df_parent["fixed_effective_date"].astype(str)
)
# Detect duplicates
df_parent["is_duplicate"] = df_parent.groupby(grouper_col)["_dup_key"].transform(
lambda x: x.duplicated(keep=False)
)
# Update only parent rows in original df
df.loc[mask, "is_duplicate"] = df_parent["is_duplicate"]
# Clean temp cols (only in df_parent, since they werent in df originally)
return df
def rank_parents(df, grouper_col):
"""
Assigns a rank to 'parent' contracts within each group defined by `grouper_col`.
Ranking logic:
1. Only rows where `parent == True` are ranked.
2. Ranking is based on earliest 'fixed_effective_date' first (NaT placed last).
3. Each parent gets its own rank (duplicates are not merged).
4. Null dates are ranked last.
"""
df = df.copy()
df["parent_rank"] = np.nan # Initialize
# Ensure effective_date is datetime
df["_sort_date"] = pd.to_datetime(df["fixed_effective_date"], errors="coerce")
for group_val, group_df in df.groupby(grouper_col):
parents = group_df[group_df["parent"] == True].copy()
if parents.empty:
continue
# Replace NaT with a large future date for sorting → ensures null dates go last
parents["_rank_date"] = parents["_sort_date"].fillna(pd.Timestamp.max)
# Sort and assign unique ranks
parents = parents.sort_values("_rank_date")
ranks = range(1, len(parents) + 1)
df.loc[parents.index, "parent_rank"] = list(ranks)
return df.drop(columns=["_sort_date", "_rank_date"], errors="ignore")
def flag_multi_parent_with_early_dates(
df: pd.DataFrame, grouper_col: str
) -> pd.DataFrame:
"""
Flags groups where:
1. There are multiple unique parent ranks.
2. The group has children.
3. At least two parents have effective dates earlier than at least one child.
Parameters
----------
df : pd.DataFrame
The DataFrame containing at least:
- 'parent' (bool)
- 'parent_rank'
- 'fixed_effective_date'
- Grouping column specified by `grouper_col`
grouper_col : str
The column name to group by (e.g., 'TIN').
Returns
-------
pd.DataFrame
The input DataFrame with an added column:
'flag_multi_parent_with_early_dates' containing either the flag value
'multiple_parents_early_children' or NaN.
"""
df = df.copy()
df["fixed_effective_date"] = pd.to_datetime(
df["fixed_effective_date"], errors="coerce"
)
flag_dict = {}
for group_val, group_df in df.groupby(grouper_col):
parents_df = group_df[group_df["parent"] == True]
children_df = group_df[group_df["parent"] == False]
# Skip if less than 2 unique parents or no children
unique_parents = parents_df["parent_rank"].dropna().unique()
if len(unique_parents) <= 1 or children_df.empty:
continue
# Get all child dates
child_dates = children_df["fixed_effective_date"].dropna()
if child_dates.empty:
continue
# Count parents with earlier dates than any child
count_early_parents = sum(
pd.notna(parent_row["fixed_effective_date"])
and (child_dates > parent_row["fixed_effective_date"]).any()
for _, parent_row in parents_df.iterrows()
)
if count_early_parents >= 2:
flag_dict[group_val] = "multiple_parents_early_children"
# Map flags back to DataFrame
df["flag_multi_parent_with_early_dates"] = df[grouper_col].map(flag_dict)
return df
def assign_parents_with_state_and_payer(
df, grouper_col, flag_col="flag_multi_parent_with_early_dates"
):
"""
Assigns each child contract to a parent contract with strict TIN, state, payer, and date logic.
"""
df = df.copy()
df["fixed_effective_date"] = pd.to_datetime(
df["fixed_effective_date"], errors="coerce"
)
parents = df[df["parent"]].copy()
children = df[~df["parent"]].copy()
assignments = []
for idx, child in children.iterrows():
# Step 1: Filter parents within TIN group
group_parents = parents[parents[grouper_col] == child[grouper_col]].copy()
if group_parents.empty:
assignments.append(("no_parent", idx))
continue
# Step 2: Filter by matching state
state_parents = group_parents[
group_parents["PAYER_STATE"] == child["PAYER_STATE"]
]
if state_parents.empty:
assignments.append(("no_parent", idx))
continue
# Step 3: Apply effective date constraint (parent date <= child date)
if pd.notna(child["fixed_effective_date"]):
state_parents = state_parents[
pd.to_datetime(state_parents["fixed_effective_date"], errors="coerce")
<= child["fixed_effective_date"]
]
if state_parents.empty:
assignments.append(("no_parent", idx))
continue
# Step 4: Tie-breaker on payer_name
if len(state_parents) > 1:
if pd.notna(child.get("payer_name_cleaned")):
payer_match = state_parents[
state_parents["payer_name_cleaned"] == child["payer_name_cleaned"]
]
if len(payer_match) == 1:
assignments.append((payer_match.iloc[0]["parent_rank"], idx))
continue
elif len(payer_match) > 1:
# Still multiple → pick latest
latest_parent = payer_match.sort_values(
"parent_rank", ascending=False
).iloc[0]
assignments.append((latest_parent["parent_rank"], idx))
continue
else:
# No matching payer → no parent
assignments.append(("no_parent", idx))
continue
else:
# Child payer_name null → no parent
assignments.append(("no_parent", idx))
continue
else:
# Only one parent left → assign
assignments.append((state_parents.iloc[0]["parent_rank"], idx))
children["assigned_parent_rank"] = None
for rank, idx in assignments:
children.at[idx, "assigned_parent_rank"] = rank
return pd.concat([parents, children]).sort_index()
def assign_no_parent_by_title(
df, grouper_col, contract_title_col="contract_title_cleaned"
):
"""
Second-pass assignment for children with 'no_parent' in assigned_parent_rank.
Assigns based on contract title substring match, with same restrictions:
- Must match on grouper_col (e.g., TIN).
- Parent's contract title must be substring of child's title (case-insensitive).
- Parent's effective date must be <= child's date.
- Pick latest parent_rank if multiple.
Parameters
----------
df : pd.DataFrame
Must contain:
- 'assigned_parent_rank'
- 'parent' (bool)
- 'parent_rank'
- 'fixed_effective_date'
- contract_title_col (str)
- grouper_col (str)
grouper_col : str
Column name for grouping (e.g., 'TIN').
contract_title_col : str
Column name containing the contract title.
Returns
-------
pd.DataFrame
Updated DataFrame with some 'no_parent' children now assigned.
"""
# Validate input DataFrame is not empty
if df.empty:
logging.info("Warning: Input DataFrame is empty")
return df.copy()
# Validate required columns exist
required_cols = [
"assigned_parent_rank",
"parent",
"parent_rank",
"fixed_effective_date",
contract_title_col,
grouper_col,
]
missing_cols = [col for col in required_cols if col not in df.columns]
if missing_cols:
logging.info(f"Available columns: {list(df.columns)}")
raise KeyError(f"Missing required columns: {missing_cols}")
logging.info(f"Processing {len(df)} rows")
logging.info(f"Using columns - grouper: {grouper_col}, title: {contract_title_col}")
# Create a copy and ensure we're working with the right data types
df_work = df.copy()
# Convert date column with explicit error handling
try:
df_work["fixed_effective_date"] = pd.to_datetime(
df_work["fixed_effective_date"], errors="coerce"
)
except Exception as e:
logging.info(f"Error converting dates: {e}")
return df.copy()
# Filter parents and children with no parent
parents = df_work[df_work["parent"] == True].copy()
children_no_parent = df_work[df_work["assigned_parent_rank"] == "no_parent"].copy()
logging.info(
f"Found {len(parents)} parents and {len(children_no_parent)} children with no_parent"
)
if len(parents) == 0 or len(children_no_parent) == 0:
logging.info("No work to do - either no parents or no unassigned children")
return df.copy()
assignments_made = 0
# Process each child that needs a parent
for idx, child in children_no_parent.iterrows():
# Find potential parents in same group
try:
possible_parents = parents[
parents[grouper_col] == child[grouper_col]
].copy()
except KeyError as e:
logging.info(f"Error accessing grouper column '{grouper_col}': {e}")
continue
if len(possible_parents) == 0:
continue
child_date = child["fixed_effective_date"]
if pd.isna(child_date):
continue # skip if child's date missing
child_title = child[contract_title_col]
if not isinstance(child_title, str):
continue # skip if child title is not a string
child_title_lower = child_title.lower()
# Find parents whose title is substring of child's title (case-insensitive)
def title_match(parent_title):
if not isinstance(parent_title, str):
return False
return parent_title.lower() in child_title_lower
title_matches = possible_parents[
possible_parents[contract_title_col].apply(title_match)
]
if len(title_matches) == 0:
continue
# Filter by effective date: parent date <= child date
eligible = title_matches[title_matches["fixed_effective_date"] <= child_date]
if not eligible.empty:
# Choose parent with latest parent_rank
chosen_parent_rank = eligible.sort_values(
"parent_rank", ascending=False
).iloc[0]["parent_rank"]
df_work.at[idx, "assigned_parent_rank"] = chosen_parent_rank
assignments_made += 1
logging.info(f"Made {assignments_made} new parent assignments")
return df_work
def assign_child_ranks(df, grouper_col):
"""
Assigns sequential child ranks to non-parent contracts.
Ranking logic:
- Parent rank assigned earlier is used as the base.
- Orphans (assigned_parent_rank == 'no_parent') start with base rank "0".
- Children are ordered by 'fixed_effective_date' (null dates last).
- No deduplication: every child gets a unique sequential rank.
- The child rank is always a string (e.g., "1.1", "1.2", "0.1").
Returns
-------
pd.DataFrame
DataFrame with added 'child_rank' and 'combined_rank' columns (as string).
"""
df = df.copy()
# Separate parents and children
parents = df[df["parent"]].copy()
children = df[~df["parent"]].copy()
# Assign children to parents (and orphans)
for (grp_val, parent_rank), grp_df in children.groupby(
[grouper_col, "assigned_parent_rank"]
):
if parent_rank == "no_parent": # orphan group
base_rank = "0"
else:
base_rank = str(parent_rank)
# Sort by date (null dates last)
grp_df = grp_df.assign(
sort_date=grp_df["fixed_effective_date"].fillna(pd.Timestamp.max)
).sort_values("sort_date")
# Sequential rank without deduplication
for i, idx in enumerate(grp_df.index, start=1):
rank_val = f"{base_rank}.{i}"
children.loc[idx, "child_rank"] = rank_val
logging.info(children.columns)
# Combine back
result = pd.concat([parents, children]).sort_index()
# Make sure child_rank is string
result["child_rank"] = result["child_rank"].astype(str)
# If parent rank is "no_parent", fall back to child rank
def pick_rank(row):
if row.get("assigned_parent_rank") == "no_parent":
return row["child_rank"]
return (
str(row["parent_rank"])
if pd.notna(row["parent_rank"])
else row["child_rank"]
)
result["combined_rank"] = result.apply(pick_rank, axis=1)
return result
def assign_orphans_based_on_unique_parent_and_payer(
df,
grouper_col="TIN",
payer_col="payer_name_cleaned",
assigned_parent_rank_col="assigned_parent_rank",
):
"""
Assigns orphans to their unique parent when:
- There's only one unique payer in the TIN group.
- There's exactly one unique parent in the TIN group (based on 'parent_rank').
- The child has 'no_parent' in 'assigned_parent_rank'.
- There is at least one parent in the TIN group.
- The parents effective date is older than or equal to the childs effective date, or the childs effective date is missing.
- If there are multiple unique parents and exactly one unique payer, assign to the latest eligible parent (i.e., the lowest `parent_rank`).
Parameters
----------
df : pd.DataFrame
The input DataFrame.
grouper_col : str
Column name for the grouping (e.g., 'TIN').
payer_col : str
Column name for the PAYER_NAME (e.g., 'payer_name_cleaned').
assigned_parent_rank_col : str
Column for the assigned parent rank (e.g., 'assigned_parent_rank').
Returns
-------
pd.DataFrame
Updated DataFrame with some orphans assigned to their unique parent.
"""
# Ensure 'assigned_parent_rank' is initialized
if assigned_parent_rank_col not in df.columns:
raise KeyError(f"'{assigned_parent_rank_col}' column is missing.")
# Step 1: Filter TIN groups where there are orphans ('no_parent') and at least one parent exists
df_work = df.copy()
# Filter TIN groups where:
# - There are orphans (i.e., 'no_parent' in 'assigned_parent_rank')
# - At least one parent exists in the group
tin_present = df_work.groupby(grouper_col).filter(
lambda group: (group[assigned_parent_rank_col] == "no_parent").any()
and (group["parent"] == True).any()
)
if tin_present.empty:
logging.info("No TIN groups meet the filtering criteria.")
return df_work
# Step 2: Identify orphans (children with 'no_parent')
orphan_groups = tin_present[tin_present[assigned_parent_rank_col] == "no_parent"]
if orphan_groups.empty:
logging.info("No orphans to assign.")
return df_work
# Step 3: Iterate over TIN groups to assign orphans
assignments_made = 0
for tin_group, group_df in orphan_groups.groupby(grouper_col):
# Find the parents in this TIN group
possible_parents = tin_present[
(tin_present[grouper_col] == tin_group) & (tin_present["parent"] == True)
]
# Step 4: Check for unique parent ranks (i.e., exactly one unique parent in the TIN group)
unique_parent_ranks = possible_parents["parent_rank"].nunique()
# Step 5: Check for unique PAYER_NAMEs (including NaN values)
unique_payer_names = possible_parents[payer_col].nunique(dropna=False)
# Debug: Check unique parent ranks and PAYER_NAMEs in the group
# logging.info(f"Processing TIN group {tin_group}:")
# logging.info(f"Unique parent ranks: {unique_parent_ranks}")
# logging.info(f"Unique PAYER_NAMEs (including NaN): {unique_payer_names}")
# Step 6: If there is more than one unique payer, do not assign any parent
if unique_payer_names > 1:
# logging.info(f"More than one unique payer in TIN group {tin_group}. Skipping assignment.")
continue
# Step 7: If there's exactly one unique parent and one unique payer, assign to that parent
if unique_parent_ranks == 1:
parent = possible_parents.iloc[0]
parent_rank = parent["parent_rank"]
parent_effective_date = pd.to_datetime(
parent["fixed_effective_date"], errors="coerce"
)
# Step 8: Assign the single parent to the orphans in this TIN group
for _, orphan in group_df.iterrows():
child_effective_date = pd.to_datetime(
orphan["fixed_effective_date"], errors="coerce"
)
# Check if the child's effective date is older than the parent's
if (
pd.isna(child_effective_date)
or parent_effective_date <= child_effective_date
):
df_work.at[orphan.name, assigned_parent_rank_col] = parent_rank
assignments_made += 1
# logging.info(f"Assigned parent rank {parent_rank} to orphan {orphan.name}")
# Step 9: If there's exactly one unique payer and more than one unique parent, assign to the latest parent (lowest rank)
elif unique_parent_ranks > 1:
# logging.info(f"Multiple unique parents in TIN group {tin_group}. Assigning to the latest parent (lowest rank).")
# Sort parents by parent_rank (which is based on effective date)
sorted_parents = possible_parents.sort_values(
"parent_rank", ascending=False
) # Lower rank is the latest parent
# Assign the orphan to the latest eligible parent
for _, orphan in group_df.iterrows():
child_effective_date = pd.to_datetime(
orphan["fixed_effective_date"], errors="coerce"
)
# Find the first parent who meets the effective date condition
for _, parent in sorted_parents.iterrows():
parent_effective_date = pd.to_datetime(
parent["fixed_effective_date"], errors="coerce"
)
if (
pd.isna(child_effective_date)
or parent_effective_date <= child_effective_date
):
df_work.at[orphan.name, assigned_parent_rank_col] = parent[
"parent_rank"
]
assignments_made += 1
# logging.info(f"Assigned parent rank {parent['parent_rank']} to orphan {orphan.name}")
break
return df_work
def assign_parent_names(df, rank_col="combined_rank", name_col="CONTRACT_NAME"):
"""
Adds a 'parent_name' column mapping each child to its parent's name.
- Parents are detected as rows whose rank has only one number part (e.g. '1', '1.0', '001')
- Children starting with '0.' get no parent (None)
- Works with different numeric formats
Parameters
----------
df : pd.DataFrame
Input dataframe
rank_col : str
Column containing rank (e.g. 'combined_rank')
name_col : str
Column containing name (e.g. 'CONTRACT_NAME')
Returns
-------
pd.DataFrame
Dataframe with 'parent_name' column added
"""
df = df.copy()
df[rank_col] = df[rank_col].astype(str).str.strip()
# Detect parents: only one part when split by dot, or dot followed by zeros
parents_df = df[df[rank_col].str.match(r"^\d+(\.0+)?$", na=False)].copy()
# Normalize parent rank to plain integer string (e.g. '1.0' -> '1')
parents_df["parent_key"] = (
parents_df[rank_col].str.split(".").str[0].astype(int).astype(str)
)
# Create mapping parent_key -> parent name
parent_map = parents_df.set_index("parent_key")[name_col].to_dict()
# Function to get parent name
def get_parent_name(rank):
rank = str(rank).strip()
if rank.startswith("0.") or rank in parent_map:
return None
parent_key = rank.split(".")[0]
return parent_map.get(parent_key)
df["parent_name"] = df[rank_col].apply(get_parent_name)
return df
def extract_ordinal(row):
num_dict = io_utils.convert_json_to_dict("constants/mappings/numeric_mappigs.json")
word_to_num = num_dict.get("ordinal_word_to_number", {})
cardinal_to_num = num_dict.get("cardinal_word_to_number", {})
ordinal_pattern = num_dict.get("ordinal_regex_pattern", "")
# Combine all text-based mappings
all_words_to_num = {**word_to_num, **cardinal_to_num}
text = str(row["contract_title_cleaned"]).lower()
# Match ordinal or cardinal words
for word, num in all_words_to_num.items():
if word in text:
return num
# Match numeric ordinals (e.g., 1st, 2nd, 3rd, 25th)
num_match = re.search(ordinal_pattern, text)
if num_match:
value = re.sub(r"(st|nd|rd|th)$", "", num_match.group(0))
if value.isdigit():
return int(value)
return None
def assign_parent_name_by_rank(df):
"""
Assigns the parent's 'FILE_NAME' to child rows based on TIN and a
modified combined_rank.
Args:
df (pd.DataFrame): DataFrame with 'TIN', 'parent_child_flag',
'combined_rank', and 'FILE_NAME' columns.
'parent_child_flag' should be 'parent' or 'child'.
Returns:
pd.DataFrame: DataFrame with an added 'parent_name' column.
"""
df = df.copy()
df["parent_name"] = None # Initialize parent_name column
# Separate parents and children based on parent_child_flag
parents_df = df[df["parent_child_flag"] == "Parent"].copy()
children_df = df[df["parent_child_flag"] == "Child"].copy()
# Create a mapping for parents: (TIN, combined_rank) -> FILE_NAME
# This will be used to look up the parent's name efficiently
parent_lookup = parents_df.set_index(["TIN", "combined_rank"])[
"FILE_NAME"
].to_dict()
# Iterate through each child to find its parent
for index, child_row in children_df.iterrows():
child_tin = child_row["TIN"]
child_combined_rank = str(
child_row["combined_rank"]
) # Ensure string for splitting
# 2. Remove the last decimal to get the conceptual parent rank
if "." in child_combined_rank:
parent_conceptual_rank = ".".join(child_combined_rank.split(".")[:-1])
else:
# If a child rank doesn't have a decimal (e.g., '1'), it might not have a parent of this type
parent_conceptual_rank = (
None # Or handle as appropriate, maybe it's a top-level child
)
if parent_conceptual_rank:
# 3. Try to find a parent with the matching TIN and conceptual rank
parent_contract_name = parent_lookup.get(
(child_tin, parent_conceptual_rank)
)
# 3. Add parent name to the row where the corresponding child is in
if parent_contract_name:
df.at[index, "parent_name"] = parent_contract_name
return df
# # molina
def parent_child_mapping(cleaned_df):
grouper_column = "TIN"
logging.info(
f"Unique Contract Count: {cleaned_df['contract_name_cleaned'].nunique()}"
)
tin_present = create_df_per_tin_presence(cleaned_df, True)
logging.info(
f"Contract Count after dropping contracts w/o TINs: {tin_present.shape}"
)
tin_present = filter_on_exclude_keywords(tin_present, True)
tin_present = filter_on_missing_title(tin_present)
logging.info(
f"Contract Count after filtering on exclude keywords and missing title: {tin_present.shape}"
)
# tin_present = flag_duplicates_within_group(tin_present, grouper_col=grouper_column)
tin_present = rank_parents(tin_present, grouper_col=grouper_column)
tin_present = flag_multi_parent_with_early_dates(
tin_present, grouper_col=grouper_column
)
tin_present = assign_parents_with_state_and_payer(
tin_present, grouper_col=grouper_column
)
tin_present = assign_no_parent_by_title(tin_present, grouper_col=grouper_column)
tin_present = assign_orphans_based_on_unique_parent_and_payer(
tin_present, grouper_col="TIN", payer_col="payer_name_cleaned"
)
tin_present = assign_child_ranks(tin_present, grouper_col=grouper_column)
tin_present["numbering"] = tin_present.apply(extract_ordinal, axis=1)
tin_present = tin_present[
[
"FILE_NAME",
"CONTRACT_TITLE",
"PROV_GROUP_NAME_FULL",
"PAYER_STATE",
"contract_name_cleaned",
"contract_title_cleaned",
"TIN",
"fixed_effective_date",
"AARETE_DERIVED_TERMINATION_DT",
"payer_name_cleaned",
"PROV_GROUP_NAME_FULL_cleaned",
"consolidated_lob",
"consolidated_service_term",
"parent",
"combined_rank",
"numbering",
]
]
tin_present["parent_child_flag"] = tin_present["parent"].map(
{True: "Parent", False: "Child"}
)
tin_present.loc[
(tin_present["parent"] == False)
& (tin_present["combined_rank"].astype(str).str.count(r"\.") == 1),
"parent_child_flag",
] = "Orphan"
tin_present = clean_termination_date(tin_present, "AARETE_DERIVED_TERMINATION_DT")
cols_order = [
"TIN",
"PROV_GROUP_NAME_FULL_cleaned",
"consolidated_lob",
"fixed_effective_date",
"parent_child_flag",
"combined_rank",
]
tin_present["naming_col"] = (
tin_present[cols_order]
.replace("", "xx")
.fillna("xx")
.astype(str)
.agg("___".join, axis=1)
)
tin_present["FILE_NAME"] = tin_present["FILE_NAME"].str.replace("Filename: ", "")
logging.info(f"tin present: {tin_present.columns}")
tin_present = assign_parent_name_by_rank(tin_present)
tin_present = tin_present[
[
"FILE_NAME",
"CONTRACT_TITLE",
"PROV_GROUP_NAME_FULL",
"PAYER_STATE",
"contract_name_cleaned",
"contract_title_cleaned",
"TIN",
"fixed_effective_date",
"fixed_termination_date",
"payer_name_cleaned",
"PROV_GROUP_NAME_FULL_cleaned",
"consolidated_lob",
"consolidated_service_term",
"parent",
"combined_rank",
"numbering",
"parent_child_flag",
"parent_name",
"naming_col",
]
]
return tin_present
@@ -0,0 +1,496 @@
import warnings
warnings.filterwarnings("ignore")
import pandas as pd
import nltk
nltk.download("stopwords")
from nltk.corpus import stopwords
stop_words = stopwords.words("english")
import re
import string
import src.config as config
from datetime import datetime
import logging
def clean_filename(all_fields_df, contact_name_col="FILE_NAME"): # Fixing FILE_NAME
all_fields_df["contract_name_cleaned"] = (
all_fields_df[contact_name_col]
.str.replace(r"Filename: ?", "", regex=True)
.str.replace(r".txt", "", regex=True)
.str.replace(r".pdf", "", regex=True)
.str.replace(r"@", "", regex=True)
.str.replace(r"^[^a-zA-Z0-9]+", "", regex=True)
)
return all_fields_df
def create_one_to_one_df(df):
"""
Create a DataFrame with AC Output columns.
"""
one_to_one_df = df.drop_duplicates(subset="FILE_NAME")
return one_to_one_df
def clean_contract_title(one_to_one_df, contact_name_col="CONTRACT_TITLE"):
# Cleaning contract title
one_to_one_df["contract_title_cleaned"] = (
one_to_one_df[contact_name_col]
.astype(str)
.str.lower()
.str.replace(r"inc", "", regex=True)
.str.replace(r"[^a-zA-Z0-9]", " ", regex=True)
.apply(
lambda x: " ".join([word for word in x.split() if word not in stop_words])
)
)
return one_to_one_df
def clean_effective_date(
one_to_one_df, effective_date_col="AARETE_DERIVED_EFFECTIVE_DT"
):
one_to_one_df["fixed_effective_date"] = pd.to_datetime(
one_to_one_df[effective_date_col], errors="coerce"
).dt.strftime("%m/%d/%Y")
return one_to_one_df
def clean_termination_date(
one_to_one_df, termination_date_col="AARETE_DERIVED_TERMINATION_DT"
):
def format_value(x):
# Handle null values
if pd.isna(x):
return None
# Handle "UNKNOWN" string
if isinstance(x, str) and x == "UNKNOWN":
return None
# Handle datetime objects
if isinstance(x, (pd.Timestamp, datetime)):
return x.strftime("%m/%d/%Y")
# Try to convert anything else
try:
return pd.to_datetime(x).strftime("%m/%d/%Y")
except:
return x
one_to_one_df["fixed_termination_date"] = one_to_one_df[termination_date_col].apply(
format_value
)
return one_to_one_df
def clean_payer_name(text, hit_words):
"""
Processes a string by:
1. Lowercasing it.
2. Removing everything after (and including) the first occurrence of any word from `hit_words`.
3. Removing all punctuation.
Args:
text (str): The input string.
hit_words (list): A list of words (strings) that, if found, will truncate the string.
Returns:
str: The cleaned and truncated string.
"""
text = text.lower()
# exception of molina healthcare
if "molina healthcare" in text or "molina heathcare" in text or text == "molina":
return "molina healthcare"
# 2. Remove all types of punctuation
text = text.translate(str.maketrans("", "", string.punctuation))
# 3. Remove everything after and including the first occurrence of a hit word
if hit_words: # Check if there are hit words to avoid unnecessary processing
# Create a regex pattern to match any of the hit words, case-insensitive
# and capture the part before the hit word
hit_pattern = (
r"(.*?)(?:\b" + "|".join(re.escape(word) for word in hit_words) + r"\b.*)"
)
match = re.search(hit_pattern, text)
if match:
text = match.group(1).strip() # Keep only the part before the hit word
# Remove any extra spaces that might have been created
text = re.sub(r"\s+", " ", text).strip()
# standadrizing for molina healthcare
# if text == 'molina':
# return 'molina healthcare'
return text
def clean_PROV_GROUP_NAME_FULL(
name, cutoff_words=["inc", "ltd", "llc", "corp", "corporation", "incorporated"]
):
"""
Cleans a string with the following steps:
1. Lowercases everything
2. Removes everything after the first occurrence of any word in cutoff_words
3. Replaces all non-alphanumeric characters with a space
4. Removes extra spaces
Parameters
----------
name : str
Input string
cutoff_words : list[str], optional
List of keywords to cut off text after first occurrence (e.g. ['inc', 'llc', 'd/b/a'])
Returns
-------
str or None
"""
if pd.isna(name):
return None
text = str(name).lower().strip()
# Step 2: cutoff after first keyword match
if cutoff_words:
pattern = r"\b(" + "|".join(map(re.escape, cutoff_words)) + r")\b.*"
text = re.sub(pattern, "", text)
# Step 3: replace non-alphanumeric with space
text = re.sub(r"[^a-z0-9]", " ", text)
# Step 4: collapse spaces
text = re.sub(r"\s+", " ", text).strip()
return text
def extract_tin_like(value):
if not isinstance(value, str):
return None
val = value.strip()
# Check for pattern: XX-XXXXXXX
if re.match(r"^\d{2}-\d{7}", val[:10]):
return val[:10]
# Check for pattern: XXXXXXXXX
if re.match(r"^\d{9}", val[:9]):
return val[:2] + "-" + val[2:9]
return None
def consolidate_one_to_n_fields(contract_name_col, df, fields_to_consolidate):
"""
Consolidate specified fields for each contract into unique, lowercased lists.
Parameters
----------
df : pandas.DataFrame
DataFrame containing 'FILE_NAME' and fields to consolidate.
fields_to_consolidate : list of str
List of column names to consolidate (e.g., ['LOB', 'SERVICE_TERM']).
Returns
-------
pandas.DataFrame
DataFrame with 'FILE_NAME' and one consolidated column per field in fields_to_consolidate,
named as 'consolidated_<field>'.
"""
# Aggregate into lists (drop NaN so they don't turn into "nan")
agg_dict = {
field: (lambda x: [", ".join(x.dropna().astype(str))])
for field in fields_to_consolidate
}
grouped = df.groupby(contract_name_col).agg(agg_dict).reset_index()
for field in fields_to_consolidate:
new_col = f'consolidated_{field.lower().replace(" ", "_")}'
grouped[new_col] = grouped[field].apply(
lambda x: ", ".join(
sorted(
set(
item.strip().lower()
for sublist in x
for item in sublist.split(",")
if item.strip() != "" and item.strip().lower() != "nan"
)
)
)
)
grouped = grouped.drop(columns=[field])
# Keep 'FILE_NAME' and all consolidated columns
cols = [contract_name_col] + [
f'consolidated_{field.lower().replace(" ", "_")}'
for field in fields_to_consolidate
]
return grouped[cols]
def drop_duplicates(df, subset=["contract_name_cleaned", "fixed_effective_date"]):
"""
Drop duplicates based on the specified subset of columns, treating nulls as distinct values.
Parameters
----------
df : pandas.DataFrame
DataFrame to drop duplicates from.
subset : list of str
List of column names to consider for identifying duplicates.
Returns
-------
pandas.DataFrame
DataFrame with duplicates dropped based on the specified subset, treating nulls as distinct.
"""
logging.info(f"Shape before dropping duplicates: {df.shape}")
# Create a helper column that combines the subset columns as a tuple, with None for nulls
temp = df[subset].apply(
lambda row: tuple(x if pd.notnull(x) else None for x in row), axis=1
)
# Keep only the first occurrence of each unique tuple
mask = ~temp.duplicated()
df = df[mask]
logging.info(f"Shape after dropping duplicates: {df.shape}")
return df
def drop_forms(
df,
contract_name_col="contract_name_cleaned",
exception_words=["amd", "add", "amendment", "amend", "adden", "addendum"],
partial_match=False,
):
"""
Drop rows containing 'form' or 'forms', unless one of the exception words is present.
:param df: DataFrame
:param contract_name_col: Column to check
:param exception_words: List of words; if any are present, row is kept
:param partial_match: If True, exception words can match substrings (e.g., 'adden' matches 'addendum')
"""
if exception_words is None:
exception_words = []
logging.info(f"Shape before dropping Forms: {df.shape}")
# Pattern to detect 'form' or 'forms' (corrected regex)
form_pattern = r"\bform?\b"
# Boolean mask for rows containing 'form' or 'forms'
has_form = df[contract_name_col].str.contains(form_pattern, case=False, na=False)
# Boolean mask for exception words
if exception_words:
if partial_match:
exc_pattern = "|".join([re.escape(word) for word in exception_words])
else:
exc_pattern = "|".join(
[rf"\b{re.escape(word)}\b" for word in exception_words]
)
has_exception = df[contract_name_col].str.contains(
exc_pattern, case=False, na=False
)
else:
has_exception = pd.Series(False, index=df.index)
# Keep rows where either 'form' not present OR exception word present
df = df[~has_form | has_exception]
logging.info(f"Shape after dropping Forms: {df.shape}")
return df
def drop_exact_matches(
df,
contract_name_col="contract_name_cleaned",
exact_patterns=None,
exception_words=None,
partial_match=False,
):
"""
Drop rows that contain specified exact patterns (case-insensitive),
with or without spaces around dashes, unless one of the exception words is present.
:param df: DataFrame
:param contract_name_col: Column to check
:param exact_patterns: List of exact patterns to match (as they appear in the list)
(e.g., ['- loi -', 'com - com', 'agmt - coll', '- hdo -', 'term - term'])
Will match with or without spaces around dashes
:param exception_words: List of words; if any are present, row is kept
:param partial_match: If True, exception words can match substrings
"""
if exact_patterns is None:
exact_patterns = [
"- loi -",
"com - com",
"agmt - coll",
"- hdo -",
"term - term",
"- ros -",
]
if exception_words is None:
exception_words = []
logging.info(f"Shape before dropping exact patterns: {df.shape}")
# Boolean mask for rows that contain any of the exact patterns
contains_pattern = pd.Series(False, index=df.index)
for pattern in exact_patterns:
# Normalize the pattern by removing spaces around dashes
# Then create regex that allows optional spaces around dashes
# E.g., "- loi -" becomes pattern that matches "-loi-", "- loi -", "-loi -", etc.
normalized = pattern.strip()
# Replace " - " with flexible dash pattern, and handle edge cases
# Split by dashes to get the parts
parts = [p.strip() for p in normalized.split("-")]
parts = [p for p in parts if p] # Remove empty parts
# Build regex pattern with flexible spacing around dashes
if len(parts) == 1:
# Pattern like "- loi -" (single word between dashes)
regex_pattern = rf"-\s*{re.escape(parts[0])}\s*-"
elif len(parts) == 2:
# Pattern like "com - com" or "agmt - coll" (two words with dash)
regex_pattern = rf"{re.escape(parts[0])}\s*-\s*{re.escape(parts[1])}"
else:
# For any other pattern, just escape and allow flexible spaces
regex_pattern = re.escape(normalized).replace(r"\ ", r"\s*")
contains_pattern |= df[contract_name_col].str.contains(
regex_pattern, case=False, na=False, regex=True
)
# Boolean mask for exception words
if exception_words:
if partial_match:
exc_pattern = "|".join([re.escape(word) for word in exception_words])
else:
exc_pattern = "|".join(
[rf"\b{re.escape(word)}\b" for word in exception_words]
)
has_exception = df[contract_name_col].str.contains(
exc_pattern, case=False, na=False
)
else:
has_exception = pd.Series(False, index=df.index)
# Keep rows where either pattern not present OR exception word present
df = df[~contains_pattern | has_exception]
logging.info(f"Shape after dropping exact patterns: {df.shape}")
return df
def drop_letter(
df,
contract_name_col="contract_title_cleaned",
patterns=None,
exception_words=None,
partial_match=False,
):
"""
Drop rows that contain 'letter' or 'letter intent' as whole words (case-insensitive),
unless one of the exception words is present.
:param df: DataFrame
:param contract_name_col: Column to check (default: 'contract_title_cleaned')
:param patterns: List of patterns to match as whole words
(default: ['letter', 'letter intent'])
:param exception_words: List of words; if any are present, row is kept
:param partial_match: If True, exception words can match substrings
"""
if patterns is None:
patterns = ["letter intent", "extension letter"]
if exception_words is None:
exception_words = []
logging.info(f"Shape before dropping Letter: {df.shape}")
# Boolean mask for rows that contain any of the patterns
contains_pattern = pd.Series(False, index=df.index)
for pattern in patterns:
# Create regex pattern for whole word/phrase matching
# Handle multi-word patterns like "letter intent"
words = pattern.split()
if len(words) == 1:
# Single word - use word boundaries
regex_pattern = rf"\b{re.escape(pattern)}\b"
else:
# Multi-word phrase - use word boundary at start and end
regex_pattern = (
r"\b" + r"\s+".join([re.escape(word) for word in words]) + r"\b"
)
contains_pattern |= df[contract_name_col].str.contains(
regex_pattern, case=False, na=False, regex=True
)
# Boolean mask for exception words
if exception_words:
if partial_match:
exc_pattern = "|".join([re.escape(word) for word in exception_words])
else:
exc_pattern = "|".join(
[rf"\b{re.escape(word)}\b" for word in exception_words]
)
has_exception = df[contract_name_col].str.contains(
exc_pattern, case=False, na=False
)
else:
has_exception = pd.Series(False, index=df.index)
# Keep rows where either pattern not present OR exception word present
df = df[~contains_pattern | has_exception]
logging.info(f"Shape after dropping Letter: {df.shape}")
return df
def parent_child_preprocessing(df_read):
all_fields_df = clean_filename(df_read)
one_to_one_df = create_one_to_one_df(all_fields_df)
one_to_one_df = clean_contract_title(one_to_one_df)
one_to_one_df = clean_effective_date(one_to_one_df, "AARETE_DERIVED_EFFECTIVE_DT")
one_to_one_df["payer_name_cleaned"] = one_to_one_df["PAYER_NAME"].apply(
lambda x: clean_payer_name(str(x), ["inc", "llc", "dba"])
)
one_to_one_df["PROV_GROUP_NAME_FULL_cleaned"] = one_to_one_df[
"PROV_GROUP_NAME_FULL"
].apply(clean_PROV_GROUP_NAME_FULL)
one_to_one_df["TIN"] = one_to_one_df["contract_name_cleaned"].apply(
extract_tin_like
)
consolidated_one_to_n_fields = consolidate_one_to_n_fields(
"contract_name_cleaned", all_fields_df, ["LOB", "SERVICE_TERM"]
)
df = pd.merge(
one_to_one_df,
consolidated_one_to_n_fields,
on="contract_name_cleaned",
how="left",
)
df = drop_duplicates(df)
df = drop_forms(df, partial_match=True)
df = drop_exact_matches(df)
df = drop_letter(df)
logging.info(f"df after preprocessing: {df.columns}")
return df
+72 -7
View File
@@ -3,6 +3,7 @@ import logging
import os
import tempfile
from io import StringIO
import json
from threading import Lock
import pandas as pd
@@ -10,7 +11,8 @@ import src.tracking.tracking_utils as tracking_utils
from botocore.exceptions import ClientError
from pyxlsb import open_workbook
from src import config
from pathlib import Path
from typing import Optional
# Thread-safe lock for JSON file operations
json_write_lock = Lock()
@@ -337,7 +339,7 @@ def read_xlsb(
def write_local(
df: pd.DataFrame, filename: str, run_timestamp: str, output_type: str
) -> None:
) -> Optional[str]:
"""
Saves a DataFrame as a CSV file in a local directory.
@@ -345,10 +347,11 @@ def write_local(
df (pd.DataFrame): The DataFrame to save.
filename (str): The base filename to use when saving the file.
run_timestamp (str): The timestamp used to organize output files.
output_type (str): Type of output, either "final" or "individual".
output_type (str): Type of output, either "final", "parent_child" or "individual".
Returns:
None; this function is called for its side effect of writing a file to local.
- Output path if parent-child mapping is performed.
- None, when this function is called for its side effect of uploading a file to S3.
Notes:
- Creates output directories if they do not exist
@@ -366,6 +369,8 @@ def write_local(
index=False,
quoting=1,
)
if config.PERFORM_PARENT_CHILD_MAPPING:
return os.path.join(output_dir, f"{config.BATCH_ID}-RESULTS.csv")
return None
elif output_type == "individual":
base_filename = os.path.splitext(filename)[0].strip()
@@ -375,6 +380,18 @@ def write_local(
index=False,
quoting=1,
)
if config.PERFORM_PARENT_CHILD_MAPPING:
return os.path.join(config.OUTPUT_DIRECTORY, f"{base_filename}_-RESULTS.csv")
return None
elif output_type == "parent_child":
output_dir = os.path.join(config.CONSOLIDATED_OUTPUT_DIRECTORY, run_timestamp)
os.makedirs(output_dir, exist_ok=True)
df.to_csv(
os.path.join(output_dir, f"{config.BATCH_ID}-Parent_Child.csv"),
index=False,
quoting=1,
)
return None
elif output_type == "error":
output_dir = os.path.join(config.CONSOLIDATED_OUTPUT_DIRECTORY, run_timestamp)
@@ -384,6 +401,8 @@ def write_local(
index=False,
quoting=1,
)
if config.PERFORM_PARENT_CHILD_MAPPING:
return os.path.join(output_dir, f"{config.BATCH_ID}-RESULTS.csv")
return None
elif output_type == "usage":
output_dir = os.path.join(config.CONSOLIDATED_OUTPUT_DIRECTORY, run_timestamp, "tracking")
@@ -430,15 +449,16 @@ def write_local(
def write_s3(
df: pd.DataFrame, filename: str, run_timestamp: str, output_type: str
) -> None:
) -> Optional[str]:
"""
Uploads a DataFrame to an S3 bucket as a CSV file.
Returns Output path if parent-child mapping is performed.
Args:
df (pd.DataFrame): The DataFrame to be written to S3.
filename (str): The base filename to use when saving the file.
run_timestamp (str): A timestamp representing the current run.
output_type (str): Type of output, either "final", "individual", or "error".
output_type (str): Type of output, either "final", "individual", "parent_child" or "error".
The function generates an S3 key based on the configured BATCH_ID and
uploads the DataFrame as a CSV file to the configured S3 bucket.
@@ -446,7 +466,9 @@ def write_s3(
convention "<BATCH_ID>-RESULTS.csv".
Returns:
None; this function is called for its side effect of uploading a file to S3.
- Output path if parent-child mapping is performed.
- None, when this function is called for its side effect of uploading a file to S3.
Configurations used:
- config.BATCH_ID: Identifier for the batch process.
@@ -462,6 +484,10 @@ def write_s3(
output_path = (
f"{config.BATCH_ID}/{run_timestamp}/individual/{filename}-RESULTS.csv"
)
elif output_type == "parent_child":
output_path = (
f"{config.BATCH_ID}/{run_timestamp}/parent_child/{filename}-RESULTS.csv"
)
elif output_type == "error":
output_path = f"{config.BATCH_ID}/{run_timestamp}/{config.BATCH_ID}-ERRORS.csv"
elif output_type == "usage":
@@ -485,6 +511,8 @@ def write_s3(
logging.info(f"Saved batch output file: {output_path}")
elif output_type == "individual":
logging.info(f"Saved individual output file: {output_path}")
elif output_type == "parent_child":
logging.info(f"Saved parent_child output file: {output_path}")
elif output_type == "error":
logging.info(f"Saved error output file: {output_path}")
elif output_type == "usage":
@@ -497,6 +525,10 @@ def write_s3(
logging.info(f"Saved DTC summary file: {output_path}")
else:
logging.info(f"Saved unknown output type file: {output_path}")
if config.PERFORM_PARENT_CHILD_MAPPING:
return f"s3://{config.S3_OUTPUT_BUCKET}/{output_path}"
return None
except ClientError as e:
logging.error(f"Error uploading to S3: {e}")
@@ -548,3 +580,36 @@ def save_result_to_json(filename, result, json_folder):
# Save result to individual JSON file
with open(json_file_path, "w") as f:
json.dump({"filename": filename, "result": result}, f, indent=2)
def read_DataFrame(file_path: str) -> pd.DataFrame:
"""
Takes in a file path and returns a pandas DataFrame.
Supports reading from local file system and only CSV from S3.
"""
is_s3 = file_path.startswith('s3://')
if is_s3:
file = read_s3_csv(file_path)
return file
file = read_local(file_path)
return file
def convert_json_to_dict(json_file_path: str) -> dict:
"""
Reads a JSON file and converts it to a dictionary.
Converts string keys 'True' and 'False' to boolean keys True and False.
"""
json_str = Path(json_file_path).read_text()
try:
data = json.loads(json_str)
# Convert "True"/"False" string keys to boolean keys
converted = {
(True if k == "True" else False if k == "False" else k): v
for k, v in data.items()
}
return converted
except json.JSONDecodeError:
return {}