Merged in feature/PC_logic_cleanup_output (pull request #984)

Feature/PC logic cleanup output

* Few tweaks PC_logics

* Fixed orphan_ranking

* black formatting

* Changes on output field and ranking method

* updated few hotfixes

* black format fix


Approved-by: Katon Minhas
This commit is contained in:
Rahul Ailaboina
2026-04-27 15:04:07 +00:00
committed by Katon Minhas
parent f351ced7ea
commit e4650064aa
5 changed files with 386 additions and 159 deletions
+19 -3
View File
@@ -38,6 +38,9 @@ class ColumnMapper:
r"^agreement.*title$",
r"^doc.*title$",
],
"AARETE_DERIVED_PAYER_NAME": [
r"^aarete_derived_payer_name$",
],
"PAYER_NAME": [
r"^payer.*name$",
r"^payer$",
@@ -74,6 +77,15 @@ class ColumnMapper:
r"^national.*provider.*id$",
r"^provider.*npi$",
],
"PROV_OTHER_TIN": [
r"^prov.*other.*tin$",
],
"PROV_OTHER_NPI": [
r"^prov.*other.*npi$",
],
"PROV_OTHER_NAME_FULL": [
r"^prov.*other.*name.*full$",
],
"EFFECTIVE_DATE": [
r"aarete_derived_effective_dt",
r"^.*effective.*date$",
@@ -124,6 +136,10 @@ class ColumnMapper:
"PROV_GROUP_NPI",
"AARETE_DERIVED_AMENDMENT_NUM",
"AARETE_DERIVED_PROVIDER_NAME",
"AARETE_DERIVED_PAYER_NAME",
"PROV_OTHER_TIN",
"PROV_OTHER_NPI",
"PROV_OTHER_NAME_FULL",
]
def __init__(self, df: pd.DataFrame):
@@ -154,9 +170,9 @@ class ColumnMapper:
Returns:
Matched column name or None
"""
for col in available_columns:
normalized_col = self._normalize_column_name(col)
for pattern in patterns:
for pattern in patterns:
for col in available_columns:
normalized_col = self._normalize_column_name(col)
if re.match(pattern, normalized_col, re.IGNORECASE):
return col
return None
+166 -57
View File
@@ -1,5 +1,6 @@
from __future__ import annotations
import json
import pandas as pd
import re
from pathlib import Path
@@ -27,6 +28,60 @@ _MODULE_DIR = Path(__file__).resolve().parent
_CONSTANTS_DIR = _MODULE_DIR.parent / "constants" / "mappings"
def _build_provider_group_label(df: pd.DataFrame) -> pd.Series:
"""Human-readable grouping label for reviewers.
Uses the cleaned provider name, title-cased. When the same name appears
with more than one TIN in the dataset, disambiguate by appending the TIN
(e.g. "Storr Medical Center - 123456789"). Falls back to grouping_key
for rows with no usable provider name.
"""
name_col = "PROV_GROUP_NAME_FULL_cleaned"
tin_col = "PROV_GROUP_TIN"
if name_col not in df.columns:
return df.get("grouping_key", pd.Series([""] * len(df), index=df.index)).astype(
"string"
)
name = df[name_col].fillna("").astype(str).str.strip()
tin = (
df[tin_col].fillna("").astype(str).str.strip()
if tin_col in df.columns
else pd.Series([""] * len(df), index=df.index)
)
# Count distinct non-empty TINs per name; names with >1 TIN need
# disambiguation.
mask = (name != "") & (tin != "")
tins_per_name: Dict[str, int] = (
pd.DataFrame({"_n": name[mask], "_t": tin[mask]})
.groupby("_n")["_t"]
.nunique()
.to_dict()
)
fallback = (
df["grouping_key"].fillna("").astype(str)
if "grouping_key" in df.columns
else pd.Series([""] * len(df), index=df.index)
)
def _label(n: str, t: str, fb: str) -> str:
if not n:
return fb
display = n.title()
if tins_per_name.get(n, 0) > 1 and t:
return f"{display} - {t}"
return display
return pd.Series(
[_label(n, t, fb) for n, t, fb in zip(name, tin, fallback)],
index=df.index,
dtype="string",
)
def create_grouping_key_and_tier(df):
"""
Creates grouping key based on available non-null values in PROV_GROUP_TIN,
@@ -322,6 +377,41 @@ def build_grouping_string(child_row, grouping_cols):
return " and ".join(parts)
def _parse_identifier_set(value) -> set:
"""Normalize a TIN/NPI field into a set of individual ID strings.
Source data often stores identifiers as JSON-encoded lists
(e.g. `'["133757370", "840611484"]'`) because a single contract can
span multiple legal entities. Plain scalars (`'133757370'`) are also
supported. Returns an empty set for NaN / empty / "nan" values.
Normalizing to a set lets downstream matchers use intersection
instead of string equality — two rows that share a TIN compare as
compatible even if one encodes it as `["A", "B"]` and the other as
`["B", "C", "D"]`.
"""
if value is None:
return set()
try:
if pd.isna(value):
return set()
except (TypeError, ValueError):
pass
s = str(value).strip()
if not s or s.lower() == "nan":
return set()
# JSON list (the common case from upstream extraction).
if s.startswith("[") and s.endswith("]"):
try:
parsed = json.loads(s)
if isinstance(parsed, list):
return {str(v).strip() for v in parsed if str(v).strip()}
except (json.JSONDecodeError, ValueError):
pass
# Plain scalar — wrap as a singleton set.
return {s}
def check_grouping_compatibility(parent_key, child_key):
"""
Checks if a parent's grouping key is compatible with a child's grouping key.
@@ -329,6 +419,9 @@ def check_grouping_compatibility(parent_key, child_key):
Rules:
- Extract values from both keys
- Compatible if they share AT LEAST ONE matching value
- For TIN / NPI, values may be JSON-list strings (e.g. `["A","B"]`);
we compare as sets so a parent holding `["A","B"]` is compatible
with a child holding `["B","C"]` via the shared TIN "B"
- This allows cross-tier matching (e.g., Tier 1 parent can match Tier 3 child via TIN)
Examples:
@@ -336,9 +429,9 @@ def check_grouping_compatibility(parent_key, child_key):
Child: "TIN:123|NPI:456"
Result: ✅ Compatible (share TIN and NPI)
Parent: "TIN:123|NPI:456|NAME:ABC"
Child: "TIN:123|NPI:999"
Result: ✅ Compatible (share TIN:123, even though NPI differs)
Parent: "TIN:[\"123\",\"456\"]"
Child: "TIN:[\"456\",\"789\"]"
Result: ✅ Compatible (share TIN 456 in the list)
Parent: "TIN:123|NPI:456|NAME:ABC"
Child: "TIN:999"
@@ -361,10 +454,17 @@ def check_grouping_compatibility(parent_key, child_key):
child_dict = parse_key(child_key)
# Check if they share AT LEAST ONE matching value
for key, child_value in child_dict.items():
parent_value = parent_dict.get(key)
if parent_value == child_value:
return True # Found at least one match!
for field, child_value in child_dict.items():
parent_value = parent_dict.get(field)
if parent_value is None:
continue
if field in ("TIN", "NPI"):
# Set-intersection comparison for list-encoded identifiers.
if _parse_identifier_set(parent_value) & _parse_identifier_set(child_value):
return True
else:
if parent_value == child_value:
return True
return False # No matching values found
@@ -380,6 +480,11 @@ def create_df_per_tin_presence(df, is_tin_present):
def filter_on_exclude_keywords(df, is_tin_present):
"""
Flags rows as 'parent' based on exclusion keywords.
`contract_title_cleaned` is the primary signal. The cleaned filename is
consulted only when the title is absent — otherwise a row with no title
but an obvious child-type filename (e.g. "... Amendment.pdf", "... AMD ...")
would silently default to parent. When a title is present, it wins.
"""
df = df.copy()
df["parent"] = True
@@ -392,9 +497,28 @@ def filter_on_exclude_keywords(df, is_tin_present):
r"\b(?:" + "|".join(re.escape(kw) for kw in exclude_keywords_list) + r")\b"
)
df["parent"] = ~(
df["contract_title_cleaned"].str.contains(exclude_pattern, case=False, na=False)
title = df["contract_title_cleaned"].astype("string").fillna("").str.strip()
title_has_kw = title.str.contains(exclude_pattern, case=False, na=False)
title_missing = (title == "") | (title.str.lower() == "nan")
if "contract_name_cleaned" in df.columns:
filename_source = df["contract_name_cleaned"]
else:
filename_source = df.get("FILE_NAME", pd.Series([""] * len(df), index=df.index))
# Normalize the filename so regex word boundaries land cleanly. Digits,
# underscores and punctuation are regex word chars, which means `\bamd\b`
# won't match "AMD1" or "_AMD#". Replacing non-letters with spaces first
# gives us `BCBSAZ AMD to MSA` and `\bamd\b` catches it.
filename_normalized = (
filename_source.astype("string")
.fillna("")
.str.replace(r"[^a-zA-Z]+", " ", regex=True)
)
filename_has_kw = filename_normalized.str.contains(
exclude_pattern, case=False, na=False
)
df["parent"] = ~(title_has_kw | (title_missing & filename_has_kw))
return df
@@ -478,23 +602,20 @@ def build_matching_columns_string(parent_row, child_row):
"""
matched_parts = []
# Check TIN
parent_tin = parent_row.get("PROV_GROUP_TIN")
child_tin = child_row.get("PROV_GROUP_TIN")
if pd.notna(parent_tin) and pd.notna(child_tin) and parent_tin == child_tin:
if isinstance(child_tin, float) and child_tin == int(child_tin):
matched_parts.append(f"TIN ({int(child_tin)})")
else:
matched_parts.append(f"TIN ({child_tin})")
# Check TIN — parent/child may each encode multiple TINs as JSON lists;
# report the intersection so the reasoning shows which ID actually matched.
shared_tins = _parse_identifier_set(parent_row.get("PROV_GROUP_TIN")) & (
_parse_identifier_set(child_row.get("PROV_GROUP_TIN"))
)
if shared_tins:
matched_parts.append(f"TIN ({', '.join(sorted(shared_tins))})")
# Check NPI
parent_npi = parent_row.get("PROV_GROUP_NPI")
child_npi = child_row.get("PROV_GROUP_NPI")
if pd.notna(parent_npi) and pd.notna(child_npi) and parent_npi == child_npi:
if isinstance(child_npi, float) and child_npi == int(child_npi):
matched_parts.append(f"NPI ({int(child_npi)})")
else:
matched_parts.append(f"NPI ({child_npi})")
shared_npis = _parse_identifier_set(parent_row.get("PROV_GROUP_NPI")) & (
_parse_identifier_set(child_row.get("PROV_GROUP_NPI"))
)
if shared_npis:
matched_parts.append(f"NPI ({', '.join(sorted(shared_npis))})")
# Check PROV_NAME
parent_name = parent_row.get("PROV_GROUP_NAME_FULL")
@@ -993,7 +1114,9 @@ def assign_child_ranks(df, grouper_col="grouping_key"):
assigned_rank = child_row.get("assigned_parent_rank")
if assigned_rank == ASSIGNMENT_NO_PARENT or pd.isna(assigned_rank):
children.at[idx, "_parent_identity"] = "orphan"
gk = child_row.get("grouping_key")
gk_str = str(gk) if pd.notna(gk) and gk not in (None, "") else "__none__"
children.at[idx, "_parent_identity"] = f"orphan__{gk_str}"
else:
# Find any parent with this rank that's compatible with the child
found = False
@@ -1016,7 +1139,10 @@ def assign_child_ranks(df, grouper_col="grouping_key"):
has_amendment_col = "AARETE_DERIVED_AMENDMENT_NUM" in children.columns
for parent_id, grp_df in children.groupby("_parent_identity"):
if parent_id == "orphan":
is_orphan_bucket = isinstance(parent_id, str) and parent_id.startswith(
"orphan__"
)
if is_orphan_bucket:
parent_rank_str = "0"
else:
# Extract parent_rank from the child's assigned_parent_rank
@@ -1041,6 +1167,9 @@ def assign_child_ranks(df, grouper_col="grouping_key"):
grp_df["sort_date"] = grp_df["fixed_effective_date"].fillna(pd.Timestamp.max)
grp_df = grp_df.sort_values("sort_date")
# Standalone orphan (alone in its grouping_key): no sequence suffix.
standalone_orphan = is_orphan_bucket and len(grp_df) == 1
# Assign sequential ranks using amendment number instead of 0
for i, idx in enumerate(grp_df.index, start=1):
amendment_val = 0
@@ -1055,7 +1184,12 @@ def assign_child_ranks(df, grouper_col="grouping_key"):
amendment_val = ord(raw_str.upper()) - ord("A") + 1
else:
amendment_val = raw
rank_val = f"{parent_rank_str}.{amendment_val}.{i}"
if standalone_orphan:
# Only one orphan in this group — fixed order=1 (never
# sequences up against unrelated rows).
rank_val = f"{parent_rank_str}.{amendment_val}.1"
else:
rank_val = f"{parent_rank_str}.{amendment_val}.{i}"
children.loc[idx, "child_rank"] = rank_val
# Clean up temp column
@@ -1080,33 +1214,6 @@ def assign_child_ranks(df, grouper_col="grouping_key"):
return result
def extract_ordinal(row):
"""
Extracts ordinal numbers from contract titles.
"""
num_dict = io_utils.convert_json_to_dict(
str(_CONSTANTS_DIR / "numeric_mappings.json")
)
word_to_num = num_dict.get("ordinal_word_to_number", {})
cardinal_to_num = num_dict.get("cardinal_word_to_number", {})
ordinal_pattern = num_dict.get("ordinal_regex_pattern", "")
all_words_to_num = {**word_to_num, **cardinal_to_num}
text = str(row["contract_title_cleaned"]).lower()
for word, num in all_words_to_num.items():
if word in text:
return num
num_match = re.search(ordinal_pattern, text)
if num_match:
value = re.sub(r"(st|nd|rd|th)$", "", num_match.group(0))
if value.isdigit():
return int(value)
return None
def parent_child_mapping(
cleaned_df: pd.DataFrame, original_df: pd.DataFrame
) -> Tuple[pd.DataFrame, pd.DataFrame]:
@@ -1281,7 +1388,6 @@ def parent_child_mapping(
# Assign child ranks
pc_df = assign_child_ranks(pc_df, grouper_col="grouping_key")
pc_df["numbering"] = pc_df.apply(extract_ordinal, axis=1)
# Set parent_child_flag
pc_df["parent_child_flag"] = pc_df["parent"].map({True: "Parent", False: "Child"})
@@ -1334,6 +1440,8 @@ def parent_child_mapping(
pc_df["FILE_NAME"] = pc_df["FILE_NAME"].str.replace("Filename: ", "")
pc_df["provider_group"] = _build_provider_group_label(pc_df)
# Final summary
logging.info("=" * 80)
logging.info("FINAL SUMMARY")
@@ -1372,7 +1480,8 @@ def parent_child_mapping(
columns=["contract_name_cleaned", "fixed_effective_date"]
)
# Build pc_df output columns dynamically based on available columns
# Build pc_df output columns dynamically based on available columns.
cols_to_keep_pc_tab = [
"FILE_NAME",
"fixed_effective_date",
@@ -1384,9 +1493,9 @@ def parent_child_mapping(
"PAYER_STATE",
"AARETE_DERIVED_AMENDMENT_NUM",
"grouping_key",
"provider_group",
"parent",
"combined_rank",
"numbering",
"parent_child_flag",
"assignment_reasoning",
"parent_name",
+48 -13
View File
@@ -841,21 +841,27 @@ def parent_child_preprocessing(
one_to_one_df = create_one_to_one_df(all_fields_df)
one_to_one_df = clean_contract_title(one_to_one_df)
one_to_one_df["payer_name_cleaned"] = one_to_one_df[payer_name_col].apply(
lambda x: clean_payer_name(str(x), ["inc", "llc", "dba"])
)
# Use AARETE_DERIVED_PROVIDER_NAME if available (already cleaned/standardized upstream)
# Otherwise fall back to raw PROV_GROUP_NAME_FULL with full cleaning
# Payer: prefer AARETE_DERIVED_PAYER_NAME (already cleaned upstream).
# Fall back to raw PAYER_NAME with cleaning only when derived is absent.
if "AARETE_DERIVED_PAYER_NAME" in one_to_one_df.columns:
logging.info("Using AARETE_DERIVED_PAYER_NAME as-is (no cleaning)")
one_to_one_df["payer_name_cleaned"] = one_to_one_df["AARETE_DERIVED_PAYER_NAME"]
else:
logging.info("AARETE_DERIVED_PAYER_NAME not found, cleaning raw PAYER_NAME")
one_to_one_df["payer_name_cleaned"] = one_to_one_df[payer_name_col].apply(
lambda x: clean_payer_name(str(x), ["inc", "llc", "dba"])
)
# Provider: prefer AARETE_DERIVED_PROVIDER_NAME as-is (already cleaned upstream).
# Fall back to raw PROV_GROUP_NAME_FULL with full cleaning only when derived is absent.
if "AARETE_DERIVED_PROVIDER_NAME" in one_to_one_df.columns:
logging.info(
"Using AARETE_DERIVED_PROVIDER_NAME for provider grouping (lightweight cleaning)"
)
one_to_one_df = lightweight_clean_provider_name(
one_to_one_df,
provider_col="AARETE_DERIVED_PROVIDER_NAME",
)
one_to_one_df = standardize_provider_groups(one_to_one_df)
logging.info("Using AARETE_DERIVED_PROVIDER_NAME as-is (no cleaning)")
one_to_one_df["PROV_GROUP_NAME_FULL_cleaned"] = one_to_one_df[
"AARETE_DERIVED_PROVIDER_NAME"
]
if "DBA_Name" not in one_to_one_df.columns:
one_to_one_df["DBA_Name"] = ""
else:
logging.info(
"AARETE_DERIVED_PROVIDER_NAME not found, falling back to raw provider name cleaning"
@@ -864,6 +870,35 @@ def parent_child_preprocessing(
one_to_one_df = update_group(one_to_one_df)
one_to_one_df = standardize_provider_groups(one_to_one_df)
# Per-row fallback: when a row has no PROV_GROUP_* identifiers at all,
# promote PROV_OTHER_* into the PROV_GROUP_* slots so the downstream
# grouping-key / tier logic treats them as real identifiers instead of
# dropping the row to the filename-based fallback.
other_cols = ("PROV_OTHER_TIN", "PROV_OTHER_NPI", "PROV_OTHER_NAME_FULL")
if any(c in one_to_one_df.columns for c in other_cols):
def _is_blank(series):
as_str = series.astype("string").fillna("").str.strip()
return (as_str == "") | (as_str.str.lower() == "nan")
blank_group = (
_is_blank(one_to_one_df["PROV_GROUP_TIN"])
& _is_blank(one_to_one_df["PROV_GROUP_NPI"])
& _is_blank(one_to_one_df["PROV_GROUP_NAME_FULL_cleaned"])
)
for src_col, dst_col in (
("PROV_OTHER_TIN", "PROV_GROUP_TIN"),
("PROV_OTHER_NPI", "PROV_GROUP_NPI"),
("PROV_OTHER_NAME_FULL", "PROV_GROUP_NAME_FULL_cleaned"),
):
if src_col in one_to_one_df.columns:
one_to_one_df.loc[blank_group, dst_col] = one_to_one_df.loc[
blank_group, src_col
]
logging.info(
f"PROV_OTHER_* fallback applied to {int(blank_group.sum())} rows with no group identifiers"
)
# Determine which fields to consolidate based on available columns
fields_to_consolidate = []
if "LOB" in all_fields_df.columns:
+145 -45
View File
@@ -2,6 +2,7 @@
Parent-Child Contract Mapping and Ranking System
"""
import json
import re
import sys
import time
@@ -235,9 +236,37 @@ class TextProcessor:
normalized = re.sub(r"[^0-9A-Za-z]", "", value.strip())
return normalized.upper()
@staticmethod
def _identifier_parts(val_str: str) -> List[str]:
"""Split a raw identifier value into individual IDs.
Handles:
- JSON lists: '["275440611", "274211365"]' → ['275440611', '274211365']
- Pipe-separated: '275440611|274211365' → ['275440611', '274211365']
- Plain scalars: '275440611' → ['275440611']
JSON-list parsing keeps multi-TIN contracts readable in the grouping
key — without it, `normalize_single_identifier` strips the brackets
and commas and concatenates every digit into one blob.
"""
if val_str.startswith("[") and val_str.endswith("]"):
try:
parsed = json.loads(val_str)
if isinstance(parsed, list):
return [str(p) for p in parsed]
except (json.JSONDecodeError, ValueError):
pass
if "|" in val_str:
return val_str.split("|")
return [val_str]
@staticmethod
def normalize_identifier(value: Any) -> str:
"""Normalize TIN/NPI identifiers (handles pipe-separated values)."""
"""Normalize TIN/NPI identifiers.
Accepts JSON-list, pipe-separated, or scalar input. Returns the
unique IDs joined by `|`, sorted for deterministic bucketing.
"""
if value is None or pd.isna(value):
return ""
@@ -245,17 +274,11 @@ class TextProcessor:
if not val_str:
return ""
if "|" in val_str:
parts = [
TextProcessor.normalize_single_identifier(p) for p in val_str.split("|")
]
seen = set()
unique_parts = []
for p in parts:
if p and p not in seen:
seen.add(p)
unique_parts.append(p)
return "|".join(unique_parts) if unique_parts else ""
parts = TextProcessor._identifier_parts(val_str)
if len(parts) > 1:
normalized = [TextProcessor.normalize_single_identifier(p) for p in parts]
unique = sorted({p for p in normalized if p})
return "|".join(unique) if unique else ""
return TextProcessor.normalize_single_identifier(val_str)
@@ -266,7 +289,10 @@ class TextProcessor:
@staticmethod
def get_identifier_values(value: Any) -> Set[str]:
"""Extract all individual identifier values (handles pipe-separated)."""
"""Extract all individual identifier values.
Handles JSON-list, pipe-separated, and scalar input.
"""
if value is None or pd.isna(value):
return set()
@@ -274,11 +300,10 @@ class TextProcessor:
if not val_str:
return set()
if "|" in val_str:
parts = [
TextProcessor.normalize_single_identifier(p) for p in val_str.split("|")
]
return {p for p in parts if p}
parts = TextProcessor._identifier_parts(val_str)
if len(parts) > 1:
normalized = [TextProcessor.normalize_single_identifier(p) for p in parts]
return {p for p in normalized if p}
normalized = TextProcessor.normalize_single_identifier(val_str)
return {normalized} if normalized else set()
@@ -646,9 +671,15 @@ class ParentIdentifier:
raise ValueError(f"Unknown parent identification strategy: {strategy_type}")
def _identify_by_keyword(self, df: pd.DataFrame, config: Dict) -> pd.Series:
"""Identify parents by excluding keywords."""
"""Identify parents by excluding keywords.
`field` (typically the cleaned contract title) is the primary signal.
`fallback_field` (typically the filename) is consulted only when the
primary field is blank/missing for a row.
"""
field = config.get("field")
exclude_keywords = config.get("exclude_keywords", [])
fallback_field = config.get("fallback_field")
if not field or field not in df.columns:
raise ValueError(f"keyword strategy missing field: {field}")
if not exclude_keywords:
@@ -661,9 +692,24 @@ class ParentIdentifier:
self.pattern_cache[pattern_str] = re.compile(pattern_str, re.IGNORECASE)
compiled = self.pattern_cache[pattern_str]
has_keyword = (
df[field].astype("string").str.contains(compiled, na=False, regex=True)
)
primary = df[field].astype("string").fillna("").str.strip()
has_keyword = primary.str.contains(compiled, na=False, regex=True)
if fallback_field and fallback_field in df.columns:
primary_missing = (primary == "") | (primary.str.lower() == "nan")
# Normalize the fallback (typically FILE_NAME) so regex word
# boundaries land cleanly. Digits, underscores and punctuation
# are regex word chars, so `\bamd\b` won't match "AMD1" or
# "_AMD#" without this preprocessing step.
fallback_has_kw = (
df[fallback_field]
.astype("string")
.fillna("")
.str.replace(r"[^a-zA-Z]+", " ", regex=True)
.str.contains(compiled, na=False, regex=True)
)
has_keyword = has_keyword | (primary_missing & fallback_has_kw)
return ~has_keyword
def _identify_by_field_value(self, df: pd.DataFrame, config: Dict) -> pd.Series:
@@ -687,7 +733,7 @@ class ParentIdentifier:
if not field or field not in df.columns:
raise ValueError(f"source_label strategy missing field: {field}")
labels = TextProcessor.normalize_label(df[field])
return (labels == "Parent").fillna(False)
return (labels == "PARENT").fillna(False)
def _identify_bcbs_custom(self, df: pd.DataFrame, config: Dict) -> pd.Series:
"""
@@ -1609,16 +1655,19 @@ class ConfigFactory:
parent_identification={
"strategy_type": "keyword",
"field": "title_clean",
"fallback_field": "FILE_NAME",
"exclude_keywords": [
"exhibit",
"amendment",
"amend",
"amd",
"addendum",
"adden",
"renewal",
"extension",
"modification",
"attachment",
"agenda",
"letter",
"rate letter",
"notice",
@@ -2109,30 +2158,49 @@ class ParentChildEngine:
tmp.index, "child_rank_dest"
]
# Orphan ranks
# Orphan ranks — bucketed by grouping_key so numbering resets per group.
# Standalone orphans (alone in their grouping_key) get no sequence suffix
# ("0.<amendment>"), so numbering doesn't balloon across unrelated rows.
# Multi-orphan groups keep the "0.<amendment>.<seq>" layout with seq
# resetting per group. When the grouping_key column isn't present (e.g.
# in isolated test harnesses), every orphan is treated as its own bucket.
pm = df["is_parent"]
cm = df["assigned_parent_idx"].notna()
orphan_mask = (~pm) & (~cm)
if orphan_mask.any():
tmp = df.loc[orphan_mask, ["eff_date", "input_row_order"]].copy()
sort_cols = ["eff_date", "input_row_order"]
tmp = df.loc[orphan_mask, sort_cols].copy()
if "grouping_key" in df.columns:
tmp["_gk"] = (
df.loc[orphan_mask, "grouping_key"].fillna("__none__").astype(str)
)
else:
# No grouping info → every orphan is its own bucket (singleton).
tmp["_gk"] = [f"__row_{i}__" for i in range(len(tmp))]
tmp["eff_sort"] = tmp["eff_date"].fillna(pd.Timestamp.max)
tmp = tmp.sort_values(["eff_sort", "input_row_order"], kind="mergesort")
seq = pd.Series(np.arange(1, len(tmp) + 1, dtype=np.int32), index=tmp.index)
tmp = tmp.sort_values(
["_gk", "eff_sort", "input_row_order"], kind="mergesort"
)
seq_int = tmp.groupby("_gk", sort=False).cumcount() + 1
group_size = tmp.groupby("_gk", sort=False)["_gk"].transform("size")
if "AARETE_DERIVED_AMENDMENT_NUM" in df.columns:
amendment_str = df.loc[tmp.index, "AARETE_DERIVED_AMENDMENT_NUM"].apply(
self._normalize_amendment_val
)
df.loc[tmp.index, "orphan_rank_dest"] = (
"0." + amendment_str + "." + seq.astype("string")
).values
else:
df.loc[tmp.index, "orphan_rank_dest"] = (
"0.0." + seq.astype("string")
).values
df.loc[tmp.index, "combined_rank_dest"] = df.loc[
tmp.index, "orphan_rank_dest"
]
amendment_str = pd.Series(["0"] * len(tmp), index=tmp.index)
seq_str = seq_int.astype("string")
multi_rank = ("0." + amendment_str + "." + seq_str).values
# Singletons always have order=1 (nothing to sequence against).
single_rank = ("0." + amendment_str + ".1").values
ranks = np.where(group_size.values == 1, single_rank, multi_rank)
df.loc[tmp.index, "orphan_rank_dest"] = ranks
df.loc[tmp.index, "combined_rank_dest"] = ranks
return df
@@ -2280,20 +2348,52 @@ def qc_main(df_in, client):
engine = ParentChildEngine(cfg_dict)
df_out = engine.run(df_in)
# Export
input_cols = df_in.columns.tolist()
# Reviewer-facing column order (user-specified).
out_cols = df_out.columns.tolist()
extra_cols = [
preferred_order = [
"FILE_NAME",
"fixed_effective_date",
"contract_title_cleaned",
"PROV_GROUP_NAME_FULL_cleaned",
"PROV_GROUP_TIN",
"PROV_GROUP_NPI",
"payer_name_cleaned",
"PAYER_STATE",
"AARETE_DERIVED_AMENDMENT_NUM",
"grouping_key",
"provider_group",
"parent",
"combined_rank",
"parent_child_flag",
"assignment_reasoning",
"parent_name",
"parent_child_flag_dest",
"combined_rank_dest",
"Manual_Review_Flag",
"assignment_reasoning",
"parent_name_dest",
]
keep_cols = [c for c in input_cols if c in out_cols]
keep_cols.extend([c for c in extra_cols if c in out_cols and c not in keep_cols])
keep_cols = [c for c in preferred_order if c in out_cols]
df_final = df_out.loc[:, keep_cols].copy()
# df_final.to_csv(output_file, index=False, encoding="utf-8", quoting=1)
# logger.info("Exported: %s", output_file)
# Reviewer-friendly names, then uppercase every header so the sheet
# reads consistently. Data values keep their original case. The three
# AARETE_DERIVED_* renames relabel the header to match the source of
# truth — the values in these columns are already populated from the
# derived fields upstream in preprocessing.
df_final = df_final.rename(
columns={
"fixed_effective_date": "AARETE_DERIVED_EFFECTIVE_DT",
"PROV_GROUP_NAME_FULL_cleaned": "AARETE_DERIVED_PROVIDER_NAME",
"payer_name_cleaned": "AARETE_DERIVED_PAYER_NAME",
"provider_group": "Group",
"parent_child_flag": "Flag",
"combined_rank": "Rank",
"parent_name": "Parent_Child_Name",
"assignment_reasoning": "Reasoning",
"parent_child_flag_dest": "QC_Flag",
"combined_rank_dest": "QC_Rank",
}
)
df_final.columns = [str(c).upper() for c in df_final.columns]
return df_final
+8 -41
View File
@@ -25,7 +25,6 @@ from src.parent_child.pipeline import (
build_matching_columns_string,
flag_multi_parent_with_early_dates,
assign_child_ranks,
extract_ordinal,
)
from src.parent_child.__main__ import extract_client_name
from src.parent_child.qc import (
@@ -567,40 +566,6 @@ class TestAssignChildRanks(unittest.TestCase):
self.assertIn(".2", ranks[1])
class TestExtractOrdinal(unittest.TestCase):
"""Tests for extract_ordinal function."""
def test_extract_numeric_ordinal(self):
"""Extracts numeric ordinals like '1st', '2nd', '3rd'."""
row = pd.Series({"contract_title_cleaned": "1st Amendment to Agreement"})
result = extract_ordinal(row)
self.assertEqual(result, 1)
row = pd.Series({"contract_title_cleaned": "2nd Amendment"})
result = extract_ordinal(row)
self.assertEqual(result, 2)
row = pd.Series({"contract_title_cleaned": "3rd Addendum"})
result = extract_ordinal(row)
self.assertEqual(result, 3)
def test_extract_word_ordinal(self):
"""Extracts word ordinals like 'first', 'second', 'third'."""
row = pd.Series({"contract_title_cleaned": "First Amendment to Agreement"})
result = extract_ordinal(row)
self.assertEqual(result, 1)
row = pd.Series({"contract_title_cleaned": "Second Amendment"})
result = extract_ordinal(row)
self.assertEqual(result, 2)
def test_no_ordinal_returns_none(self):
"""Returns None when no ordinal found."""
row = pd.Series({"contract_title_cleaned": "Provider Agreement"})
result = extract_ordinal(row)
self.assertIsNone(result)
class TestConfigFactory(unittest.TestCase):
"""Tests for ConfigFactory class."""
@@ -919,7 +884,7 @@ class TestComputeRanksWithAmendmentNum(unittest.TestCase):
self.assertEqual(result.at[1, "child_rank_dest"], "1.0.2")
def test_orphan_rank_dest_with_amendment_num(self):
"""orphan_rank_dest uses amendment number instead of 0."""
"""Standalone orphan: order fixed at 1, amendment preserved."""
engine = self._make_engine()
df = self._make_base_df(
extra_cols={
@@ -928,11 +893,12 @@ class TestComputeRanksWithAmendmentNum(unittest.TestCase):
)
result = engine.compute_ranks(df)
# Index 3 is orphan with amendment=7
# Index 3 is alone in its grouping bucket, so order is fixed at 1
# (not sequenced against unrelated orphans); amendment=7 preserved.
self.assertEqual(result.at[3, "orphan_rank_dest"], "0.7.1")
def test_orphan_rank_dest_without_amendment_column(self):
"""orphan_rank_dest defaults to 0 when column is missing."""
"""Standalone orphan without amendment column gets 0.0.1."""
engine = self._make_engine()
df = self._make_base_df() # No amendment column
result = engine.compute_ranks(df)
@@ -986,7 +952,8 @@ class TestComputeRanksWithAmendmentNum(unittest.TestCase):
)
result = engine.compute_ranks(df)
self.assertEqual(result.at[3, "orphan_rank_dest"], "0.1.1") # A=1
# Standalone orphan: order fixed at 1, A=1.
self.assertEqual(result.at[3, "orphan_rank_dest"], "0.1.1")
def test_child_rank_dest_with_mixed_numeric_and_character_amendment(self):
"""child_rank_dest handles mix of numeric and character amendment values."""
@@ -1002,8 +969,8 @@ class TestComputeRanksWithAmendmentNum(unittest.TestCase):
# Index 1 (eff 2024-03-01, amendment=A=1) -> seq 2
self.assertEqual(result.at[2, "child_rank_dest"], "1.3.1")
self.assertEqual(result.at[1, "child_rank_dest"], "1.1.2") # A=1
# Index 3 is orphan with amendment=B=2
self.assertEqual(result.at[3, "orphan_rank_dest"], "0.2.1") # B=2
# Standalone orphan: order fixed at 1, B=2.
self.assertEqual(result.at[3, "orphan_rank_dest"], "0.2.1")
class TestNormalizeAmendmentVal(unittest.TestCase):