From e4650064aa618cd2371c55149fa05b2359797f35 Mon Sep 17 00:00:00 2001 From: Rahul Ailaboina Date: Mon, 27 Apr 2026 15:04:07 +0000 Subject: [PATCH] Merged in feature/PC_logic_cleanup_output (pull request #984) Feature/PC logic cleanup output * Few tweaks PC_logics * Fixed orphan_ranking * black formatting * Changes on output field and ranking method * updated few hotfixes * black format fix Approved-by: Katon Minhas --- src/parent_child/column_mapper.py | 22 ++- src/parent_child/pipeline.py | 223 ++++++++++++++++++++++-------- src/parent_child/preprocessing.py | 61 ++++++-- src/parent_child/qc.py | 190 +++++++++++++++++++------ src/tests/test_parent_child.py | 49 ++----- 5 files changed, 386 insertions(+), 159 deletions(-) diff --git a/src/parent_child/column_mapper.py b/src/parent_child/column_mapper.py index 90d4c25..5f16788 100644 --- a/src/parent_child/column_mapper.py +++ b/src/parent_child/column_mapper.py @@ -38,6 +38,9 @@ class ColumnMapper: r"^agreement.*title$", r"^doc.*title$", ], + "AARETE_DERIVED_PAYER_NAME": [ + r"^aarete_derived_payer_name$", + ], "PAYER_NAME": [ r"^payer.*name$", r"^payer$", @@ -74,6 +77,15 @@ class ColumnMapper: r"^national.*provider.*id$", r"^provider.*npi$", ], + "PROV_OTHER_TIN": [ + r"^prov.*other.*tin$", + ], + "PROV_OTHER_NPI": [ + r"^prov.*other.*npi$", + ], + "PROV_OTHER_NAME_FULL": [ + r"^prov.*other.*name.*full$", + ], "EFFECTIVE_DATE": [ r"aarete_derived_effective_dt", r"^.*effective.*date$", @@ -124,6 +136,10 @@ class ColumnMapper: "PROV_GROUP_NPI", "AARETE_DERIVED_AMENDMENT_NUM", "AARETE_DERIVED_PROVIDER_NAME", + "AARETE_DERIVED_PAYER_NAME", + "PROV_OTHER_TIN", + "PROV_OTHER_NPI", + "PROV_OTHER_NAME_FULL", ] def __init__(self, df: pd.DataFrame): @@ -154,9 +170,9 @@ class ColumnMapper: Returns: Matched column name or None """ - for col in available_columns: - normalized_col = self._normalize_column_name(col) - for pattern in patterns: + for pattern in patterns: + for col in available_columns: + normalized_col = self._normalize_column_name(col) if re.match(pattern, normalized_col, re.IGNORECASE): return col return None diff --git a/src/parent_child/pipeline.py b/src/parent_child/pipeline.py index d5ad7b6..00fa25b 100644 --- a/src/parent_child/pipeline.py +++ b/src/parent_child/pipeline.py @@ -1,5 +1,6 @@ from __future__ import annotations +import json import pandas as pd import re from pathlib import Path @@ -27,6 +28,60 @@ _MODULE_DIR = Path(__file__).resolve().parent _CONSTANTS_DIR = _MODULE_DIR.parent / "constants" / "mappings" +def _build_provider_group_label(df: pd.DataFrame) -> pd.Series: + """Human-readable grouping label for reviewers. + + Uses the cleaned provider name, title-cased. When the same name appears + with more than one TIN in the dataset, disambiguate by appending the TIN + (e.g. "Storr Medical Center - 123456789"). Falls back to grouping_key + for rows with no usable provider name. + """ + name_col = "PROV_GROUP_NAME_FULL_cleaned" + tin_col = "PROV_GROUP_TIN" + + if name_col not in df.columns: + return df.get("grouping_key", pd.Series([""] * len(df), index=df.index)).astype( + "string" + ) + + name = df[name_col].fillna("").astype(str).str.strip() + tin = ( + df[tin_col].fillna("").astype(str).str.strip() + if tin_col in df.columns + else pd.Series([""] * len(df), index=df.index) + ) + + # Count distinct non-empty TINs per name; names with >1 TIN need + # disambiguation. + mask = (name != "") & (tin != "") + tins_per_name: Dict[str, int] = ( + pd.DataFrame({"_n": name[mask], "_t": tin[mask]}) + .groupby("_n")["_t"] + .nunique() + .to_dict() + ) + + fallback = ( + df["grouping_key"].fillna("").astype(str) + if "grouping_key" in df.columns + else pd.Series([""] * len(df), index=df.index) + ) + + def _label(n: str, t: str, fb: str) -> str: + if not n: + return fb + display = n.title() + if tins_per_name.get(n, 0) > 1 and t: + return f"{display} - {t}" + return display + + return pd.Series( + [_label(n, t, fb) for n, t, fb in zip(name, tin, fallback)], + index=df.index, + dtype="string", + ) + + def create_grouping_key_and_tier(df): """ Creates grouping key based on available non-null values in PROV_GROUP_TIN, @@ -322,6 +377,41 @@ def build_grouping_string(child_row, grouping_cols): return " and ".join(parts) +def _parse_identifier_set(value) -> set: + """Normalize a TIN/NPI field into a set of individual ID strings. + + Source data often stores identifiers as JSON-encoded lists + (e.g. `'["133757370", "840611484"]'`) because a single contract can + span multiple legal entities. Plain scalars (`'133757370'`) are also + supported. Returns an empty set for NaN / empty / "nan" values. + + Normalizing to a set lets downstream matchers use intersection + instead of string equality — two rows that share a TIN compare as + compatible even if one encodes it as `["A", "B"]` and the other as + `["B", "C", "D"]`. + """ + if value is None: + return set() + try: + if pd.isna(value): + return set() + except (TypeError, ValueError): + pass + s = str(value).strip() + if not s or s.lower() == "nan": + return set() + # JSON list (the common case from upstream extraction). + if s.startswith("[") and s.endswith("]"): + try: + parsed = json.loads(s) + if isinstance(parsed, list): + return {str(v).strip() for v in parsed if str(v).strip()} + except (json.JSONDecodeError, ValueError): + pass + # Plain scalar — wrap as a singleton set. + return {s} + + def check_grouping_compatibility(parent_key, child_key): """ Checks if a parent's grouping key is compatible with a child's grouping key. @@ -329,6 +419,9 @@ def check_grouping_compatibility(parent_key, child_key): Rules: - Extract values from both keys - Compatible if they share AT LEAST ONE matching value + - For TIN / NPI, values may be JSON-list strings (e.g. `["A","B"]`); + we compare as sets so a parent holding `["A","B"]` is compatible + with a child holding `["B","C"]` via the shared TIN "B" - This allows cross-tier matching (e.g., Tier 1 parent can match Tier 3 child via TIN) Examples: @@ -336,9 +429,9 @@ def check_grouping_compatibility(parent_key, child_key): Child: "TIN:123|NPI:456" Result: ✅ Compatible (share TIN and NPI) - Parent: "TIN:123|NPI:456|NAME:ABC" - Child: "TIN:123|NPI:999" - Result: ✅ Compatible (share TIN:123, even though NPI differs) + Parent: "TIN:[\"123\",\"456\"]" + Child: "TIN:[\"456\",\"789\"]" + Result: ✅ Compatible (share TIN 456 in the list) Parent: "TIN:123|NPI:456|NAME:ABC" Child: "TIN:999" @@ -361,10 +454,17 @@ def check_grouping_compatibility(parent_key, child_key): child_dict = parse_key(child_key) # Check if they share AT LEAST ONE matching value - for key, child_value in child_dict.items(): - parent_value = parent_dict.get(key) - if parent_value == child_value: - return True # Found at least one match! + for field, child_value in child_dict.items(): + parent_value = parent_dict.get(field) + if parent_value is None: + continue + if field in ("TIN", "NPI"): + # Set-intersection comparison for list-encoded identifiers. + if _parse_identifier_set(parent_value) & _parse_identifier_set(child_value): + return True + else: + if parent_value == child_value: + return True return False # No matching values found @@ -380,6 +480,11 @@ def create_df_per_tin_presence(df, is_tin_present): def filter_on_exclude_keywords(df, is_tin_present): """ Flags rows as 'parent' based on exclusion keywords. + + `contract_title_cleaned` is the primary signal. The cleaned filename is + consulted only when the title is absent — otherwise a row with no title + but an obvious child-type filename (e.g. "... Amendment.pdf", "... AMD ...") + would silently default to parent. When a title is present, it wins. """ df = df.copy() df["parent"] = True @@ -392,9 +497,28 @@ def filter_on_exclude_keywords(df, is_tin_present): r"\b(?:" + "|".join(re.escape(kw) for kw in exclude_keywords_list) + r")\b" ) - df["parent"] = ~( - df["contract_title_cleaned"].str.contains(exclude_pattern, case=False, na=False) + title = df["contract_title_cleaned"].astype("string").fillna("").str.strip() + title_has_kw = title.str.contains(exclude_pattern, case=False, na=False) + + title_missing = (title == "") | (title.str.lower() == "nan") + if "contract_name_cleaned" in df.columns: + filename_source = df["contract_name_cleaned"] + else: + filename_source = df.get("FILE_NAME", pd.Series([""] * len(df), index=df.index)) + # Normalize the filename so regex word boundaries land cleanly. Digits, + # underscores and punctuation are regex word chars, which means `\bamd\b` + # won't match "AMD1" or "_AMD#". Replacing non-letters with spaces first + # gives us `BCBSAZ AMD to MSA` and `\bamd\b` catches it. + filename_normalized = ( + filename_source.astype("string") + .fillna("") + .str.replace(r"[^a-zA-Z]+", " ", regex=True) ) + filename_has_kw = filename_normalized.str.contains( + exclude_pattern, case=False, na=False + ) + + df["parent"] = ~(title_has_kw | (title_missing & filename_has_kw)) return df @@ -478,23 +602,20 @@ def build_matching_columns_string(parent_row, child_row): """ matched_parts = [] - # Check TIN - parent_tin = parent_row.get("PROV_GROUP_TIN") - child_tin = child_row.get("PROV_GROUP_TIN") - if pd.notna(parent_tin) and pd.notna(child_tin) and parent_tin == child_tin: - if isinstance(child_tin, float) and child_tin == int(child_tin): - matched_parts.append(f"TIN ({int(child_tin)})") - else: - matched_parts.append(f"TIN ({child_tin})") + # Check TIN — parent/child may each encode multiple TINs as JSON lists; + # report the intersection so the reasoning shows which ID actually matched. + shared_tins = _parse_identifier_set(parent_row.get("PROV_GROUP_TIN")) & ( + _parse_identifier_set(child_row.get("PROV_GROUP_TIN")) + ) + if shared_tins: + matched_parts.append(f"TIN ({', '.join(sorted(shared_tins))})") # Check NPI - parent_npi = parent_row.get("PROV_GROUP_NPI") - child_npi = child_row.get("PROV_GROUP_NPI") - if pd.notna(parent_npi) and pd.notna(child_npi) and parent_npi == child_npi: - if isinstance(child_npi, float) and child_npi == int(child_npi): - matched_parts.append(f"NPI ({int(child_npi)})") - else: - matched_parts.append(f"NPI ({child_npi})") + shared_npis = _parse_identifier_set(parent_row.get("PROV_GROUP_NPI")) & ( + _parse_identifier_set(child_row.get("PROV_GROUP_NPI")) + ) + if shared_npis: + matched_parts.append(f"NPI ({', '.join(sorted(shared_npis))})") # Check PROV_NAME parent_name = parent_row.get("PROV_GROUP_NAME_FULL") @@ -993,7 +1114,9 @@ def assign_child_ranks(df, grouper_col="grouping_key"): assigned_rank = child_row.get("assigned_parent_rank") if assigned_rank == ASSIGNMENT_NO_PARENT or pd.isna(assigned_rank): - children.at[idx, "_parent_identity"] = "orphan" + gk = child_row.get("grouping_key") + gk_str = str(gk) if pd.notna(gk) and gk not in (None, "") else "__none__" + children.at[idx, "_parent_identity"] = f"orphan__{gk_str}" else: # Find any parent with this rank that's compatible with the child found = False @@ -1016,7 +1139,10 @@ def assign_child_ranks(df, grouper_col="grouping_key"): has_amendment_col = "AARETE_DERIVED_AMENDMENT_NUM" in children.columns for parent_id, grp_df in children.groupby("_parent_identity"): - if parent_id == "orphan": + is_orphan_bucket = isinstance(parent_id, str) and parent_id.startswith( + "orphan__" + ) + if is_orphan_bucket: parent_rank_str = "0" else: # Extract parent_rank from the child's assigned_parent_rank @@ -1041,6 +1167,9 @@ def assign_child_ranks(df, grouper_col="grouping_key"): grp_df["sort_date"] = grp_df["fixed_effective_date"].fillna(pd.Timestamp.max) grp_df = grp_df.sort_values("sort_date") + # Standalone orphan (alone in its grouping_key): no sequence suffix. + standalone_orphan = is_orphan_bucket and len(grp_df) == 1 + # Assign sequential ranks using amendment number instead of 0 for i, idx in enumerate(grp_df.index, start=1): amendment_val = 0 @@ -1055,7 +1184,12 @@ def assign_child_ranks(df, grouper_col="grouping_key"): amendment_val = ord(raw_str.upper()) - ord("A") + 1 else: amendment_val = raw - rank_val = f"{parent_rank_str}.{amendment_val}.{i}" + if standalone_orphan: + # Only one orphan in this group — fixed order=1 (never + # sequences up against unrelated rows). + rank_val = f"{parent_rank_str}.{amendment_val}.1" + else: + rank_val = f"{parent_rank_str}.{amendment_val}.{i}" children.loc[idx, "child_rank"] = rank_val # Clean up temp column @@ -1080,33 +1214,6 @@ def assign_child_ranks(df, grouper_col="grouping_key"): return result -def extract_ordinal(row): - """ - Extracts ordinal numbers from contract titles. - """ - num_dict = io_utils.convert_json_to_dict( - str(_CONSTANTS_DIR / "numeric_mappings.json") - ) - word_to_num = num_dict.get("ordinal_word_to_number", {}) - cardinal_to_num = num_dict.get("cardinal_word_to_number", {}) - ordinal_pattern = num_dict.get("ordinal_regex_pattern", "") - - all_words_to_num = {**word_to_num, **cardinal_to_num} - text = str(row["contract_title_cleaned"]).lower() - - for word, num in all_words_to_num.items(): - if word in text: - return num - - num_match = re.search(ordinal_pattern, text) - if num_match: - value = re.sub(r"(st|nd|rd|th)$", "", num_match.group(0)) - if value.isdigit(): - return int(value) - - return None - - def parent_child_mapping( cleaned_df: pd.DataFrame, original_df: pd.DataFrame ) -> Tuple[pd.DataFrame, pd.DataFrame]: @@ -1281,7 +1388,6 @@ def parent_child_mapping( # Assign child ranks pc_df = assign_child_ranks(pc_df, grouper_col="grouping_key") - pc_df["numbering"] = pc_df.apply(extract_ordinal, axis=1) # Set parent_child_flag pc_df["parent_child_flag"] = pc_df["parent"].map({True: "Parent", False: "Child"}) @@ -1334,6 +1440,8 @@ def parent_child_mapping( pc_df["FILE_NAME"] = pc_df["FILE_NAME"].str.replace("Filename: ", "") + pc_df["provider_group"] = _build_provider_group_label(pc_df) + # Final summary logging.info("=" * 80) logging.info("FINAL SUMMARY") @@ -1372,7 +1480,8 @@ def parent_child_mapping( columns=["contract_name_cleaned", "fixed_effective_date"] ) - # Build pc_df output columns dynamically based on available columns + # Build pc_df output columns dynamically based on available columns. + cols_to_keep_pc_tab = [ "FILE_NAME", "fixed_effective_date", @@ -1384,9 +1493,9 @@ def parent_child_mapping( "PAYER_STATE", "AARETE_DERIVED_AMENDMENT_NUM", "grouping_key", + "provider_group", "parent", "combined_rank", - "numbering", "parent_child_flag", "assignment_reasoning", "parent_name", diff --git a/src/parent_child/preprocessing.py b/src/parent_child/preprocessing.py index 558ed3b..bcce01d 100644 --- a/src/parent_child/preprocessing.py +++ b/src/parent_child/preprocessing.py @@ -841,21 +841,27 @@ def parent_child_preprocessing( one_to_one_df = create_one_to_one_df(all_fields_df) one_to_one_df = clean_contract_title(one_to_one_df) - one_to_one_df["payer_name_cleaned"] = one_to_one_df[payer_name_col].apply( - lambda x: clean_payer_name(str(x), ["inc", "llc", "dba"]) - ) - # Use AARETE_DERIVED_PROVIDER_NAME if available (already cleaned/standardized upstream) - # Otherwise fall back to raw PROV_GROUP_NAME_FULL with full cleaning + # Payer: prefer AARETE_DERIVED_PAYER_NAME (already cleaned upstream). + # Fall back to raw PAYER_NAME with cleaning only when derived is absent. + if "AARETE_DERIVED_PAYER_NAME" in one_to_one_df.columns: + logging.info("Using AARETE_DERIVED_PAYER_NAME as-is (no cleaning)") + one_to_one_df["payer_name_cleaned"] = one_to_one_df["AARETE_DERIVED_PAYER_NAME"] + else: + logging.info("AARETE_DERIVED_PAYER_NAME not found, cleaning raw PAYER_NAME") + one_to_one_df["payer_name_cleaned"] = one_to_one_df[payer_name_col].apply( + lambda x: clean_payer_name(str(x), ["inc", "llc", "dba"]) + ) + + # Provider: prefer AARETE_DERIVED_PROVIDER_NAME as-is (already cleaned upstream). + # Fall back to raw PROV_GROUP_NAME_FULL with full cleaning only when derived is absent. if "AARETE_DERIVED_PROVIDER_NAME" in one_to_one_df.columns: - logging.info( - "Using AARETE_DERIVED_PROVIDER_NAME for provider grouping (lightweight cleaning)" - ) - one_to_one_df = lightweight_clean_provider_name( - one_to_one_df, - provider_col="AARETE_DERIVED_PROVIDER_NAME", - ) - one_to_one_df = standardize_provider_groups(one_to_one_df) + logging.info("Using AARETE_DERIVED_PROVIDER_NAME as-is (no cleaning)") + one_to_one_df["PROV_GROUP_NAME_FULL_cleaned"] = one_to_one_df[ + "AARETE_DERIVED_PROVIDER_NAME" + ] + if "DBA_Name" not in one_to_one_df.columns: + one_to_one_df["DBA_Name"] = "" else: logging.info( "AARETE_DERIVED_PROVIDER_NAME not found, falling back to raw provider name cleaning" @@ -864,6 +870,35 @@ def parent_child_preprocessing( one_to_one_df = update_group(one_to_one_df) one_to_one_df = standardize_provider_groups(one_to_one_df) + # Per-row fallback: when a row has no PROV_GROUP_* identifiers at all, + # promote PROV_OTHER_* into the PROV_GROUP_* slots so the downstream + # grouping-key / tier logic treats them as real identifiers instead of + # dropping the row to the filename-based fallback. + other_cols = ("PROV_OTHER_TIN", "PROV_OTHER_NPI", "PROV_OTHER_NAME_FULL") + if any(c in one_to_one_df.columns for c in other_cols): + + def _is_blank(series): + as_str = series.astype("string").fillna("").str.strip() + return (as_str == "") | (as_str.str.lower() == "nan") + + blank_group = ( + _is_blank(one_to_one_df["PROV_GROUP_TIN"]) + & _is_blank(one_to_one_df["PROV_GROUP_NPI"]) + & _is_blank(one_to_one_df["PROV_GROUP_NAME_FULL_cleaned"]) + ) + for src_col, dst_col in ( + ("PROV_OTHER_TIN", "PROV_GROUP_TIN"), + ("PROV_OTHER_NPI", "PROV_GROUP_NPI"), + ("PROV_OTHER_NAME_FULL", "PROV_GROUP_NAME_FULL_cleaned"), + ): + if src_col in one_to_one_df.columns: + one_to_one_df.loc[blank_group, dst_col] = one_to_one_df.loc[ + blank_group, src_col + ] + logging.info( + f"PROV_OTHER_* fallback applied to {int(blank_group.sum())} rows with no group identifiers" + ) + # Determine which fields to consolidate based on available columns fields_to_consolidate = [] if "LOB" in all_fields_df.columns: diff --git a/src/parent_child/qc.py b/src/parent_child/qc.py index 929c302..c66c88f 100644 --- a/src/parent_child/qc.py +++ b/src/parent_child/qc.py @@ -2,6 +2,7 @@ Parent-Child Contract Mapping and Ranking System """ +import json import re import sys import time @@ -235,9 +236,37 @@ class TextProcessor: normalized = re.sub(r"[^0-9A-Za-z]", "", value.strip()) return normalized.upper() + @staticmethod + def _identifier_parts(val_str: str) -> List[str]: + """Split a raw identifier value into individual IDs. + + Handles: + - JSON lists: '["275440611", "274211365"]' → ['275440611', '274211365'] + - Pipe-separated: '275440611|274211365' → ['275440611', '274211365'] + - Plain scalars: '275440611' → ['275440611'] + + JSON-list parsing keeps multi-TIN contracts readable in the grouping + key — without it, `normalize_single_identifier` strips the brackets + and commas and concatenates every digit into one blob. + """ + if val_str.startswith("[") and val_str.endswith("]"): + try: + parsed = json.loads(val_str) + if isinstance(parsed, list): + return [str(p) for p in parsed] + except (json.JSONDecodeError, ValueError): + pass + if "|" in val_str: + return val_str.split("|") + return [val_str] + @staticmethod def normalize_identifier(value: Any) -> str: - """Normalize TIN/NPI identifiers (handles pipe-separated values).""" + """Normalize TIN/NPI identifiers. + + Accepts JSON-list, pipe-separated, or scalar input. Returns the + unique IDs joined by `|`, sorted for deterministic bucketing. + """ if value is None or pd.isna(value): return "" @@ -245,17 +274,11 @@ class TextProcessor: if not val_str: return "" - if "|" in val_str: - parts = [ - TextProcessor.normalize_single_identifier(p) for p in val_str.split("|") - ] - seen = set() - unique_parts = [] - for p in parts: - if p and p not in seen: - seen.add(p) - unique_parts.append(p) - return "|".join(unique_parts) if unique_parts else "" + parts = TextProcessor._identifier_parts(val_str) + if len(parts) > 1: + normalized = [TextProcessor.normalize_single_identifier(p) for p in parts] + unique = sorted({p for p in normalized if p}) + return "|".join(unique) if unique else "" return TextProcessor.normalize_single_identifier(val_str) @@ -266,7 +289,10 @@ class TextProcessor: @staticmethod def get_identifier_values(value: Any) -> Set[str]: - """Extract all individual identifier values (handles pipe-separated).""" + """Extract all individual identifier values. + + Handles JSON-list, pipe-separated, and scalar input. + """ if value is None or pd.isna(value): return set() @@ -274,11 +300,10 @@ class TextProcessor: if not val_str: return set() - if "|" in val_str: - parts = [ - TextProcessor.normalize_single_identifier(p) for p in val_str.split("|") - ] - return {p for p in parts if p} + parts = TextProcessor._identifier_parts(val_str) + if len(parts) > 1: + normalized = [TextProcessor.normalize_single_identifier(p) for p in parts] + return {p for p in normalized if p} normalized = TextProcessor.normalize_single_identifier(val_str) return {normalized} if normalized else set() @@ -646,9 +671,15 @@ class ParentIdentifier: raise ValueError(f"Unknown parent identification strategy: {strategy_type}") def _identify_by_keyword(self, df: pd.DataFrame, config: Dict) -> pd.Series: - """Identify parents by excluding keywords.""" + """Identify parents by excluding keywords. + + `field` (typically the cleaned contract title) is the primary signal. + `fallback_field` (typically the filename) is consulted only when the + primary field is blank/missing for a row. + """ field = config.get("field") exclude_keywords = config.get("exclude_keywords", []) + fallback_field = config.get("fallback_field") if not field or field not in df.columns: raise ValueError(f"keyword strategy missing field: {field}") if not exclude_keywords: @@ -661,9 +692,24 @@ class ParentIdentifier: self.pattern_cache[pattern_str] = re.compile(pattern_str, re.IGNORECASE) compiled = self.pattern_cache[pattern_str] - has_keyword = ( - df[field].astype("string").str.contains(compiled, na=False, regex=True) - ) + primary = df[field].astype("string").fillna("").str.strip() + has_keyword = primary.str.contains(compiled, na=False, regex=True) + + if fallback_field and fallback_field in df.columns: + primary_missing = (primary == "") | (primary.str.lower() == "nan") + # Normalize the fallback (typically FILE_NAME) so regex word + # boundaries land cleanly. Digits, underscores and punctuation + # are regex word chars, so `\bamd\b` won't match "AMD1" or + # "_AMD#" without this preprocessing step. + fallback_has_kw = ( + df[fallback_field] + .astype("string") + .fillna("") + .str.replace(r"[^a-zA-Z]+", " ", regex=True) + .str.contains(compiled, na=False, regex=True) + ) + has_keyword = has_keyword | (primary_missing & fallback_has_kw) + return ~has_keyword def _identify_by_field_value(self, df: pd.DataFrame, config: Dict) -> pd.Series: @@ -687,7 +733,7 @@ class ParentIdentifier: if not field or field not in df.columns: raise ValueError(f"source_label strategy missing field: {field}") labels = TextProcessor.normalize_label(df[field]) - return (labels == "Parent").fillna(False) + return (labels == "PARENT").fillna(False) def _identify_bcbs_custom(self, df: pd.DataFrame, config: Dict) -> pd.Series: """ @@ -1609,16 +1655,19 @@ class ConfigFactory: parent_identification={ "strategy_type": "keyword", "field": "title_clean", + "fallback_field": "FILE_NAME", "exclude_keywords": [ "exhibit", "amendment", "amend", + "amd", "addendum", "adden", "renewal", "extension", "modification", "attachment", + "agenda", "letter", "rate letter", "notice", @@ -2109,30 +2158,49 @@ class ParentChildEngine: tmp.index, "child_rank_dest" ] - # Orphan ranks + # Orphan ranks — bucketed by grouping_key so numbering resets per group. + # Standalone orphans (alone in their grouping_key) get no sequence suffix + # ("0."), so numbering doesn't balloon across unrelated rows. + # Multi-orphan groups keep the "0.." layout with seq + # resetting per group. When the grouping_key column isn't present (e.g. + # in isolated test harnesses), every orphan is treated as its own bucket. pm = df["is_parent"] cm = df["assigned_parent_idx"].notna() orphan_mask = (~pm) & (~cm) if orphan_mask.any(): - tmp = df.loc[orphan_mask, ["eff_date", "input_row_order"]].copy() + sort_cols = ["eff_date", "input_row_order"] + tmp = df.loc[orphan_mask, sort_cols].copy() + if "grouping_key" in df.columns: + tmp["_gk"] = ( + df.loc[orphan_mask, "grouping_key"].fillna("__none__").astype(str) + ) + else: + # No grouping info → every orphan is its own bucket (singleton). + tmp["_gk"] = [f"__row_{i}__" for i in range(len(tmp))] tmp["eff_sort"] = tmp["eff_date"].fillna(pd.Timestamp.max) - tmp = tmp.sort_values(["eff_sort", "input_row_order"], kind="mergesort") - seq = pd.Series(np.arange(1, len(tmp) + 1, dtype=np.int32), index=tmp.index) + tmp = tmp.sort_values( + ["_gk", "eff_sort", "input_row_order"], kind="mergesort" + ) + + seq_int = tmp.groupby("_gk", sort=False).cumcount() + 1 + group_size = tmp.groupby("_gk", sort=False)["_gk"].transform("size") + if "AARETE_DERIVED_AMENDMENT_NUM" in df.columns: amendment_str = df.loc[tmp.index, "AARETE_DERIVED_AMENDMENT_NUM"].apply( self._normalize_amendment_val ) - df.loc[tmp.index, "orphan_rank_dest"] = ( - "0." + amendment_str + "." + seq.astype("string") - ).values else: - df.loc[tmp.index, "orphan_rank_dest"] = ( - "0.0." + seq.astype("string") - ).values - df.loc[tmp.index, "combined_rank_dest"] = df.loc[ - tmp.index, "orphan_rank_dest" - ] + amendment_str = pd.Series(["0"] * len(tmp), index=tmp.index) + + seq_str = seq_int.astype("string") + multi_rank = ("0." + amendment_str + "." + seq_str).values + # Singletons always have order=1 (nothing to sequence against). + single_rank = ("0." + amendment_str + ".1").values + ranks = np.where(group_size.values == 1, single_rank, multi_rank) + + df.loc[tmp.index, "orphan_rank_dest"] = ranks + df.loc[tmp.index, "combined_rank_dest"] = ranks return df @@ -2280,20 +2348,52 @@ def qc_main(df_in, client): engine = ParentChildEngine(cfg_dict) df_out = engine.run(df_in) - # Export - input_cols = df_in.columns.tolist() + # Reviewer-facing column order (user-specified). out_cols = df_out.columns.tolist() - extra_cols = [ + preferred_order = [ + "FILE_NAME", + "fixed_effective_date", + "contract_title_cleaned", + "PROV_GROUP_NAME_FULL_cleaned", + "PROV_GROUP_TIN", + "PROV_GROUP_NPI", + "payer_name_cleaned", + "PAYER_STATE", + "AARETE_DERIVED_AMENDMENT_NUM", + "grouping_key", + "provider_group", + "parent", + "combined_rank", + "parent_child_flag", + "assignment_reasoning", + "parent_name", "parent_child_flag_dest", "combined_rank_dest", "Manual_Review_Flag", - "assignment_reasoning", - "parent_name_dest", ] - keep_cols = [c for c in input_cols if c in out_cols] - keep_cols.extend([c for c in extra_cols if c in out_cols and c not in keep_cols]) + + keep_cols = [c for c in preferred_order if c in out_cols] df_final = df_out.loc[:, keep_cols].copy() - # df_final.to_csv(output_file, index=False, encoding="utf-8", quoting=1) - # logger.info("Exported: %s", output_file) + + # Reviewer-friendly names, then uppercase every header so the sheet + # reads consistently. Data values keep their original case. The three + # AARETE_DERIVED_* renames relabel the header to match the source of + # truth — the values in these columns are already populated from the + # derived fields upstream in preprocessing. + df_final = df_final.rename( + columns={ + "fixed_effective_date": "AARETE_DERIVED_EFFECTIVE_DT", + "PROV_GROUP_NAME_FULL_cleaned": "AARETE_DERIVED_PROVIDER_NAME", + "payer_name_cleaned": "AARETE_DERIVED_PAYER_NAME", + "provider_group": "Group", + "parent_child_flag": "Flag", + "combined_rank": "Rank", + "parent_name": "Parent_Child_Name", + "assignment_reasoning": "Reasoning", + "parent_child_flag_dest": "QC_Flag", + "combined_rank_dest": "QC_Rank", + } + ) + df_final.columns = [str(c).upper() for c in df_final.columns] return df_final diff --git a/src/tests/test_parent_child.py b/src/tests/test_parent_child.py index 0e1527f..bb8b6e0 100644 --- a/src/tests/test_parent_child.py +++ b/src/tests/test_parent_child.py @@ -25,7 +25,6 @@ from src.parent_child.pipeline import ( build_matching_columns_string, flag_multi_parent_with_early_dates, assign_child_ranks, - extract_ordinal, ) from src.parent_child.__main__ import extract_client_name from src.parent_child.qc import ( @@ -567,40 +566,6 @@ class TestAssignChildRanks(unittest.TestCase): self.assertIn(".2", ranks[1]) -class TestExtractOrdinal(unittest.TestCase): - """Tests for extract_ordinal function.""" - - def test_extract_numeric_ordinal(self): - """Extracts numeric ordinals like '1st', '2nd', '3rd'.""" - row = pd.Series({"contract_title_cleaned": "1st Amendment to Agreement"}) - result = extract_ordinal(row) - self.assertEqual(result, 1) - - row = pd.Series({"contract_title_cleaned": "2nd Amendment"}) - result = extract_ordinal(row) - self.assertEqual(result, 2) - - row = pd.Series({"contract_title_cleaned": "3rd Addendum"}) - result = extract_ordinal(row) - self.assertEqual(result, 3) - - def test_extract_word_ordinal(self): - """Extracts word ordinals like 'first', 'second', 'third'.""" - row = pd.Series({"contract_title_cleaned": "First Amendment to Agreement"}) - result = extract_ordinal(row) - self.assertEqual(result, 1) - - row = pd.Series({"contract_title_cleaned": "Second Amendment"}) - result = extract_ordinal(row) - self.assertEqual(result, 2) - - def test_no_ordinal_returns_none(self): - """Returns None when no ordinal found.""" - row = pd.Series({"contract_title_cleaned": "Provider Agreement"}) - result = extract_ordinal(row) - self.assertIsNone(result) - - class TestConfigFactory(unittest.TestCase): """Tests for ConfigFactory class.""" @@ -919,7 +884,7 @@ class TestComputeRanksWithAmendmentNum(unittest.TestCase): self.assertEqual(result.at[1, "child_rank_dest"], "1.0.2") def test_orphan_rank_dest_with_amendment_num(self): - """orphan_rank_dest uses amendment number instead of 0.""" + """Standalone orphan: order fixed at 1, amendment preserved.""" engine = self._make_engine() df = self._make_base_df( extra_cols={ @@ -928,11 +893,12 @@ class TestComputeRanksWithAmendmentNum(unittest.TestCase): ) result = engine.compute_ranks(df) - # Index 3 is orphan with amendment=7 + # Index 3 is alone in its grouping bucket, so order is fixed at 1 + # (not sequenced against unrelated orphans); amendment=7 preserved. self.assertEqual(result.at[3, "orphan_rank_dest"], "0.7.1") def test_orphan_rank_dest_without_amendment_column(self): - """orphan_rank_dest defaults to 0 when column is missing.""" + """Standalone orphan without amendment column gets 0.0.1.""" engine = self._make_engine() df = self._make_base_df() # No amendment column result = engine.compute_ranks(df) @@ -986,7 +952,8 @@ class TestComputeRanksWithAmendmentNum(unittest.TestCase): ) result = engine.compute_ranks(df) - self.assertEqual(result.at[3, "orphan_rank_dest"], "0.1.1") # A=1 + # Standalone orphan: order fixed at 1, A=1. + self.assertEqual(result.at[3, "orphan_rank_dest"], "0.1.1") def test_child_rank_dest_with_mixed_numeric_and_character_amendment(self): """child_rank_dest handles mix of numeric and character amendment values.""" @@ -1002,8 +969,8 @@ class TestComputeRanksWithAmendmentNum(unittest.TestCase): # Index 1 (eff 2024-03-01, amendment=A=1) -> seq 2 self.assertEqual(result.at[2, "child_rank_dest"], "1.3.1") self.assertEqual(result.at[1, "child_rank_dest"], "1.1.2") # A=1 - # Index 3 is orphan with amendment=B=2 - self.assertEqual(result.at[3, "orphan_rank_dest"], "0.2.1") # B=2 + # Standalone orphan: order fixed at 1, B=2. + self.assertEqual(result.at[3, "orphan_rank_dest"], "0.2.1") class TestNormalizeAmendmentVal(unittest.TestCase):