Fix REIMB_TERM being converted to list in lesser_of_distribution

- Fixed prompt_lesser_of_distribution to extract string from list when parser returns list
- Added debug print statements to trace REIMB_TERM format through processing pipeline
- Removed defensive normalization checks that are no longer needed upstream
- Added debug output in reimbursement_level and methodology_breakout_single_row to track data flow
This commit is contained in:
ppanchigar
2026-02-03 17:15:14 -06:00
parent 4c0d2930d5
commit a633f67c62
10 changed files with 343 additions and 72 deletions
@@ -119,6 +119,32 @@ def prompt_reimbursement_primary(
try:
llm_answer_final = _parser(llm_answer_raw)
# Normalize SERVICE_TERM and REIMB_TERM to strings immediately after parsing
# These should always be strings per row, not lists
if isinstance(llm_answer_final, list):
for answer_dict in llm_answer_final:
if isinstance(answer_dict, dict):
# SERVICE_TERM should be a string per row
if "SERVICE_TERM" in answer_dict:
service_term = answer_dict["SERVICE_TERM"]
if isinstance(service_term, list):
# If list, join with comma (shouldn't happen, but defensive)
answer_dict["SERVICE_TERM"] = ", ".join(str(item) for item in service_term if item)
elif service_term is not None:
answer_dict["SERVICE_TERM"] = str(service_term)
else:
answer_dict["SERVICE_TERM"] = ""
# REIMB_TERM should be a string per row
if "REIMB_TERM" in answer_dict:
reimb_term = answer_dict["REIMB_TERM"]
if isinstance(reimb_term, list):
# If list, join with comma (shouldn't happen, but defensive)
answer_dict["REIMB_TERM"] = ", ".join(str(item) for item in reimb_term if item)
elif reimb_term is not None:
answer_dict["REIMB_TERM"] = str(reimb_term)
else:
answer_dict["REIMB_TERM"] = ""
except ValueError as e:
logging.error(f"Error parsing LLM response: {e}")
logging.error(f"Raw LLM output: {llm_answer_raw}")
@@ -151,6 +177,19 @@ def prompt_methodology_breakout(
logging.debug(f"LLM Response for {filename}: {llm_response}")
try:
methodology_breakout_answers = _parser(llm_response)
# Normalize single-value fields that should be strings (not lists)
# FEE_SCHEDULE should be a string per methodology breakout result
if isinstance(methodology_breakout_answers, list):
for methodology_dict in methodology_breakout_answers:
if isinstance(methodology_dict, dict) and "FEE_SCHEDULE" in methodology_dict:
fee_schedule = methodology_dict["FEE_SCHEDULE"]
if isinstance(fee_schedule, list):
# If list, take first element (shouldn't happen, but defensive)
methodology_dict["FEE_SCHEDULE"] = fee_schedule[0] if fee_schedule else ""
elif fee_schedule is not None:
methodology_dict["FEE_SCHEDULE"] = str(fee_schedule)
else:
methodology_dict["FEE_SCHEDULE"] = ""
except ValueError as e:
logging.error(f"Error parsing LLM response: {e}")
methodology_breakout_answers = []
@@ -166,9 +205,16 @@ def prompt_fee_schedule_breakout(
Call FEE_SCHEDULE_BREAKOUT prompt with cached instruction.
Note: Field definitions are now included in FEE_SCHEDULE_BREAKOUT_INSTRUCTION() for caching.
"""
# Normalize FEE_SCHEDULE to string before using in prompt (should already be normalized from prompt_methodology_breakout)
fee_schedule = methodology_breakout_dict.get("FEE_SCHEDULE", "")
if isinstance(fee_schedule, list):
# If list, take first element (shouldn't happen, but defensive)
fee_schedule = fee_schedule[0] if fee_schedule else ""
fee_schedule = str(fee_schedule) if fee_schedule else ""
prompt, _parser = prompt_templates.FEE_SCHEDULE_BREAKOUT(
reimbursement_method,
methodology_breakout_dict.get("FEE_SCHEDULE"),
fee_schedule,
)
logging.debug(f"Fee Schedule Breakout Prompt for {filename}: {prompt}")
llm_answer_raw = llm_utils.invoke_claude(
@@ -299,7 +345,13 @@ def prompt_lob_relationship(
instruction=prompt_templates.LOB_RELATIONSHIP_INSTRUCTION(),
)
llm_answer_final = _parser(llm_answer_raw)
return llm_answer_final
# LOB_RELATIONSHIP returns a list (e.g., ["Inclusive"]), but we need a string
if isinstance(llm_answer_final, list) and len(llm_answer_final) > 0:
return str(llm_answer_final[0])
elif isinstance(llm_answer_final, str):
return llm_answer_final
else:
return "Exclusive" # Default fallback
def prompt_special_case_assignment(
@@ -126,6 +126,32 @@ def prompt_reimbursement_primary(
try:
llm_answer_final = _parser(llm_answer_raw)
# Normalize SERVICE_TERM and REIMB_TERM to strings immediately after parsing
# These should always be strings per row, not lists
if isinstance(llm_answer_final, list):
for answer_dict in llm_answer_final:
if isinstance(answer_dict, dict):
# SERVICE_TERM should be a string per row
if "SERVICE_TERM" in answer_dict:
service_term = answer_dict["SERVICE_TERM"]
if isinstance(service_term, list):
# If list, join with comma (shouldn't happen, but defensive)
answer_dict["SERVICE_TERM"] = ", ".join(str(item) for item in service_term if item)
elif service_term is not None:
answer_dict["SERVICE_TERM"] = str(service_term)
else:
answer_dict["SERVICE_TERM"] = ""
# REIMB_TERM should be a string per row
if "REIMB_TERM" in answer_dict:
reimb_term = answer_dict["REIMB_TERM"]
if isinstance(reimb_term, list):
# If list, join with comma (shouldn't happen, but defensive)
answer_dict["REIMB_TERM"] = ", ".join(str(item) for item in reimb_term if item)
elif reimb_term is not None:
answer_dict["REIMB_TERM"] = str(reimb_term)
else:
answer_dict["REIMB_TERM"] = ""
except ValueError as e:
logging.error(f"Error parsing LLM response: {e}")
logging.error(f"Raw LLM output: {llm_answer_raw}")
@@ -158,6 +184,19 @@ def prompt_methodology_breakout(
logging.debug(f"LLM Response for {filename}: {llm_response}")
try:
methodology_breakout_answers = _parser(llm_response)
# Normalize single-value fields that should be strings (not lists)
# FEE_SCHEDULE should be a string per methodology breakout result
if isinstance(methodology_breakout_answers, list):
for methodology_dict in methodology_breakout_answers:
if isinstance(methodology_dict, dict) and "FEE_SCHEDULE" in methodology_dict:
fee_schedule = methodology_dict["FEE_SCHEDULE"]
if isinstance(fee_schedule, list):
# If list, take first element (shouldn't happen, but defensive)
methodology_dict["FEE_SCHEDULE"] = fee_schedule[0] if fee_schedule else ""
elif fee_schedule is not None:
methodology_dict["FEE_SCHEDULE"] = str(fee_schedule)
else:
methodology_dict["FEE_SCHEDULE"] = ""
except ValueError as e:
logging.error(f"Error parsing LLM response: {e}")
methodology_breakout_answers = []
@@ -173,9 +212,16 @@ def prompt_fee_schedule_breakout(
Call FEE_SCHEDULE_BREAKOUT prompt with cached instruction.
Note: Field definitions are now included in FEE_SCHEDULE_BREAKOUT_INSTRUCTION() for caching.
"""
# Normalize FEE_SCHEDULE to string before using in prompt (should already be normalized from prompt_methodology_breakout)
fee_schedule = methodology_breakout_dict.get("FEE_SCHEDULE", "")
if isinstance(fee_schedule, list):
# If list, take first element (shouldn't happen, but defensive)
fee_schedule = fee_schedule[0] if fee_schedule else ""
fee_schedule = str(fee_schedule) if fee_schedule else ""
prompt, _parser = prompt_templates.FEE_SCHEDULE_BREAKOUT(
reimbursement_method,
methodology_breakout_dict.get("FEE_SCHEDULE"),
fee_schedule,
)
logging.debug(f"Fee Schedule Breakout Prompt for {filename}: {prompt}")
llm_answer_raw = llm_utils.invoke_claude(
@@ -306,7 +352,13 @@ def prompt_lob_relationship(
instruction=prompt_templates.LOB_RELATIONSHIP_INSTRUCTION(),
)
llm_answer_final = _parser(llm_answer_raw)
return llm_answer_final
# LOB_RELATIONSHIP returns a list (e.g., ["Inclusive"]), but we need a string
if isinstance(llm_answer_final, list) and len(llm_answer_final) > 0:
return str(llm_answer_final[0])
elif isinstance(llm_answer_final, str):
return llm_answer_final
else:
return "Exclusive" # Default fallback
def prompt_special_case_assignment(
+20 -3
View File
@@ -126,6 +126,11 @@ def prompt_reimbursement_primary(
try:
llm_answer_final = _parser(llm_answer_raw)
# DEBUG: Print raw parsed output
print(f"\n[DEBUG REIMBURSEMENT_PRIMARY] Parsed output for {filename}:")
print(llm_answer_final)
print()
except ValueError as e:
logging.error(f"Error parsing LLM response: {e}")
logging.error(f"Raw LLM output: {llm_answer_raw}")
@@ -173,9 +178,11 @@ def prompt_fee_schedule_breakout(
Call FEE_SCHEDULE_BREAKOUT prompt with cached instruction.
Note: Field definitions are now included in FEE_SCHEDULE_BREAKOUT_INSTRUCTION() for caching.
"""
# Normalize FEE_SCHEDULE to string before using in prompt (should already be normalized from prompt_methodology_breakout)
fee_schedule = methodology_breakout_dict.get("FEE_SCHEDULE", "")
prompt, _parser = prompt_templates.FEE_SCHEDULE_BREAKOUT(
reimbursement_method,
methodology_breakout_dict.get("FEE_SCHEDULE"),
fee_schedule,
)
logging.debug(f"Fee Schedule Breakout Prompt for {filename}: {prompt}")
llm_answer_raw = llm_utils.invoke_claude(
@@ -306,7 +313,13 @@ def prompt_lob_relationship(
instruction=prompt_templates.LOB_RELATIONSHIP_INSTRUCTION(),
)
llm_answer_final = _parser(llm_answer_raw)
return llm_answer_final
# LOB_RELATIONSHIP returns a list (e.g., ["Inclusive"]), but we need a string
if isinstance(llm_answer_final, list) and len(llm_answer_final) > 0:
return str(llm_answer_final[0])
elif isinstance(llm_answer_final, str):
return llm_answer_final
else:
return "Exclusive" # Default fallback
def prompt_special_case_assignment(
@@ -687,13 +700,17 @@ def prompt_lesser_of_distribution(
return reimb_term # Return original term unchanged
else:
llm_answer_final = _parser(llm_answer_raw)
# DEBUG: Print raw parsed output
print(f"\n[DEBUG prompt_lesser_of_distribution] Raw parsed output:")
print(llm_answer_final)
logging.debug(
f"Applied lesser-of to '{service_term}' on page {page_num}: "
f"{reimb_term}{llm_answer_final[:60]}..."
)
return llm_answer_final
return llm_answer_final[0] if isinstance(llm_answer_final, list) else llm_answer_final
def prompt_lesser_of_check(
@@ -234,8 +234,20 @@ def add_one_to_one_field(
# Only search for these fields when NO LOB has been found
if field_to_add.field_name in ["PROGRAM", "PRODUCT", "NETWORK"]:
lob_values = [answer_dict.get("LOB", "") for answer_dict in answer_dicts]
unique_lobs = set([lob for lob in lob_values if not string_utils.is_empty(lob)])
# LOB can be a string or a list (from JSON format)
# Extract all LOB values, handling both string and list formats
lob_values = []
for answer_dict in answer_dicts:
lob_value = answer_dict.get("LOB", "")
if isinstance(lob_value, list):
# If it's a list, extract each item and ensure it's a string
for item in lob_value:
if not string_utils.is_empty(item):
lob_values.append(str(item))
elif not string_utils.is_empty(lob_value):
# If it's a string, add it directly (already hashable)
lob_values.append(lob_value)
unique_lobs = set(lob_values)
if len(unique_lobs) > 0:
return one_to_one_fields
@@ -283,8 +295,22 @@ def get_dynamic_one_to_one_fields(
# Only search for PROGRAM, PRODUCT, NETWORK when NO LOB has been found
# NOTE: Check raw LOB field, not AARETE_DERIVED_LOB, because AARETE_DERIVED_LOB
# is populated later via crosswalk and may be derived from PROGRAM/PRODUCT
lob_values = [answer_dict.get("LOB", "") for answer_dict in answer_dicts]
unique_lobs = set([lob for lob in lob_values if not string_utils.is_empty(lob)])
# LOB can be a string or a list (from JSON format)
# Extract all LOB values, handling both string and list formats
# All values must be hashable (strings) to create a set
lob_values = []
for answer_dict in answer_dicts:
lob_value = answer_dict.get("LOB", "")
if isinstance(lob_value, list):
# If it's a list, extract each item and ensure it's a string
for item in lob_value:
if not string_utils.is_empty(item):
# Convert to string to ensure hashability (handles edge cases)
lob_values.append(str(item))
elif not string_utils.is_empty(lob_value):
# If it's a string, add it directly (already hashable)
lob_values.append(lob_value)
unique_lobs = set(lob_values)
has_lob = len(unique_lobs) > 0
# Handle ALL empty fields
@@ -98,6 +98,13 @@ def reimbursement_level(
reimbursement_primary_answers, constants, filename
)
# ============================================================================
# DEBUG: Print reimbursement_primary_answers after filtering
# ============================================================================
print(f"\n[DEBUG reimbursement_level] After filtering (line 101):")
print(reimbursement_primary_answers)
# ============================================================================
return reimbursement_primary_answers
@@ -108,10 +115,11 @@ def clean_reimbursement_primary(
) -> list[dict[str, str]]:
"""
Cleans and processes reimbursement primary answers through several steps:
1. Split compound reimbursements
2. Filter out any non-reimbursement content from split answers
3. Deduplicate against previously seen (SERVICE_TERM, REIMB_TERM) pairs
4. Apply exhibit lesser-of statement if it exists
1. Normalize SERVICE_TERM and REIMB_TERM to strings (they should never be lists per-row)
2. Split compound reimbursements
3. Filter out any non-reimbursement content from split answers
4. Deduplicate against previously seen (SERVICE_TERM, REIMB_TERM) pairs
5. Apply exhibit lesser-of statement if it exists
Args:
reimbursement_primary_answers (list[str]): List of dictionaries containing reimbursement details.
@@ -122,6 +130,7 @@ def clean_reimbursement_primary(
"""
if reimbursement_primary_answers: # If any reimbursements found
# If needed, add Reimbursement Splitting here
# If needed, add deduplication here
@@ -330,9 +339,17 @@ def methodology_breakout_single_row(
list[dict]: Updated list with methodology breakout details and fee schedule information.
"""
reimb_term, service_term = answer_dict.get("REIMB_TERM"), answer_dict.get(
"SERVICE_TERM"
)
# DEBUG: Check what answer_dict contains
print(f"\n[DEBUG methodology_breakout_single_row] answer_dict REIMB_TERM: {repr(answer_dict.get('REIMB_TERM'))} (type: {type(answer_dict.get('REIMB_TERM')).__name__})")
print(f"[DEBUG methodology_breakout_single_row] answer_dict SERVICE_TERM: {repr(answer_dict.get('SERVICE_TERM'))} (type: {type(answer_dict.get('SERVICE_TERM')).__name__})")
reimb_term = answer_dict.get("REIMB_TERM", "")
service_term = answer_dict.get("SERVICE_TERM", "")
# DEBUG: Check what we extracted
print(f"[DEBUG methodology_breakout_single_row] Extracted reimb_term: {repr(reimb_term)} (type: {type(reimb_term).__name__})")
print(f"[DEBUG methodology_breakout_single_row] Extracted service_term: {repr(service_term)} (type: {type(service_term).__name__})\n")
methodology_breakout_questions = FieldSet(
config.FIELD_JSON_PATH, field_type="methodology_breakout"
)
@@ -781,6 +798,31 @@ def one_to_n_cleaning(
################################ Split REIMB_DATES ################################
all_exhibit_rows = split_reimb_dates(all_exhibit_rows, filename)
################################ Normalize Single-Value Fields to Strings ################################
# REIMB_TERM, SERVICE_TERM, and LOB_PROGRAM_RELATIONSHIP should always be strings (not lists)
for answer_dict in all_exhibit_rows:
# LOB_PROGRAM_RELATIONSHIP should be a string (not a list)
if "LOB_PROGRAM_RELATIONSHIP" in answer_dict:
lob_program_rel = answer_dict["LOB_PROGRAM_RELATIONSHIP"]
if isinstance(lob_program_rel, list):
# If list, take first element (shouldn't happen, but defensive)
answer_dict["LOB_PROGRAM_RELATIONSHIP"] = str(lob_program_rel[0]) if lob_program_rel else "Exclusive"
elif lob_program_rel is not None:
answer_dict["LOB_PROGRAM_RELATIONSHIP"] = str(lob_program_rel)
else:
answer_dict["LOB_PROGRAM_RELATIONSHIP"] = "Exclusive"
# LOB_PRODUCT_RELATIONSHIP should be a string (not a list)
if "LOB_PRODUCT_RELATIONSHIP" in answer_dict:
lob_product_rel = answer_dict["LOB_PRODUCT_RELATIONSHIP"]
if isinstance(lob_product_rel, list):
# If list, take first element (shouldn't happen, but defensive)
answer_dict["LOB_PRODUCT_RELATIONSHIP"] = str(lob_product_rel[0]) if lob_product_rel else "Exclusive"
elif lob_product_rel is not None:
answer_dict["LOB_PRODUCT_RELATIONSHIP"] = str(lob_product_rel)
else:
answer_dict["LOB_PRODUCT_RELATIONSHIP"] = "Exclusive"
return all_exhibit_rows
@@ -32,18 +32,26 @@ def fill_na_mapping(answer_dicts):
all_lob_values = set()
# Get existing AARETE_DERIVED_LOB values (if any)
# Handle both list and string formats (JSON lists or pipe-delimited strings)
existing_lob_list = answer_dict.get("AARETE_DERIVED_LOB", "")
if not string_utils.is_empty(existing_lob_list):
for lob_val in existing_lob_list:
# Normalize to list format
if isinstance(existing_lob_list, list):
lob_items = existing_lob_list
elif isinstance(existing_lob_list, str):
# Handle pipe-delimited format (backward compatibility)
if "|" in existing_lob_list:
all_lob_values.update(
v.strip()
for v in lob_val.split("|")
if v.strip() and v.strip() != "N/A"
)
elif lob_val.strip() and lob_val.strip() != "N/A":
all_lob_values.add(lob_val)
lob_items = [v.strip() for v in existing_lob_list.split("|")]
else:
lob_items = [existing_lob_list]
else:
lob_items = [str(existing_lob_list)]
# Process each LOB value
for lob_val in lob_items:
if isinstance(lob_val, str) and lob_val.strip() and lob_val.strip() != "N/A":
all_lob_values.add(lob_val.strip())
# Get AARETE_DERIVED_LOB from AARETE_DERIVED_PROGRAM crosswalk (always check if PROGRAM exists)
program_lob_values = set()
@@ -58,16 +66,24 @@ def fill_na_mapping(answer_dicts):
not string_utils.is_empty(program_filled_value_list)
and "N/A" not in program_filled_value_list
):
for program_lob_val in program_filled_value_list:
if "|" in program_lob_val:
program_lob_values = set(
v.strip()
for v in program_lob_val.split("|")
if v.strip() and v.strip() != "N/A"
)
elif program_lob_val.strip() and program_lob_val.strip() != "N/A":
program_lob_values = {program_lob_val.strip()}
all_lob_values.update(program_lob_values)
# Normalize to list format (handle both list and string)
if isinstance(program_filled_value_list, list):
program_lob_items = program_filled_value_list
elif isinstance(program_filled_value_list, str):
# Handle pipe-delimited format (backward compatibility)
if "|" in program_filled_value_list:
program_lob_items = [v.strip() for v in program_filled_value_list.split("|")]
else:
program_lob_items = [program_filled_value_list]
else:
program_lob_items = [str(program_filled_value_list)]
# Process each program LOB value
for program_lob_val in program_lob_items:
if isinstance(program_lob_val, str) and program_lob_val.strip() and program_lob_val.strip() != "N/A":
program_lob_values.add(program_lob_val.strip())
# Update all_lob_values after processing all program LOB values
all_lob_values.update(program_lob_values)
# Get AARETE_DERIVED_LOB from PRODUCT crosswalk (always check if PRODUCT exists)
product_lob_values = set()
@@ -82,16 +98,24 @@ def fill_na_mapping(answer_dicts):
not string_utils.is_empty(product_filled_value_list)
and "N/A" not in product_filled_value_list
):
for product_lob_val in product_filled_value_list:
if "|" in product_lob_val:
product_lob_values = set(
v.strip()
for v in product_filled_value_list.split("|")
if v.strip() and v.strip() != "N/A"
)
elif product_lob_val.strip() and product_lob_val.strip() != "N/A":
product_lob_values = {product_lob_val.strip()}
all_lob_values.update(product_lob_values)
# Normalize to list format (handle both list and string)
if isinstance(product_filled_value_list, list):
product_lob_items = product_filled_value_list
elif isinstance(product_filled_value_list, str):
# Handle pipe-delimited format (backward compatibility)
if "|" in product_filled_value_list:
product_lob_items = [v.strip() for v in product_filled_value_list.split("|")]
else:
product_lob_items = [product_filled_value_list]
else:
product_lob_items = [str(product_filled_value_list)]
# Process each product LOB value
for product_lob_val in product_lob_items:
if isinstance(product_lob_val, str) and product_lob_val.strip() and product_lob_val.strip() != "N/A":
product_lob_values.add(product_lob_val.strip())
# Update all_lob_values after processing all product LOB values
all_lob_values.update(product_lob_values)
# Set the merged, deduplicated value
if all_lob_values:
@@ -165,11 +189,16 @@ def get_crosswalk_fields(answer_dicts: list, constants: Constants):
if "AARETE_DERIVED" in to_field_name:
to_field_answer_list.append(individual_from_field_value)
else:
to_field_answer_list.append(
crosswalk.create_reverse_mapping().get(
individual_from_field_value
)
)
# Reverse mapping returns dict[str, list[str]], so .get() returns a list
# We need to extend, not append, to avoid nested lists
reverse_mapping = crosswalk.create_reverse_mapping()
reverse_result = reverse_mapping.get(individual_from_field_value)
if reverse_result is not None:
# reverse_result is a list, extend it
to_field_answer_list.extend(reverse_result)
else:
# If not found in reverse mapping, keep original value
to_field_answer_list.append(individual_from_field_value)
answer_dict[to_field_name] = to_field_answer_list
# update from field value to list if not already
if not isinstance(from_field_value, list):
+41 -7
View File
@@ -26,7 +26,8 @@ JSON_LIST_OF_DICTS_FORMAT_INSTRUCTIONS = """Return your final answer as a valid
- Each item in the list should be a dictionary with the specified field names as keys.
- If multiple items are found, include all of them as separate dictionaries in the list.
- Use "N/A" for fields within each dictionary where the information doesn't apply.
- Ensure the JSON is properly formatted with square brackets, curly braces, and double quotes."""
- Ensure the JSON is properly formatted with square brackets, curly braces, and double quotes.
- The Keys and Values must be strings, and not lists."""
# Parser helper functions for prompt templates
@@ -486,7 +487,7 @@ def REIMBURSEMENT_PRIMARY(context) -> Tuple[str, Callable[[str], list]]:
Call REIMBURSEMENT_PRIMARY_INSTRUCTION() separately for the cached instruction.
Returns:
Tuple of (prompt_string, parser_function) where parser expects JSON list output.
Tuple of (prompt_string, parser_function) where parser expects JSON list output. Should return a list of dicts of strings
"""
prompt = f"""
[START CONTEXT]
@@ -500,7 +501,7 @@ Briefly talk through your reasoning. Then return a properly formatted JSON list
def REIMBURSEMENT_PRIMARY_INSTRUCTION() -> str:
return """[OBJECTIVE]
return f"""[OBJECTIVE]
Extract Reimbursement Terms from a section of a payer-provider contract
The output from this prompt will be used to populate downstream fields in the system.
@@ -536,6 +537,17 @@ Some types of Reimbursements are special and require additional thought.
- Stop Loss and Outliers - These reimbursements usually consist of multiple rates and/or thresholds. When a Stop-Loss or Outlier is found, include the entire statement (which may be multiple sentences) as a single REIMB_TERM.
- Discounts, Escalators and Premiums - These will also include multiple rates. Capture all relevant numbers
[OUTPUT FORMAT]
You MUST return your answer as a JSON list of dictionaries. Each dictionary must contain exactly two keys: "SERVICE_TERM" and "REIMB_TERM". Both values must be strings (not lists).
Example output format:
[
{{"SERVICE_TERM": "Service 1", "REIMB_TERM": "Reimb 1"}},
{{"SERVICE_TERM": "Service 2", "REIMB_TERM": "Reimb 2"}}
]
{JSON_LIST_OF_DICTS_FORMAT_INSTRUCTIONS}
[ANALYSIS CONTEXT]"""
@@ -852,7 +864,7 @@ Populate a list of JSON dictionaries with the following fields:
[OUTPUT FORMAT]
The output should always be a JSON list/array of dictionaries, even when there is only one methodology.
For any fields that don't apply to a particular methodology, use "N/A" as the value.
{JSON_LIST_OF_DICTS_FORMAT_INSTRUCTIONS}
[CONTEXT]"""
@@ -874,16 +886,20 @@ Populate a JSON dictionary with the following fields:
{fields_text}
[OUTPUT FORMAT]
Briefly explain your reasoning, then write your JSON dictionary."""
{JSON_DICT_FORMAT_INSTRUCTIONS}"""
def FEE_SCHEDULE_BREAKOUT(
methodology, fee_schedule
methodology:str, fee_schedule:str
) -> Tuple[str, Callable[[str], dict]]:
"""Returns ONLY dynamic content for fee schedule breakout.
Call FEE_SCHEDULE_BREAKOUT_INSTRUCTION() separately for the cached instruction.
Note: Field definitions are now included in FEE_SCHEDULE_BREAKOUT_INSTRUCTION() for caching.
Args:
methodology: The reimbursement methodology string (already normalized to string in prompt_calls)
fee_schedule: The fee schedule string (already normalized to string in prompt_calls)
Returns:
Tuple of (prompt_string, parser_function) where parser expects JSON dict output.
"""
@@ -899,7 +915,6 @@ def GROUPER_BREAKOUT_INSTRUCTION() -> str:
Contains objective, extraction rules, field definitions, and output format.
Field definitions are loaded from investment_prompts.json.
"""
# Load fields from investment_prompts.json with resolved valid_values
fields_text = _get_fields_text("grouper_breakout")
return f"""[OBJECTIVE]
@@ -1151,6 +1166,10 @@ def CODE_EXPLICIT(service, methodology) -> Tuple[str, Callable[[str], dict]]:
Call CODE_EXPLICIT_INSTRUCTION() separately for the cached instruction.
Note: Field definitions are now included in CODE_EXPLICIT_INSTRUCTION() for caching.
Args:
service: The service term (already normalized to string upstream)
methodology: The reimbursement methodology (already normalized to string upstream)
Returns:
Tuple of (prompt_string, parser_function) where parser expects JSON dict output.
"""
@@ -1182,6 +1201,10 @@ Explain your answer in 1-2 sentences. Write your final answer in a properly-form
def CODE_CATEGORY(service, choices) -> Tuple[str, Callable[[str], list]]:
"""Call CODE_CATEGORY_INSTRUCTION() separately for the cached instruction.
Args:
service: The service term (already normalized to string upstream)
choices: The valid descriptions to choose from
Returns:
Tuple of (prompt_string, parser_function) where parser expects JSON list output.
"""
@@ -1255,6 +1278,10 @@ Explain your answer, ensuring each point in the instructions is addressed. Then
def CODE_IMPLICIT(service, choices) -> Tuple[str, Callable[[str], list]]:
"""Call CODE_IMPLICIT_INSTRUCTION() separately for the cached instruction.
Args:
service: The service term (already normalized to string upstream)
choices: The valid descriptions to choose from
Returns:
Tuple of (prompt_string, parser_function) where parser expects JSON list output.
"""
@@ -1325,6 +1352,12 @@ def FILL_BILL_TYPE(
) -> Tuple[str, Callable[[str], list]]:
"""Call FILL_BILL_TYPE_INSTRUCTION() separately for the cached instruction.
Args:
service: The service term (already normalized to string upstream)
choices: The valid bill type descriptions to choose from
reimb_term: The reimbursement term (already normalized to string upstream, optional)
exhibit_text: Additional exhibit context (optional)
Returns:
Tuple of (prompt_string, parser_function) where parser expects JSON list output.
"""
@@ -1346,6 +1379,7 @@ If the Service Term and Reimbursement Term together do not provide sufficient in
{exhibit_text.replace('"', "'")}
"""
# service is already normalized to string upstream
prompt = f"""[VALID DESCRIPTIONS]
Here are the descriptions to choose from:
{choices}
+3 -3
View File
@@ -4,16 +4,16 @@ import logging
import src.utils.string_utils as string_utils
def apply_crosswalk(val: str, mapping: dict[str, str], default: str = "N/A") -> str:
def apply_crosswalk(val: str | list, mapping: dict[str, str], default: str = "N/A") -> list:
"""Apply a crosswalk to a value
Args:
val (str): Input value
val (str | list): Input value (can be string, list, or string representation of list)
mapping (dict[str, str]): Mapping dictionary
default (str | None, optional): Default value; if None and val is not found in the mapping, val is returned. Defaults to "".
Returns:
str: Crosswalked value
list: Crosswalked value as a list
"""
# Handle None or blank values defensively
if val is None:
+1 -1
View File
@@ -253,4 +253,4 @@ def parse_json_dict_or_list(raw_llm_output: str) -> dict[str, Any] | list[Any]:
raise ValueError(
f"No valid JSON object found in LLM output. "
f"Output preview: {raw_llm_output[:200]}..."
)
)
+29 -10
View File
@@ -1092,22 +1092,41 @@ def safe_token_to_str(tok: Any) -> str:
def format_or_preserve_tins(val: Any) -> Tuple[str, bool]:
"""
Format TINs to 9 digits, preserving pipe-delimited multiple values.
Format TINs to 9 digits, handling JSON format (direct values or list of values).
Args:
val: TIN value (may be pipe-delimited)
val: TIN value (may be a string, list of strings, or None/empty)
Returns:
Tuple of (formatted_tins, has_invalid)
"""
if pd.isna(val) or val is None or str(val).strip() == "":
return ("", True)
parts = [
part.strip()
for part in str(val).split("|")
if part is not None and part.strip() != ""
]
# Handle list/array values (e.g., from PROV_GROUP_TIN, PROV_OTHER_TIN)
if isinstance(val, (list, tuple)) or (hasattr(val, '__iter__') and not isinstance(val, str)):
# Convert list to list of strings, filtering out empty values
if len(val) == 0:
return ("", True)
parts = [
str(item).strip()
for item in val
if item is not None and str(item).strip() != ""
]
if not parts:
return ("", True)
else:
# Handle single value (string or None)
# Check for NaN/None/empty
try:
if pd.isna(val):
return ("", True)
except (ValueError, TypeError):
# pd.isna() can't handle arrays, but we've already handled lists above
pass
if val is None or str(val).strip() == "":
return ("", True)
# Single value - convert to list for consistent processing
parts = [str(val).strip()]
out, any_invalid = [], False
for part in parts: