Merged in feature/one-to-n-docs (pull request #715)

Feature/one to n docs

* Refactor function signatures and improve imports in one_to_n_funcs.py

* Enhance type hints and documentation for carveout_check and methodology_breakout functions

* Add docstring to methodology_breakout_secondary function for clarity on parameters and return values

* Add docstring to special_case_breakout function for clarity on parameters and return values

* Enhance docstring for deduplicate_with_accumulator function to clarify behavior and parameters

* Enhance docstring for filter_services_without_reimbursements function to clarify validation process and answer structure

* Enhance docstring for filter_services_without_reimbursements function to clarify filtering criteria and validation process

* Enhance docstring for split_compound_reimbursements function to clarify the splitting process and provide an example

* Enhance docstring for split_reimb_dates function to clarify date processing, parameters, return values, and provide an example

* Remove unused regex import from one_to_n_funcs.py

* Remove unused Delimiter import from one_to_n_funcs.py


Approved-by: Katon Minhas
This commit is contained in:
Alex Galarce
2025-09-30 14:04:47 +00:00
committed by Katon Minhas
parent b788ffc352
commit ae4eda6f25
+158 -22
View File
@@ -1,19 +1,17 @@
import logging import logging
import re
import src.investment.dynamic_funcs as dynamic_funcs
import src.investment.lesser_of_funcs as lesser_of_funcs
import src.investment.prompt_calls as prompt_calls
import src.prompts.prompt_templates as prompt_templates import src.prompts.prompt_templates as prompt_templates
import src.utils.llm_utils as llm_utils import src.utils.llm_utils as llm_utils
import src.utils.string_utils as string_utils import src.utils.string_utils as string_utils
from constants.constants import Constants from constants.constants import Constants
from constants.delimiters import Delimiter
from src import config from src import config
from src.prompts.fieldset import FieldSet from src.prompts.fieldset import FieldSet
import src.investment.dynamic_funcs as dynamic_funcs
import src.investment.prompt_calls as prompt_calls
import src.investment.lesser_of_funcs as lesser_of_funcs
def get_exhibit_header(all_exhibit_headers, exhibit_page): def get_exhibit_header(all_exhibit_headers: dict, exhibit_page: str) -> str:
""" """
Retrieves the exhibit header for a given exhibit page. Retrieves the exhibit header for a given exhibit page.
Args: Args:
@@ -45,7 +43,7 @@ def exhibit_level(
reimbursement_level_fields: FieldSet, reimbursement_level_fields: FieldSet,
constants: Constants, constants: Constants,
filename: str, filename: str,
): ) -> tuple[dict, FieldSet]:
""" """
Processes exhibit-level fields, including dynamic fields. Processes exhibit-level fields, including dynamic fields.
@@ -162,13 +160,13 @@ def reimbursement_level(
def clean_reimbursement_primary( def clean_reimbursement_primary(
reimbursement_primary_answers, reimbursement_primary_answers: list[dict],
exhibit_lesser_of, exhibit_lesser_of: str,
seen_pairs, seen_pairs: set,
exhibit_page, exhibit_page: str,
constants: Constants, constants: Constants,
filename: str, filename: str,
): ) -> list[dict]:
""" """
Cleans and processes reimbursement primary answers through several steps: Cleans and processes reimbursement primary answers through several steps:
1. Split compound reimbursements 1. Split compound reimbursements
@@ -182,6 +180,7 @@ def clean_reimbursement_primary(
seen_pairs (set): A set accumulator to track tuples of (base_page, SERVICE_TERM, REIMB_TERM) to avoid duplicate processing. seen_pairs (set): A set accumulator to track tuples of (base_page, SERVICE_TERM, REIMB_TERM) to avoid duplicate processing.
constants (Constants): A Constants object constants (Constants): A Constants object
exhibit_page (str): The page number of the exhibit being processed (for a unique identifier) exhibit_page (str): The page number of the exhibit being processed (for a unique identifier)
filename (str): The name of the file being processed, for logging
Returns: Returns:
list[dict]: The cleaned and processed reimbursement primary answers. list[dict]: The cleaned and processed reimbursement primary answers.
""" """
@@ -221,7 +220,7 @@ def breakout(
special_case_primary_fields: FieldSet, special_case_primary_fields: FieldSet,
filename: str, filename: str,
constants: Constants, constants: Constants,
): ) -> tuple[list[dict], dict[str, list]]:
""" """
Performs breakout processing on reimbursement primary answers and special case primary answers. Performs breakout processing on reimbursement primary answers and special case primary answers.
@@ -254,7 +253,7 @@ def breakout(
def carveout_check( def carveout_check(
reimbursement_primary_answers: list[dict], constants: Constants, filename: str reimbursement_primary_answers: list[dict], constants: Constants, filename: str
): ) -> list[dict]:
""" """
Performs carveout check on reimbursement primary answers. Performs carveout check on reimbursement primary answers.
@@ -281,12 +280,31 @@ def carveout_check(
answer_dict["CARVEOUT_IND"] = "Y" answer_dict["CARVEOUT_IND"] = "Y"
# Determine DEFAULT_IND based on specific carveout_answer values # Determine DEFAULT_IND based on specific carveout_answer values
answer_dict["DEFAULT_IND"] = "Y" if carveout_answer in ["DEFAULT_TERM", "UNLISTED_CODE"] else "N" answer_dict["DEFAULT_IND"] = (
"Y" if carveout_answer in ["DEFAULT_TERM", "UNLISTED_CODE"] else "N"
)
return reimbursement_primary_answers return reimbursement_primary_answers
def methodology_breakout(reimbursement_primary_answers, constants, filename): def methodology_breakout(
reimbursement_primary_answers: list[dict], constants: Constants, filename: str
) -> list[dict]:
"""Perform methodology breakout on reimbursement primary answers.
This function loops through each reimbursement primary answer and applies the
methodology breakout process, which may include additional prompts for fee schedule
or grouper details based on the reimbursement method.
Args:
reimbursement_primary_answers (list[dict]): List of dictionaries containing reimbursement details from the reimbursement primary step.
constants (Constants): A Constants object.
filename (str): The name of the file being processed, for logging.
Returns:
list[dict]: The processed reimbursement primary answers, augmented with methodology breakout details.
"""
final_answers = [] final_answers = []
for answer_dict in reimbursement_primary_answers: for answer_dict in reimbursement_primary_answers:
final_answers += methodology_breakout_single_row( final_answers += methodology_breakout_single_row(
@@ -369,7 +387,38 @@ def methodology_breakout_secondary(
grouper_breakout_questions: FieldSet, grouper_breakout_questions: FieldSet,
constants: Constants, constants: Constants,
filename: str, filename: str,
): ) -> dict:
"""Performs specialized breakout based on the derived reimbursement methodology type.
After the primary methodology breakout determines the AARETE_DERIVED_REIMB_METHOD,
this function applies methodology-specific prompts to extract detailed parameters:
- **Fee Schedule**: Extracts fee schedule type, version, and related details
- **Grouper**: Extracts DRG weights, severity adjustments, transfer policies, etc.
- **Other methodologies**: Sets all specialized fields to "N/A"
Additionally, it ensures mutually exclusive field population across Fee Schedule and Grouper fields.
(Fee Schedule OR Grouper, never both.)
Args:
methodology_breakout_dict (dict): Results from primary methodology breakout,
including the AARETE_DERIVED_REIMB_METHOD field which will drive the
branching logic.
service (str): The service term associated with the reimbursement.
reimbursement_method (str): The original reimbursement method text extracted
(pre-AARETE_DERIVED)
fs_breakout_questions (FieldSet): Fee Schedule-specific prompts and field definitions.
grouper_breakout_questions (FieldSet): Grouper-specific prompts and field definitions.
constants (Constants): System constants for processing.
filename (str): The filename used for LLM processing.
Returns:
dict: Specialized breakout fields populated based on methodology type.
Fee Schedule methodology → fee schedule fields + grouper fields set to "N/A"
Grouper methodology → grouper fields + fee schedule fields set to "N/A"
Other methodologies → all specialized fields set to "N/A"
"""
AARETE_DERIVED_REIMB_METHOD = methodology_breakout_dict.get( AARETE_DERIVED_REIMB_METHOD = methodology_breakout_dict.get(
"AARETE_DERIVED_REIMB_METHOD", "" "AARETE_DERIVED_REIMB_METHOD", ""
) )
@@ -417,6 +466,27 @@ def special_case_breakout(
special_case_primary_fields: FieldSet, special_case_primary_fields: FieldSet,
filename: str, filename: str,
) -> dict[str, list]: ) -> dict[str, list]:
"""Applies specialized breakout templates to special case primary answers.
For each special case field (outliers, discounts, premiums, etc.) that has extracted
terms, this function applies the field-specific breakout template to each term
to extract detailed and custom parameters (thresholds, percentages, conditions,
etc.)
Args:
special_case_primary_answers (dict): A dictionary containing the primary answers
for special case fields. This will take the form of dict[str, list[str]],
where the key is the field name and the value is a list of extracted terms
for that field.
TODO: Add examples here and to `Returns:` section below.
special_case_primary_fields (FieldSet): FieldSet containing special case primary
field definitions and breakout templates.
filename (str): The name of the file being processed, for logging.
Returns:
dict[str, list]: A dictionary containing the processed breakout answers for each
special case field.
"""
final_answer_dict = {} final_answer_dict = {}
@@ -444,11 +514,21 @@ def special_case_breakout(
def deduplicate_with_accumulator( def deduplicate_with_accumulator(
answers: list[dict], seen_pairs: set, exhibit_page: str, filename: str answers: list[dict], seen_pairs: set, exhibit_page: str, filename: str
) -> list[dict]: ) -> list[dict]:
"""Remove service-reimbursement pairs that have been seen in previous exhibits. """Remove service-reimbursement pairs that have been seen in previous pages.
Includes a base page scope to allow only subexhibits to share pairs. This is because
(a) subexhibits often repeat information due to our table splitting logic, and
(b) different exhibits may share the same (service, reimbursement) pairs,
but in meaningfully different contexts (e.g. MEDICARE FEE SCHEDULE on
exhibit page 23 vs. MARKETPLACE FEE SCHEDULE on exhibit 45). In this case
we do NOT want to deduplicate across exhibits.
Args: Args:
answers (list[dict]): List of dictionaries with SERVICE_TERM and REIMB_TERM keys. answers (list[dict]): List of dictionaries with SERVICE_TERM and REIMB_TERM keys.
seen_pairs (set): A set of previously seen (base_page, SERVICE_TERM, REIMB_TERM) pairs. seen_pairs (set): A set of previously seen (base_page, SERVICE_TERM, REIMB_TERM) pairs.
e.g. {("23", "Routine Vision", "Provider's Billed Charges"), ...}
This is the "accumulator" that persists across multiple calls. It is
modified in-place to add new pairs as they are encountered, and as such
is not returned by the function.
exhibit_page (str): The exhibit page number, for logging. exhibit_page (str): The exhibit page number, for logging.
filename (str): The name of the file being processed, for logging filename (str): The name of the file being processed, for logging
@@ -491,10 +571,13 @@ def filter_services_without_reimbursements(
""" """
Filters out services that lack actual reimbursement methodologies. Filters out services that lack actual reimbursement methodologies.
Uses a single-step approach of LLM validation for remaining entries to determine if they contain actual reimbursement information. Uses LLM validation to determine if entries contain actual reimbursement information.
Entries that contain service definitions, administrative content, or vague terms
rather than clear reimbursement methodologies are filtered out.
Args: Args:
answers (list[dict]): List of dictionaries containing reimbursement details. answers (list[dict]): List of dictionaries containing reimbursement details.
These will be of the form dict[str, str], with keys including SERVICE_TERM and REIMB_TERM.
filename (str): The name of the file being processed. filename (str): The name of the file being processed.
Returns: Returns:
@@ -533,6 +616,32 @@ def split_compound_reimbursements(
""" """
Splits reimbursement terms that contain multiple distinct methodologies. Splits reimbursement terms that contain multiple distinct methodologies.
The process involves two main steps:
1. A substring search is performed to identify potential compound reimbursements
using a predefined list of indicators (e.g. "otherwise shall be", "except for",
"for services where there is no", etc.).
2. For entries flagged as potential compounds, an LLM prompt is invoked to
attempt to split the entry into multiple distinct reimbursement terms.
If the LLM successfully splits the entry into multiple parts, these are added
to the results. If not, the original entry is retained.
Example:
INPUT:
SERVICE: "Drugs"
REIMBURSEMENT: "lesser of Provider's billed charges or: (i) Fee schedule rate for standard drugs, (ii) WAC + 5% for specialty drugs, not to exceed $50"
OUTPUT:
[
{{
"SERVICE_TERM": "standard drugs",
"REIMB_TERM": "lesser of Provider's billed charges or Fee schedule rate"
}},
{{
"SERVICE_TERM": "specialty drugs",
"REIMB_TERM": "lesser of Provider's billed charges or WAC + 5%, not to exceed $50"
}}
]
Args: Args:
answers (list[dict]): List of dictionaries containing reimbursement details. answers (list[dict]): List of dictionaries containing reimbursement details.
filename (str): The name of the file being processed. filename (str): The name of the file being processed.
@@ -546,7 +655,7 @@ def split_compound_reimbursements(
for answer_dict in answers: for answer_dict in answers:
reimb_term = answer_dict.get("REIMB_TERM", "").lower().strip() reimb_term = answer_dict.get("REIMB_TERM", "").lower().strip()
# Check if this looks like a compound reimbursement # Check if this looks like a compound reimbursement by using a substring search
has_compound = any(indicator in reimb_term for indicator in COMPOUND_INDICATORS) has_compound = any(indicator in reimb_term for indicator in COMPOUND_INDICATORS)
if has_compound: if has_compound:
@@ -574,8 +683,35 @@ def split_compound_reimbursements(
def split_reimb_dates(one_to_n_results: list, filename: str) -> list: def split_reimb_dates(one_to_n_results: list, filename: str) -> list:
"""Extract the date range from the field REIMB_DATES and split the dates into """Splits REIMB_DATES field into separate effective and termination date fields.
REIMB_EFFECTIVE_DT and REIMB_TERMINATION_DT by invoking LLM and split the dates based on the context.
Processes records containing REIMB_DATES and uses LLM to parse date ranges into
structured REIMB_EFFECTIVE_DT and REIMB_TERMINATION_DT fields. The original
REIMB_DATES field is removed after processing.
Args:
one_to_n_results (list): List of dictionaries containing reimbursement records.
Records with valid REIMB_DATES fields will be processed.
filename (str): The name of the file being processed, for logging.
Returns:
list: Updated list of dictionaries with:
- REIMB_EFFECTIVE_DT: Parsed start date from date range
- REIMB_TERMINATION_DT: Parsed end date from date range
- REIMB_DATES: Removed from all records
Example:
Input record: {"SERVICE_TERM": "Labs", "REIMB_DATES": "January 1, 2024 through December 31, 2024"}
Output record: {
"SERVICE_TERM": "Labs",
"REIMB_EFFECTIVE_DT": "2024-01-01",
"REIMB_TERMINATION_DT": "2024-12-31"
}
Note:
- Records without valid REIMB_DATES are left unchanged (but still get the new fields initialized as None)
- Processing continues even if individual records fail to parse
- All records have REIMB_DATES field removed regardless of processing success
""" """
if not one_to_n_results or not isinstance(one_to_n_results, list): if not one_to_n_results or not isinstance(one_to_n_results, list):