From e28fbd516d4e37700fcf8d87ca9ee1a812bcca25 Mon Sep 17 00:00:00 2001 From: Karan Desai Date: Fri, 3 Apr 2026 20:35:51 +0000 Subject: [PATCH] Merged in bugfix/exhibit-smart-chunking (pull request #933) Bugfix/exhibit smart chunking * keywords detection update * fallback for null header_dict * cleaned redundant functions * black formatting * pytest fixes * Merged dev into bugfix/exhibit-smart-chunking Approved-by: Katon Minhas --- .../bcbs_promise/prompts/prompt_calls.py | 22 ++-- .../clients/clover/prompts/prompt_calls.py | 22 ++-- src/pipelines/saas/prompts/prompt_calls.py | 6 +- .../shared/preprocessing/preprocess.py | 19 +++ .../preprocessing/preprocessing_funcs.py | 23 ++-- src/prompts/prompt_templates.py | 115 +++++------------- 6 files changed, 86 insertions(+), 121 deletions(-) diff --git a/src/pipelines/clients/bcbs_promise/prompts/prompt_calls.py b/src/pipelines/clients/bcbs_promise/prompts/prompt_calls.py index 542c6ee..ebd335f 100644 --- a/src/pipelines/clients/bcbs_promise/prompts/prompt_calls.py +++ b/src/pipelines/clients/bcbs_promise/prompts/prompt_calls.py @@ -4,7 +4,7 @@ import src.config as config import src.prompts.prompt_templates as prompt_templates from src.constants.constants import Constants from src.prompts.fieldset import Field, FieldSet -from src.utils import llm_utils, string_utils +from src.utils import llm_utils, string_utils, json_utils import json @@ -556,29 +556,31 @@ def prompt_exhibit_linkage(page_header, previous_header, filename): def prompt_exhibit_header(page_content, filename): """Extract exhibit header identifier from page content using pattern matching. - Uses LLM to identify exhibit headers in page content based on predefined markers - and extract the exhibit identifier. - Note: Header markers are now included in EXHIBIT_HEADER_INSTRUCTION() for caching. + Uses LLM to identify exhibit headers in page content and extract the exhibit identifier. Args: - page_content (str): First 400 characters of page content to analyze. + page_content (str): Page content to analyze. filename (str): Source filename for logging and LLM attribution. Returns: - str: Extracted exhibit header identifier or marker indicating no header found. + dict: Dictionary mapping boundary numbers to header strings, or empty dict on failure. """ - prompt, _parser = prompt_templates.EXHIBIT_HEADER(page_content[0:400]) + prompt, _ = prompt_templates.EXHIBIT_HEADER(page_content[0:400]) llm_answer_raw = llm_utils.invoke_claude( prompt, "sonnet_latest", filename, - max_tokens=300, + max_tokens=1024, cache=True, instruction=prompt_templates.EXHIBIT_HEADER_INSTRUCTION(), usage_label="EXHIBIT_HEADER", ) - llm_answer_extracted = _parser(llm_answer_raw) - return llm_answer_extracted + try: + return json_utils.parse_json_dict(llm_answer_raw) + except ValueError as e: + logging.error(f"Error parsing exhibit header response: {e}") + logging.error(f"Raw LLM output: {llm_answer_raw}") + return {} def prompt_date_fix(date: str) -> str: diff --git a/src/pipelines/clients/clover/prompts/prompt_calls.py b/src/pipelines/clients/clover/prompts/prompt_calls.py index d9aa663..a98c987 100644 --- a/src/pipelines/clients/clover/prompts/prompt_calls.py +++ b/src/pipelines/clients/clover/prompts/prompt_calls.py @@ -4,7 +4,7 @@ import src.config as config import src.prompts.prompt_templates as prompt_templates from src.constants.constants import Constants from src.prompts.fieldset import Field, FieldSet -from src.utils import llm_utils, string_utils +from src.utils import llm_utils, string_utils, json_utils import json @@ -579,29 +579,31 @@ def prompt_exhibit_linkage(page_header, previous_header, filename): def prompt_exhibit_header(page_content, filename): """Extract exhibit header identifier from page content using pattern matching. - Uses LLM to identify exhibit headers in page content based on predefined markers - and extract the exhibit identifier. - Note: Header markers are now included in EXHIBIT_HEADER_INSTRUCTION() for caching. + Uses LLM to identify exhibit headers in page content and extract the exhibit identifier. Args: - page_content (str): First 400 characters of page content to analyze. + page_content (str): Page content to analyze. filename (str): Source filename for logging and LLM attribution. Returns: - str: Extracted exhibit header identifier or marker indicating no header found. + dict: Dictionary mapping boundary numbers to header strings, or empty dict on failure. """ - prompt, _parser = prompt_templates.EXHIBIT_HEADER(page_content[0:400]) + prompt, _ = prompt_templates.EXHIBIT_HEADER(page_content[0:400]) llm_answer_raw = llm_utils.invoke_claude( prompt, "sonnet_latest", filename, - max_tokens=300, + max_tokens=1024, cache=True, instruction=prompt_templates.EXHIBIT_HEADER_INSTRUCTION(), usage_label="EXHIBIT_HEADER", ) - llm_answer_extracted = _parser(llm_answer_raw) - return llm_answer_extracted + try: + return json_utils.parse_json_dict(llm_answer_raw) + except ValueError as e: + logging.error(f"Error parsing exhibit header response: {e}") + logging.error(f"Raw LLM output: {llm_answer_raw}") + return {} def prompt_date_fix(date: str) -> str: diff --git a/src/pipelines/saas/prompts/prompt_calls.py b/src/pipelines/saas/prompts/prompt_calls.py index 582f4b2..72c561c 100644 --- a/src/pipelines/saas/prompts/prompt_calls.py +++ b/src/pipelines/saas/prompts/prompt_calls.py @@ -579,15 +579,13 @@ def prompt_exhibit_header(page_content, EXHIBIT_HEADER_MARKERS, filename): Returns empty dict if parsing fails. """ section_boundaries = preprocessing_funcs.identify_section_boundaries(page_content) - prompt = prompt_templates.EXHIBIT_HEADER_NEW( - section_boundaries, EXHIBIT_HEADER_MARKERS - ) + prompt, _ = prompt_templates.EXHIBIT_HEADER(section_boundaries) llm_answer_raw = llm_utils.invoke_claude( prompt, "sonnet_latest", filename, - max_tokens=300, + max_tokens=1024, cache=True, instruction=prompt_templates.EXHIBIT_HEADER_INSTRUCTION(), usage_label="EXHIBIT_HEADER", diff --git a/src/pipelines/shared/preprocessing/preprocess.py b/src/pipelines/shared/preprocessing/preprocess.py index 35b3cc7..e113c4c 100644 --- a/src/pipelines/shared/preprocessing/preprocess.py +++ b/src/pipelines/shared/preprocessing/preprocess.py @@ -133,6 +133,25 @@ def one_to_n_exhibit_chunking( sorted(exhibit_header_dict.items(), key=lambda x: int(x[0].split(".")[0])) ) + # Fallback: if no exhibits were detected, use the first 10 non-empty lines of the + # first page as a synthetic header so downstream extraction has something to work with + if not exhibit_header_dict: + source = pages_dict if pages_dict is not None else text_dict + sorted_pages = sorted(source.keys(), key=lambda x: int(x.split(".")[0])) + first_page_num = sorted_pages[0] + if pages_dict is not None: + first_page_text = pages_dict[first_page_num].get_text() + else: + first_page_text = text_dict[first_page_num] + first_n_lines = [line for line in first_page_text.split("\n") if line.strip()][ + :5 + ] + if first_n_lines: + exhibit_header_dict = {first_page_num: ["\n".join(first_n_lines)]} + logging.info( + f"exhibit_header_dict was empty after chunking; falling back to first 5 lines of page {first_page_num} - {filename}" + ) + return exhibit_header_dict diff --git a/src/pipelines/shared/preprocessing/preprocessing_funcs.py b/src/pipelines/shared/preprocessing/preprocessing_funcs.py index 81bcbd9..a2df178 100644 --- a/src/pipelines/shared/preprocessing/preprocessing_funcs.py +++ b/src/pipelines/shared/preprocessing/preprocessing_funcs.py @@ -600,21 +600,22 @@ def identify_section_boundaries(page_text: str, context_chars: int = 100) -> str # Healthcare contract section patterns (order matters - more specific first) section_patterns = [ - (r"^\s*EXHIBIT\s+[A-Z0-9]+", "EXHIBIT"), + (r"(?i)^\s*EXHIBIT\s+[A-Z0-9]+", "EXHIBIT"), ( - r"^\s*ARTICLE\s+[A-Z]+", + r"(?i)^\s*ARTICLE\s+[A-Z]+", "ARTICLE", ), # Catches "ARTICLE SIX", "ARTICLE ONE", etc. - (r"^\s*ARTICLE\s+[IVX0-9]+", "ARTICLE"), # Roman numerals and numbers - # (r'^\s*SECTION\s+\d+(\.\d+)*', 'SECTION'), - (r"^\s*SCHEDULE\s+[A-Z0-9]+", "SCHEDULE"), - (r"^\s*APPENDIX\s+[A-Z0-9]+", "APPENDIX"), - (r"^\s*ATTACHMENT\s+[A-Z0-9]+", "ATTACHMENT"), - # (r'^\s*\d+\.\d+\s+[A-Z]', 'SUBSECTION'), # "6.1 Indemnification" - (r"^\s*[A-Z][A-Z\s]{20,}$", "HEADER"), # All-caps headers (20+ chars) - # (r'^\s*\d+\.\s+[A-Z][A-Z\s]+', 'NUMBERED_SECTION'), # "1. DEFINITIONS" + (r"(?i)^\s*ARTICLE\s+[IVX0-9]+", "ARTICLE"), # Roman numerals and numbers + (r"(?i)^\s*SCHEDULE\s+[A-Z0-9]+", "SCHEDULE"), + (r"(?i)^\s*APPENDIX\s+[A-Z0-9]+", "APPENDIX"), + (r"(?i)^\s*ATTACHMENT\s+[A-Z0-9]+", "ATTACHMENT"), + (r"(?i)^\s*[A-Z][A-Z\s]{20,}$", "HEADER"), # All-caps headers (20+ chars) + # (r"(?i)^\s*(?=(?:[^a-z\n]*[A-Z]){10})[^a-z\n]+$", "HEADER"), # 10+ uppercase letters, no lowercase + ( + r"^\s*(?=(?:[^a-z\n]*[A-Z]){10})[^a-z\n]+$", + "HEADER", + ), # 10+ uppercase letters, no lowercase ] - lines = page_text.split("\n") potential_boundaries = [] diff --git a/src/prompts/prompt_templates.py b/src/prompts/prompt_templates.py index 98d273a..52c0a51 100644 --- a/src/prompts/prompt_templates.py +++ b/src/prompts/prompt_templates.py @@ -218,7 +218,6 @@ def get_cacheable_instructions(): # Preprocessing instructions (exhibit headers, linkage) try: cacheable_instructions["EXHIBIT_HEADER"] = EXHIBIT_HEADER_INSTRUCTION() - cacheable_instructions["EXHIBIT_HEADER_NEW"] = EXHIBIT_HEADER_NEW_INSTRUCTION() cacheable_instructions["EXHIBIT_LINKAGE"] = EXHIBIT_LINKAGE_INSTRUCTION() cacheable_instructions["EXHIBIT_TITLE_MATCH"] = ( EXHIBIT_TITLE_MATCH_INSTRUCTION() @@ -1916,93 +1915,23 @@ Briefly explain your reasoning, then put your final answer in JSON dictionary fo def EXHIBIT_HEADER_INSTRUCTION() -> str: """Static instruction for EXHIBIT_HEADER prompt caching. - Contains all inclusion/exclusion rules, header markers, and output format. - Header markers are loaded from Constants.EXHIBIT_HEADER_MARKERS. - """ - # Load exhibit header markers from Constants - constants = _get_constants() - EXHIBIT_HEADER_MARKERS = constants.EXHIBIT_HEADER_MARKERS - - markers_text = "\n* ".join(EXHIBIT_HEADER_MARKERS) - - return f"""[OBJECTIVE] -Analyze contract excerpts and extract the full exhibit/attachment header found. -Capture the complete hierarchy of document identifiers present in the text. - -[HEADER MARKERS] -The following keywords and phrases should be considered as exhibit/attachment headers: -* {markers_text} - -[RULES FOR INCLUSION] -* Include ALL text that appears to be part of the header -* Headers MUST appear at the top of the page - do not mistakenly extract Footers, which will be found at the bottom of the page with other text before them. -* Full header names and numbers/letters (e.g., "Attachment C: Commercial-Exchange") -* Any other relevant subtitles and identifiers -* Any words in the header that may specify what type of information is contained, such as "Compensation Schedule" or "Definitions". -* Provider/Entity names when listed with exhibit information -* Keep exact capitalization and formatting as shown in document -* Include any information referring to the current exhibit's relationship with another exhibit (e.g. "Exhibit 1: Provider Roster (see Exhibit 2 for Compensation Schedule)") - -[RULES FOR EXCLUSION] -* Do NOT abbreviate or summarize any part of the exhibit names -* Do NOT include page numbers -* Do NOT cherry-pick only certain parts - capture the complete hierarchy -* Do NOT include docusign IDs and other metadata -* DO NOT include unrelated text written in lowercase or title case. - -[OUTPUT FORMAT] -If there is no Exhibit Header present, return ["N/A"]. -Return your final answer as a single item in a JSON list, followed by a brief explanation. -Before returning any output, ensure that all instructions for inclusion were followed. -Example: ["Exhibit A: Commercial Services"] or ["Attachment B - Provider Compensation Schedule"] or ["N/A"] -{JSON_LIST_FORMAT_INSTRUCTIONS}""" - - -def EXHIBIT_HEADER(context) -> Tuple[str, Callable[[str], list]]: - """Returns ONLY dynamic content for exhibit header extraction. - Call EXHIBIT_HEADER_INSTRUCTION() separately for the cached instruction. - Note: Header markers are now included in EXHIBIT_HEADER_INSTRUCTION() for caching. - - Returns: - Tuple of (prompt_string, parser_function) where parser expects JSON list output. - """ - prompt = f"""[CONTEXT] -Here is the text to analyze: -{context.replace('"', "'")}""" - - return (prompt, _json_list_parser) - - -def EXHIBIT_HEADER_NEW_INSTRUCTION() -> str: - """Static instruction for EXHIBIT_HEADER_NEW prompt caching. Returns formatted instructions for extracting headers from chunked contract pages. """ - return ( - "[OBJECTIVE]\n" - "Extract section headers from contract page chunks with boundary markers.\n" - "Return a JSON dict in |pipes| with boundary numbers as keys and header as values." - ) - - -def EXHIBIT_HEADER_NEW(context, EXHIBIT_HEADER_MARKERS): - """ - Returns prompt for exhibit header extraction from chunked contract pages. - Chunks are separated by boundary markers like ===BOUNDARY_X_HEADER===. - Call EXHIBIT_HEADER_NEW_INSTRUCTION() separately for the cached instruction. - - Args: - context (str): Contract text with boundary markers separating chunks - EXHIBIT_HEADER_MARKERS (list): List of keywords/phrases that indicate headers - - Returns: - str: Formatted prompt for LLM to extract headers and return as JSON dict - """ - return f"""Analyze the following contract page chunks and extract section headers from each chunk. + return """[OBJECTIVE] +Extract section headers from contract page chunks with boundary markers. +Return a JSON dict in |pipes| with boundary numbers as keys and header as values. Each chunk is separated by a boundary marker in the format ===BOUNDARY_X_HEADER=== where X is the boundary number. The following keywords and phrases should be considered as exhibit/attachment headers: -* {'\n* '.join(EXHIBIT_HEADER_MARKERS)} +* EXHIBIT +* ATTACHMENT +* ARTICLE +* AMENDMENT +* SCHEDULE +* ADDENDUM +* APPENDIX +* PARTICIPATING PROVIDER AGREEMENT Rules for Inclusion: * Header can be multi-lined, include all lines that are part of the header @@ -2025,16 +1954,30 @@ Return a JSON dictionary enclosed in |pipes| where: - Keys are boundary numbers (e.g., "1", "2", "3") - Values are the header strings found in that chunk - If no headers are found in a chunk, return "N/A" for that key -- Return {{}} if no headers are found in any chunk +- Return {} if no headers are found in any chunk Example output: -|{{"1": "PROVIDER SERVICES AGREEMENT\\nSIGNATURE PAGE", "2": "N/A", "3": "Attachment A: Fee Schedule"}}| +|{"1": "PROVIDER SERVICES AGREEMENT\\nSIGNATURE PAGE", "2": "N/A", "3": "Attachment A: Fee Schedule"}|""" + +def EXHIBIT_HEADER(context) -> Tuple[str, Callable[[str], dict]]: + """ + Returns ONLY dynamic content for exhibit header extraction. + Call EXHIBIT_HEADER_INSTRUCTION() separately for the cached instruction. + + Args: + context (str): Contract text with boundary markers separating chunks + + Returns: + Tuple of (prompt_string, parser_function) where parser expects JSON dict output. + """ + prompt = f"""[CONTEXT] Here are the chunks to analyze: {context.replace('"', "'")} -Return ONLY the |{{dictionary}}| followed by a brief explanation. -""" +Return ONLY the |{{dictionary}}| followed by a brief explanation.""" + + return (prompt, _json_dict_parser) def EXHIBIT_LINKAGE_INSTRUCTION() -> str: