154 lines
6.7 KiB
Python
154 lines
6.7 KiB
Python
|
|
|
||
|
|
from itertools import groupby, count
|
||
|
|
import re
|
||
|
|
|
||
|
|
def clean_newlines(contract_text):
|
||
|
|
"""
|
||
|
|
Cleans up isolated newlines in a contract text, converting them into spaces to ensure text continuity.
|
||
|
|
|
||
|
|
This function uses a regular expression to identify standalone newlines (those not part of a paragraph break, which
|
||
|
|
typically consists of consecutive newlines) and replaces them with a single space. This process helps in preserving
|
||
|
|
the flow of text and making it more readable and processable, particularly for text extraction and analysis tasks that
|
||
|
|
might be sensitive to abrupt breaks in the text.
|
||
|
|
|
||
|
|
Parameters:
|
||
|
|
contract_text (str): The text of the contract that may contain sporadic newlines.
|
||
|
|
|
||
|
|
Returns:
|
||
|
|
str: The cleaned contract text with isolated newlines replaced by spaces.
|
||
|
|
"""
|
||
|
|
cleaned_text = re.sub(r'(?<!\n)\n(?!\n)', ' ', contract_text)
|
||
|
|
return cleaned_text
|
||
|
|
|
||
|
|
|
||
|
|
def split_text(text):
|
||
|
|
"""
|
||
|
|
Splits a contract text into a dictionary of pages based on a specific delimiter indicating the start of a page.
|
||
|
|
|
||
|
|
This function uses the delimiter 'Start of Page No. = ' to segment the input text into different pages.
|
||
|
|
Each page is stored in a dictionary where the key is the page number (extracted as the first word after the delimiter)
|
||
|
|
and the value is the corresponding page text. This facilitates easier access and processing of individual pages in the contract.
|
||
|
|
|
||
|
|
Parameters:
|
||
|
|
text (str): The contract text containing multiple pages separated by the delimiter.
|
||
|
|
|
||
|
|
Returns:
|
||
|
|
dict: A dictionary with page numbers as keys and the respective page texts as values.
|
||
|
|
"""
|
||
|
|
text_list = text.split('Start of Page No. = ')
|
||
|
|
text_dict = {page.split()[0] : page for page in text_list}
|
||
|
|
return text_dict
|
||
|
|
|
||
|
|
|
||
|
|
def highlight_rates(text_dict):
|
||
|
|
"""
|
||
|
|
Highlights rate-related terms (percentages and dollar amounts) in the text of each page within a dictionary.
|
||
|
|
|
||
|
|
This function processes each page's text in the provided dictionary, identifying words that contain percentage signs ('%')
|
||
|
|
or dollar signs ('$'). It wraps these words with '>>>' and '<<<' to highlight them. The modified text for each page
|
||
|
|
is then updated in the dictionary, allowing for easier identification and processing of rate-related terms in subsequent analyses.
|
||
|
|
|
||
|
|
Parameters:
|
||
|
|
text_dict (dict): A dictionary where keys are page numbers and values are the text content of those pages.
|
||
|
|
|
||
|
|
Returns:
|
||
|
|
dict: The updated dictionary with rate-related terms highlighted in the text of each page.
|
||
|
|
"""
|
||
|
|
for page, text in text_dict.items():
|
||
|
|
words = text.split()
|
||
|
|
highlighted_words = []
|
||
|
|
for word in words:
|
||
|
|
if '%' in word or '$' in word:
|
||
|
|
word = f'>>>{word}<<<'
|
||
|
|
highlighted_words.append(word)
|
||
|
|
text_dict[page] = ' '.join(highlighted_words)
|
||
|
|
return text_dict
|
||
|
|
|
||
|
|
|
||
|
|
def chunk_text(text_dict):
|
||
|
|
"""
|
||
|
|
Creates text chunks from a dictionary of page texts, focusing on pages with special characters (percentages and dollar amounts).
|
||
|
|
|
||
|
|
This function identifies pages that contain '%' or '$' signs and includes those pages along with their immediate neighbors
|
||
|
|
(previous and next pages) to form chunks. The chunks are then grouped and concatenated into single text blocks for easier
|
||
|
|
processing. Each chunk is stored in a new dictionary where the keys represent the range of pages included in the chunk.
|
||
|
|
|
||
|
|
Parameters:
|
||
|
|
text_dict (dict): A dictionary where keys are page numbers (as strings) and values are the text content of those pages.
|
||
|
|
|
||
|
|
Returns:
|
||
|
|
dict: A new dictionary where each key is a string representing the range of pages in a chunk, and each value is the concatenated text of those pages.
|
||
|
|
"""
|
||
|
|
special_pages = {int(page): text for page, text in text_dict.items() if (('%' in text) or ('$' in text) or ('percent' in text))}
|
||
|
|
page_numbers = sorted(special_pages.keys())
|
||
|
|
chunk_page_numbers = []
|
||
|
|
for page in page_numbers:
|
||
|
|
chunk_page_numbers.append(page-1)
|
||
|
|
chunk_page_numbers.append(page)
|
||
|
|
chunk_page_numbers.append(page+1)
|
||
|
|
chunk_page_numbers = list(set(chunk_page_numbers))
|
||
|
|
chunk_page_numbers = [page for page in chunk_page_numbers if str(page) in text_dict.keys()]
|
||
|
|
chunks = [list(group) for key, group in groupby(chunk_page_numbers, lambda x, c=count(): x - next(c))]
|
||
|
|
chunk_dict = {}
|
||
|
|
for item in chunks:
|
||
|
|
dict_key = f'{min(item)}-{max(item)}'
|
||
|
|
text = "".join([text_dict[str(page)] for page in item])
|
||
|
|
chunk_dict[dict_key] = text
|
||
|
|
return chunk_dict
|
||
|
|
|
||
|
|
|
||
|
|
def clean_billed_charges(contract_text):
|
||
|
|
"""
|
||
|
|
Cleans occurrences of 'billed charges' in a text by replacing them with '100% of billed charges' unless preceded by a percentage.
|
||
|
|
|
||
|
|
This function uses a regular expression to find all instances of the phrase 'billed charges'. For each match, it checks the preceding
|
||
|
|
text (up to 15 characters before the match) for a percentage sign ('%'). If a percentage is found, it leaves the match unchanged.
|
||
|
|
Otherwise, it replaces 'billed charges' with '100% of billed charges'.
|
||
|
|
|
||
|
|
Parameters:
|
||
|
|
text (str): The input text to be cleaned.
|
||
|
|
|
||
|
|
Returns:
|
||
|
|
str: The cleaned text with appropriate replacements made for 'billed charges'.
|
||
|
|
"""
|
||
|
|
|
||
|
|
substrings = [
|
||
|
|
"Physician's Billed Charges",
|
||
|
|
"Provider's Billed Charges",
|
||
|
|
"Allowable Billed Charges",
|
||
|
|
"Hospital's Billed Charges",
|
||
|
|
"Physician's Charges",
|
||
|
|
"Provider's Charges",
|
||
|
|
"Allowable Charges",
|
||
|
|
"Hospital's Charges",
|
||
|
|
"Billed Charges",
|
||
|
|
"the rates set forth in this Exhibit",
|
||
|
|
"the rutes set forth in this Exhibit",
|
||
|
|
"Provider's usual and customary charge"
|
||
|
|
]
|
||
|
|
max_substring_length = max([len(s) for s in substrings])
|
||
|
|
|
||
|
|
def find_substring_indices(contract_text, s):
|
||
|
|
indices = []
|
||
|
|
lower_contract_text = contract_text.lower()
|
||
|
|
lower_s = s.lower()
|
||
|
|
index = lower_contract_text.find(lower_s)
|
||
|
|
while index != -1:
|
||
|
|
indices.append(index)
|
||
|
|
index = lower_contract_text.find(lower_s, index + 1)
|
||
|
|
return indices
|
||
|
|
|
||
|
|
for s in substrings:
|
||
|
|
indices = find_substring_indices(contract_text, s)
|
||
|
|
index_adder = 0
|
||
|
|
if indices:
|
||
|
|
for i in indices:
|
||
|
|
index = i+index_adder
|
||
|
|
end_index = index + max_substring_length
|
||
|
|
match_part = contract_text[index:end_index]
|
||
|
|
previous = contract_text[max(0, index-30):index]
|
||
|
|
if '%' not in previous:
|
||
|
|
contract_text = contract_text[0:index] + f" 100% of {match_part}" + contract_text[end_index:]
|
||
|
|
index_adder += 9
|
||
|
|
return contract_text.replace(' ', ' ')
|