1098be5cf5
grouped keywords and IRS regex chunking * abandon old branch and rebuild * removed print for field names, prompts * pushing as part of leftovers * IRS regex chunking * Merged in feature/chunk_term_clean (pull request #262) chunk_log_added * chunk_log_added * Merge remote-tracking branch 'remotes/origin/feature_ac_chunking_clean' into feature/chunk_term_clean Approved-by: Alex Galarce * added function for regex based IRS chunking * shifted regex_match_chunk function to utils * added helper functions for regex based chunking * replaced tin_regex function to ac_funcs * removed prompt for irs_others * added regex chunking for irs * removed irs group from smart chunking * testing functionality for regex based irs fields * updated return N/A condition * Merged main into feature_ac_chunking_clean * file_processing.py edited online with Bitbucket * added back conditional prompts * minor fixed for PR * remove install types * mering clean branch * with passed test casses * added flag for regex based tin execution Approved-by: Michael McGuinness
42 lines
1.2 KiB
Python
42 lines
1.2 KiB
Python
import argparse
|
|
import os
|
|
from utils import read_local
|
|
import preprocessing_funcs
|
|
import ac_funcs
|
|
|
|
|
|
def parse_arguments():
|
|
parser = argparse.ArgumentParser(description="Smart Chunking Tester")
|
|
parser.add_argument(
|
|
"--input_dir", help="Input directory (local path or S3 URI)", default="src/new/"
|
|
)
|
|
return parser.parse_args()
|
|
|
|
|
|
def main():
|
|
args = parse_arguments()
|
|
input_dir = args.input_dir
|
|
|
|
for file in os.listdir(input_dir):
|
|
full_path = os.path.join(input_dir, file)
|
|
print("--" * 20)
|
|
print(file)
|
|
contract_text = read_local(full_path)
|
|
contract_text = preprocessing_funcs.clean_newlines(contract_text)
|
|
contract_text = preprocessing_funcs.clean_law_symbols(contract_text)
|
|
text_dict = preprocessing_funcs.split_text(
|
|
contract_text
|
|
) # return a dictionary with keys - page_num (str), values as the page_text
|
|
|
|
irs_answers = ac_funcs.tin_regex(filename=str(file), text_dict=text_dict)
|
|
if irs_answers is None:
|
|
irs_answers = {}
|
|
irs_answers["PROV_GROUP_TIN"] = "N/A"
|
|
irs_answers["PROV_TIN_OTHER"] = "N/A"
|
|
|
|
print(irs_answers)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|