Files
doczyai-pipelines/fieldExtraction/scripts/qa_qc.py
T
Faizan Mohiuddin 9a736fa80d Merged in feature/self-repair-main (pull request #313)
Feature/self repair main

* fixes

* fixes

* fixes

* fixes

* fixes

* fixes

* fixes

* fixes

* new testing

* new testing

* added unit tests under fieldExtraction/tests/main-unit-tests

* unit test fixes

* unit test fixes

* merge into main

* modified consolidate_output and slight changes to qa_qc_helpers to accomodate qaqc functionality w faizan and michael's changes

* merging self repair into main

* synced with main now

* fixes

* provider billing removed

* provider billing removed

* provider billing fixed

* commented out load dotenv in config

* Merged main into feature/self-repair-main

* modified check_s3_bucket_exists function by returning False if credentials not found for pipeline

* Merge branch 'feature/self-repair-main' of https://bitbucket.org/aarete/doczy.ai into feature/self-repair-main


Approved-by: Michael McGuinness
2024-12-17 16:34:59 +00:00

111 lines
4.1 KiB
Python

import pandas as pd
import os
import sys
from datetime import datetime
import re
from openpyxl import Workbook
from openpyxl.utils.dataframe import dataframe_to_rows
import argparse
import boto3
from botocore.exceptions import ClientError
sys.path.append("src")
import utils
import config
from qa_qc_helpers import (
perform_qc_qa,
save_qcqa_wb
)
from consolidate_output import consolidate_output
file_name_column = config.FILE_NAME_COLUMN
"""python scripts/qa_qc.py input_dir=______ output_dir=_______ ac_df=________ b_df=______ read_mode=____ df_read_mode=________
You should have both input and output (input is local or s3, ouput is just an empty folder) and then one of ac_df or b_df or both
The final report will be saved in the parent directory of when you are running under reports/
Run this from the base folder of the repo."""
def main():
timestamp = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
script_dir = os.path.dirname(os.path.abspath(__file__))
parent_dir = os.path.dirname(script_dir)
reports_dir = os.path.join(parent_dir, "reports")
os.makedirs(reports_dir, exist_ok=True)
if config.AC_DF:
print(f"AC DataFrame: {config.AC_DF}")
if config.B_DF:
print(f"B DataFrame: {config.B_DF}")
if config.LOCAL_PATH:
print(f"Input Dir: {config.LOCAL_PATH}")
if (config.S3_BUCKET) and (config.S3_PREFIX):
print(f"S3 Bucket: {config.S3_BUCKET}")
print(f"S3 Prefix: {config.S3_PREFIX}")
print(f"Output Dir: {config.OUTPUT_DIRECTORY}")
print(f"Reports Dir: {reports_dir}")
if not config.LOCAL_PATH or (config.S3_BUCKET and config.S3_PREFIX):
print("Needs an Input Directory!")
if not config.AC_DF and not config.B_DF:
print("Needs at least one of AC Outputs and B Outputs!")
return
input_dict = utils.read_input()
ac_df, b_df = utils.read_input_csv()
total_files = len(input_dict)
print(f"Total Input Files: {total_files}")
all_input_filenames = (
[utils.remove_txt_extension(f.lower()) for f in input_dict.keys()]
if input_dict
else []
)
if ac_df is not None:
ac_df = utils.remove_unnamed_columns(ac_df)
ac_df[file_name_column] = ac_df[file_name_column].apply(
lambda x: utils.remove_txt_extension(str(x)).lower()
)
ac_filenames = set(ac_df[file_name_column])
if b_df is not None:
b_df = utils.remove_unnamed_columns(b_df)
b_filenames = set(b_df[file_name_column])
b_df[file_name_column] = b_df[file_name_column].apply(
lambda x: utils.remove_txt_extension(str(x)).lower()
)
if ac_df is not None and b_df is not None:
unmatched_b_filenames = b_filenames - ac_filenames
unmatched_b_df = pd.DataFrame({file_name_column: list(unmatched_b_filenames)})
unmatched_b_output_path = os.path.join(config.OUTPUT_DIRECTORY, "unmatched_B_filenames.csv")
unmatched_b_df.to_csv(unmatched_b_output_path, index=False)
print(f"Unmatched B filenames saved to '{unmatched_b_output_path}'")
# Perform an inner merge to keep only matching rows for the main output
ac_df, b_df, merged_df = consolidate_output(timestamp, ac_df, b_df)
merged_df = utils.remove_unnamed_columns(merged_df)
ac_rows_before = len(ac_df)
b_rows_before = len(b_df)
merged_rows = len(merged_df)
unmatched_b_count = len(unmatched_b_filenames)
print(f"Rows in AC DataFrame before merge: {ac_rows_before}")
print(f"Rows in B DataFrame before merge: {b_rows_before}")
print(f"Rows in merged DataFrame: {merged_rows}")
print(f"Unique filenames in B not found in AC: {unmatched_b_count}")
merged_output_path = os.path.join(config.OUTPUT_DIRECTORY, "DOCZYAI_merged_results.csv")
merged_df.to_csv(merged_output_path, index=False)
print(f"Merged data saved to '{merged_output_path}'")
wb = perform_qc_qa(ac_df, b_df, merged_df, input_dict)
save_qcqa_wb(wb, timestamp)
if __name__ == "__main__":
main()