afb6d5185d
Feature/lesser table caching refactor hybrid * chore: Remove unused duplicate main.py from shared pipeline * fix: Correct crosswalk paths in aarete_derived.py * chore: Remove unused documentation files from fieldExtraction * docs: Add documentation files to documentation folder * docs: Update README with uv setup, expanded project structure, and branching conventions * docs: Add uv installation steps with Ubuntu/WSL emphasis * Enable prompt caching for all remaining LLM calls - Add _INSTRUCTION() functions for: EXHIBIT_HEADER, EXHIBIT_LINKAGE, EXHIBIT_TITLE_MATCH, DATE_FIX, DERIVED_TERM_DATE, CHECK_PROVIDER_NAME_MATCH, SPECIAL_CASE_ASSIGNMENT - Update all invoke_claude() calls in saas and clover pipelines to use cache=True with corresponding _INSTRUCTION() functions - Add new instructions to get_cacheable_instructions() for cache warming - Update tests for new instruction functions Functions now using caching: - prompt_exhibit_level - prompt_exhibit_lesser (EXHIBIT_LEVEL_LESSER_OF) - prompt_fee_schedule_breakout - prompt_grouper_breakout - prompt_special_case_assignment - prompt_exhibit_linkage - prompt_exhibit_header - prompt_smart_chunked (ONE_TO_ONE templates) - prompt_date_fix - prompt_derived_term_date - prompt_exhibit_title_match - provider_name_match_check 🤖 Generated with [Claude Code](https://claude.com/claude-code) Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com> * Reorder * feat: Add bcbs_promise client pipeline with OFFSET_TERM extraction - Add new bcbs_promise client with HSC-based OFFSET_TERM field extraction - Extract full paragraph text of offset/recoupment provisions from contracts - Derive OFFSET_INDICATOR (Y/N) from OFFSET_TERM presence - Fix reorder_columns to preserve extra columns not in COLUMN_ORDER - Update QC/QA output path to outputs/qc_qa/ * fix: Update dev deps and test assertions for QC/QA output path - Add pytest/pytest-mock to dev dependencies for mypy type checking - Update test assertions to expect outputs/qc_qa instead of qa_qc_output * style: Apply black formatting to prompt_templates.py * Merge main, move scripts * Archive some scripts * update py version * remove .py version file * Remove ASCII characters * Restore testbed code * restore tracking * Update testbed metrics * Enable prompt caching for CODE_LAST_CHECK, FILL_BILL_TYPE, DUAL_LOB_CHECK, and GROUPER_BREAKOUT - Add CODE_LAST_CHECK_INSTRUCTION() for service specificity classification - Add FILL_BILL_TYPE_INSTRUCTION() for bill type code determination - Add DUAL_LOB_CHECK_INSTRUCTION() for Medicare/Medicaid classification - Update code_funcs.py to use caching for CODE_LAST_CHECK, FILL_BILL_TYPE, GROUPER_BREAKOUT - Update postprocessing_funcs.py to use caching for DUAL_LOB_CHECK - Add new instructions to get_cacheable_instructions() for cache warming - Add unit tests for new instruction functions 🤖 Generated with [Claude Code](https://claude.com/claude-code) Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com> * Fix postprocessing_funcs to remove invalid columns * Merge branch 'main' into feature/lesser-table-caching-refactor-hybrid * Revert prompt caching changes from aed1b73c * update formatting * Update imports Approved-by: Sha Brown Approved-by: Praneel Panchigar
101 lines
4.0 KiB
Python
101 lines
4.0 KiB
Python
import boto3
|
|
import csv
|
|
import os
|
|
import pandas as pd
|
|
|
|
"""
|
|
This script is used for CNC diff analysis. It compares two folders (source1 and source2) and writes the comparison results to a CSV file.
|
|
|
|
source1_type and source2_type should be one of the following:
|
|
- local: for local directories
|
|
- s3: for S3 folders
|
|
- csv: for CSV files
|
|
|
|
S3 credentials should be stored in the AWS credentials file under the profile name 'temp_cred' or can be changed in the aws credentials set.
|
|
|
|
"""
|
|
|
|
def list_s3_files(bucket_name, folder):
|
|
s3 = boto3.Session(profile_name='temp_cred').client('s3')
|
|
paginator = s3.get_paginator('list_objects_v2')
|
|
operation_parameters = {'Bucket': bucket_name, 'Prefix': folder}
|
|
file_names = set()
|
|
|
|
for page in paginator.paginate(**operation_parameters):
|
|
if 'Contents' in page:
|
|
for content in page['Contents']:
|
|
file_name = content['Key']
|
|
if not file_name.endswith('/'):
|
|
file_name_without_extension = file_name.replace(folder, '', 1)[:-4]
|
|
# file_name_without_extension = file_name[:-4]
|
|
file_names.add(file_name_without_extension)
|
|
|
|
return file_names
|
|
|
|
def list_local_files(path):
|
|
files = set()
|
|
try:
|
|
for _, _, filenames in os.walk(path):
|
|
for filename in filenames:
|
|
files.add(filename[:-4])
|
|
except Exception as e:
|
|
print(f"Error accessing directory {path}: {e}")
|
|
return files
|
|
|
|
def compare_s3_folders(bucket_name, source1, folder1, source2, folder2, output_csv):
|
|
|
|
if source1 == 's3':
|
|
files_in_folder1 = list_s3_files(bucket_name, folder1)
|
|
elif source1 == 'local':
|
|
files_in_folder1 = list_local_files(folder1)
|
|
elif source1 == 'csv':
|
|
df = pd.read_csv(folder1,encoding='utf-8')
|
|
# df['File Name'] = df['File Name'][:-4]
|
|
files_in_folder1 = set(df['File Name'].to_list())
|
|
elif source1 == 'xlsb':
|
|
with pd.ExcelFile(folder1, engine='pyxlsb') as xlsb:
|
|
first_sheet = xlsb.sheet_names[0]
|
|
df = xlsb.parse(first_sheet)
|
|
files_in_folder1 = set(df['Contract Name'].to_list())
|
|
|
|
if source2 == 's3':
|
|
files_in_folder2 = list_s3_files(bucket_name, folder2)
|
|
elif source2 == 'local':
|
|
files_in_folder2 = list_local_files(folder2)
|
|
elif source2 == 'csv':
|
|
df = pd.read_csv(folder2,encoding='utf-8')
|
|
# df['File Name'] = df['File Name'][:-4]
|
|
files_in_folder2 = set(df['File Name'].to_list())
|
|
elif source2 == 'xlsb':
|
|
with pd.ExcelFile(folder2, engine='pyxlsb') as xlsb:
|
|
first_sheet = xlsb.sheet_names[0]
|
|
df = xlsb.parse(first_sheet)
|
|
files_in_folder2 = set(df['Contract Name'].to_list())
|
|
|
|
|
|
common_files = files_in_folder1 & files_in_folder2
|
|
only_in_folder1 = files_in_folder1 - files_in_folder2
|
|
only_in_folder2 = files_in_folder2 - files_in_folder1
|
|
|
|
with open(output_csv, 'w', newline='', encoding='utf-8') as csvfile:
|
|
csv_writer = csv.writer(csvfile)
|
|
csv_writer.writerow(['Common Files', f'Only in {folder1}', f'Only in {folder2}'])
|
|
|
|
max_length = max(len(common_files), len(only_in_folder1), len(only_in_folder2))
|
|
for i in range(max_length):
|
|
row = [
|
|
list(common_files)[i] if i < len(common_files) and list(common_files)[i] else '',
|
|
list(only_in_folder1)[i] if i < len(only_in_folder1) and list(only_in_folder1)[i] else '',
|
|
list(only_in_folder2)[i] if i < len(only_in_folder2) and list(only_in_folder2)[i] else ''
|
|
]
|
|
csv_writer.writerow(row)
|
|
|
|
bucket_name = 'centene-national-contracting-files'
|
|
source1_type = 'local' # local / s3 / csv / xlsb
|
|
source1 = 'Batch 6 TXT Files'
|
|
source2_type = 'csv' # local / s3 / csv / xlsb
|
|
source2 = 'Batch 6 Outputs'
|
|
output_csv = 'batch6/batch6_outputs_diff_with_tracker.csv'
|
|
|
|
compare_s3_folders(bucket_name, source1_type, source1, source2_type, source2, output_csv)
|
|
print(f'Comparison results have been written to {output_csv}') |