afb6d5185d
Feature/lesser table caching refactor hybrid * chore: Remove unused duplicate main.py from shared pipeline * fix: Correct crosswalk paths in aarete_derived.py * chore: Remove unused documentation files from fieldExtraction * docs: Add documentation files to documentation folder * docs: Update README with uv setup, expanded project structure, and branching conventions * docs: Add uv installation steps with Ubuntu/WSL emphasis * Enable prompt caching for all remaining LLM calls - Add _INSTRUCTION() functions for: EXHIBIT_HEADER, EXHIBIT_LINKAGE, EXHIBIT_TITLE_MATCH, DATE_FIX, DERIVED_TERM_DATE, CHECK_PROVIDER_NAME_MATCH, SPECIAL_CASE_ASSIGNMENT - Update all invoke_claude() calls in saas and clover pipelines to use cache=True with corresponding _INSTRUCTION() functions - Add new instructions to get_cacheable_instructions() for cache warming - Update tests for new instruction functions Functions now using caching: - prompt_exhibit_level - prompt_exhibit_lesser (EXHIBIT_LEVEL_LESSER_OF) - prompt_fee_schedule_breakout - prompt_grouper_breakout - prompt_special_case_assignment - prompt_exhibit_linkage - prompt_exhibit_header - prompt_smart_chunked (ONE_TO_ONE templates) - prompt_date_fix - prompt_derived_term_date - prompt_exhibit_title_match - provider_name_match_check 🤖 Generated with [Claude Code](https://claude.com/claude-code) Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com> * Reorder * feat: Add bcbs_promise client pipeline with OFFSET_TERM extraction - Add new bcbs_promise client with HSC-based OFFSET_TERM field extraction - Extract full paragraph text of offset/recoupment provisions from contracts - Derive OFFSET_INDICATOR (Y/N) from OFFSET_TERM presence - Fix reorder_columns to preserve extra columns not in COLUMN_ORDER - Update QC/QA output path to outputs/qc_qa/ * fix: Update dev deps and test assertions for QC/QA output path - Add pytest/pytest-mock to dev dependencies for mypy type checking - Update test assertions to expect outputs/qc_qa instead of qa_qc_output * style: Apply black formatting to prompt_templates.py * Merge main, move scripts * Archive some scripts * update py version * remove .py version file * Remove ASCII characters * Restore testbed code * restore tracking * Update testbed metrics * Enable prompt caching for CODE_LAST_CHECK, FILL_BILL_TYPE, DUAL_LOB_CHECK, and GROUPER_BREAKOUT - Add CODE_LAST_CHECK_INSTRUCTION() for service specificity classification - Add FILL_BILL_TYPE_INSTRUCTION() for bill type code determination - Add DUAL_LOB_CHECK_INSTRUCTION() for Medicare/Medicaid classification - Update code_funcs.py to use caching for CODE_LAST_CHECK, FILL_BILL_TYPE, GROUPER_BREAKOUT - Update postprocessing_funcs.py to use caching for DUAL_LOB_CHECK - Add new instructions to get_cacheable_instructions() for cache warming - Add unit tests for new instruction functions 🤖 Generated with [Claude Code](https://claude.com/claude-code) Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com> * Fix postprocessing_funcs to remove invalid columns * Merge branch 'main' into feature/lesser-table-caching-refactor-hybrid * Revert prompt caching changes from aed1b73c * update formatting * Update imports Approved-by: Sha Brown Approved-by: Praneel Panchigar
99 lines
2.8 KiB
Python
99 lines
2.8 KiB
Python
import logging
|
|
from airflow.exceptions import AirflowFailException
|
|
from airflow.operators.empty import EmptyOperator
|
|
from airflow.operators.python import PythonOperator
|
|
from datetime import datetime
|
|
from airflow import DAG
|
|
from airflow.providers.snowflake.hooks.snowflake import SnowflakeHook
|
|
|
|
# from openpyxl.workbook import Workbook
|
|
import pandas as pd
|
|
import boto3
|
|
import io
|
|
|
|
|
|
SNOWFLAKE_CONN_ID = "doczy_uat_snowflake"
|
|
TAGS = ["QC", "uat", "etl", "QC-Summery"]
|
|
DAG_ID = "qc_report_dag"
|
|
|
|
bucket = "doczy-dev-infra-mwaa-resources"
|
|
object_key = "outputs/"
|
|
|
|
DATABASE = "DOCZY_UAT"
|
|
SCHEMA = "STG"
|
|
TABLE_NAME = "TRAINING_DATA_RAW"
|
|
|
|
# Trigger rules
|
|
ALL_SUCCESS = "all_success"
|
|
ALL_FAILED = "all_failed"
|
|
ALL_DONE = "all_done"
|
|
ONE_SUCCESS = "one_success"
|
|
ONE_FAILED = "one_failed"
|
|
|
|
|
|
args = {"owner": "Airflow", "start_date": datetime(2022, 1, 1), "retries": 0}
|
|
dag = DAG(dag_id=DAG_ID, default_args=args, schedule=None, tags=TAGS)
|
|
|
|
|
|
def getData():
|
|
# Setup connection to Snowflake
|
|
dwh_hook = SnowflakeHook(snowflake_conn_id=SNOWFLAKE_CONN_ID)
|
|
conn = dwh_hook.get_conn() # Get the raw connection
|
|
|
|
# Your query and the database details
|
|
qc_query = f"SELECT * FROM {DATABASE}.{SCHEMA}.{TABLE_NAME}"
|
|
|
|
# Fetch data into a Pandas DataFrame
|
|
df = pd.read_sql(qc_query, conn)
|
|
|
|
# Initialize S3 client
|
|
s3 = boto3.client("s3")
|
|
|
|
# Create a buffer to hold the data
|
|
with io.StringIO() as csv_buffer:
|
|
df.to_csv(csv_buffer, index=False)
|
|
|
|
# Save the data to S3
|
|
response = s3.put_object(
|
|
Bucket=bucket,
|
|
Key=object_key + "snowflake_table_results.csv",
|
|
Body=csv_buffer.getvalue(),
|
|
)
|
|
|
|
status = response.get("ResponseMetadata", {}).get("HTTPStatusCode")
|
|
|
|
if status == 200:
|
|
print(f"Successful S3 put_object response. Status - {status}")
|
|
else:
|
|
raise AirflowFailException(
|
|
f"Unsuccessful S3 put_object response. Status - {status}"
|
|
)
|
|
|
|
oldData = df
|
|
newData = df
|
|
|
|
with io.BytesIO() as output:
|
|
with pd.ExcelWriter(output, engine="xlsxwriter") as writer:
|
|
oldData.to_excel(writer, sheet_name="Old")
|
|
newData.to_excel(writer, sheet_name="new")
|
|
dat = oldData.compare(newData, align_axis=0, keep_shape=True)
|
|
dat.to_excel(writer, sheet_name="qc")
|
|
response = s3.put_object(
|
|
Bucket=bucket, Key=object_key + "QC_Report.xlsx", Body=output.getvalue()
|
|
)
|
|
|
|
print("Dataframe is written to S3 successfully.")
|
|
|
|
|
|
with dag:
|
|
begin_job = EmptyOperator(task_id="Begin")
|
|
|
|
get_data_from_snowflake = PythonOperator(
|
|
task_id="get_data_from_snowflake", python_callable=getData
|
|
)
|
|
|
|
end_job = EmptyOperator(task_id="End")
|
|
|
|
|
|
begin_job >> get_data_from_snowflake >> end_job
|