Merged in feature/integrate-standard-and-complex (pull request #356)
Feature/integrate standard and complex * last edits to merge_tables_draft_v2 before moving to table_funcs * migrate to table_utils.py * fix typehint that caused mypy error * fix mypy errors * black and isort * fix error with multi-table pages * update poetry.lock * add tests, fix bug when table start marker or table end marker is missing * added more tests * tests for get_str_dictionaries_from_text() * fix edge case in get_str_dictionaries_from_text * fix mypy error * more tests * add tests, add handling for invalid dictionaries * add tests * add tests for insert_column_headers() * update requirements * add new simple/complex logic to preprocess.py * remove files used solely for testing * split pages into sub-pages by tables * add table split on end marker. Also add tests * add docstrings, change control flow * remove unused tests, add TODO for new tests * get in prompt changes and intermediate decisions * TODO for future enhancement for get_exhibit_pages Approved-by: Katon Minhas
This commit is contained in:
Generated
+211
-118
@@ -288,32 +288,32 @@ css = ["tinycss2 (>=1.1.0,<1.5)"]
|
||||
|
||||
[[package]]
|
||||
name = "boto3"
|
||||
version = "1.35.97"
|
||||
version = "1.36.4"
|
||||
description = "The AWS SDK for Python"
|
||||
optional = false
|
||||
python-versions = ">=3.8"
|
||||
files = [
|
||||
{file = "boto3-1.35.97-py3-none-any.whl", hash = "sha256:8e49416216a6e3a62c2a0c44fba4dd2852c85472e7b702516605b1363867d220"},
|
||||
{file = "boto3-1.35.97.tar.gz", hash = "sha256:7d398f66a11e67777c189d1f58c0a75d9d60f98d0ee51b8817e828930bf19e4e"},
|
||||
{file = "boto3-1.36.4-py3-none-any.whl", hash = "sha256:9f8f699e75ec63fcc98c4dd7290997c7c06c68d3ac8161ad4735fe71f5fe945c"},
|
||||
{file = "boto3-1.36.4.tar.gz", hash = "sha256:eeceeb74ef8b65634d358c27aa074917f4449dc828f79301f1075232618eb502"},
|
||||
]
|
||||
|
||||
[package.dependencies]
|
||||
botocore = ">=1.35.97,<1.36.0"
|
||||
botocore = ">=1.36.4,<1.37.0"
|
||||
jmespath = ">=0.7.1,<2.0.0"
|
||||
s3transfer = ">=0.10.0,<0.11.0"
|
||||
s3transfer = ">=0.11.0,<0.12.0"
|
||||
|
||||
[package.extras]
|
||||
crt = ["botocore[crt] (>=1.21.0,<2.0a0)"]
|
||||
|
||||
[[package]]
|
||||
name = "botocore"
|
||||
version = "1.35.97"
|
||||
version = "1.36.4"
|
||||
description = "Low-level, data-driven core of boto 3."
|
||||
optional = false
|
||||
python-versions = ">=3.8"
|
||||
files = [
|
||||
{file = "botocore-1.35.97-py3-none-any.whl", hash = "sha256:fed4f156b1a9b8ece53738f702ba5851b8c6216b4952de326547f349cc494f14"},
|
||||
{file = "botocore-1.35.97.tar.gz", hash = "sha256:88f2fab29192ffe2f2115d5bafbbd823ff4b6eb2774296e03ec8b5b0fe074f61"},
|
||||
{file = "botocore-1.36.4-py3-none-any.whl", hash = "sha256:3f183aa7bb0c1ba02171143a05f28a4438abdf89dd6b8c0a7727040375a90520"},
|
||||
{file = "botocore-1.36.4.tar.gz", hash = "sha256:ef54f5e3316040b6ff775941e6ed052c3230dda0079d17d9f9e3c757375f2027"},
|
||||
]
|
||||
|
||||
[package.dependencies]
|
||||
@@ -322,7 +322,7 @@ python-dateutil = ">=2.1,<3.0.0"
|
||||
urllib3 = {version = ">=1.25.4,<2.2.0 || >2.2.0,<3", markers = "python_version >= \"3.10\""}
|
||||
|
||||
[package.extras]
|
||||
crt = ["awscrt (==0.22.0)"]
|
||||
crt = ["awscrt (==0.23.4)"]
|
||||
|
||||
[[package]]
|
||||
name = "certifi"
|
||||
@@ -557,39 +557,113 @@ traitlets = ">=4"
|
||||
[package.extras]
|
||||
test = ["pytest"]
|
||||
|
||||
[[package]]
|
||||
name = "coverage"
|
||||
version = "7.6.10"
|
||||
description = "Code coverage measurement for Python"
|
||||
optional = false
|
||||
python-versions = ">=3.9"
|
||||
files = [
|
||||
{file = "coverage-7.6.10-cp310-cp310-macosx_10_9_x86_64.whl", hash = "sha256:5c912978f7fbf47ef99cec50c4401340436d200d41d714c7a4766f377c5b7b78"},
|
||||
{file = "coverage-7.6.10-cp310-cp310-macosx_11_0_arm64.whl", hash = "sha256:a01ec4af7dfeb96ff0078ad9a48810bb0cc8abcb0115180c6013a6b26237626c"},
|
||||
{file = "coverage-7.6.10-cp310-cp310-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:a3b204c11e2b2d883946fe1d97f89403aa1811df28ce0447439178cc7463448a"},
|
||||
{file = "coverage-7.6.10-cp310-cp310-manylinux_2_5_i686.manylinux1_i686.manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:32ee6d8491fcfc82652a37109f69dee9a830e9379166cb73c16d8dc5c2915165"},
|
||||
{file = "coverage-7.6.10-cp310-cp310-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:675cefc4c06e3b4c876b85bfb7c59c5e2218167bbd4da5075cbe3b5790a28988"},
|
||||
{file = "coverage-7.6.10-cp310-cp310-musllinux_1_2_aarch64.whl", hash = "sha256:f4f620668dbc6f5e909a0946a877310fb3d57aea8198bde792aae369ee1c23b5"},
|
||||
{file = "coverage-7.6.10-cp310-cp310-musllinux_1_2_i686.whl", hash = "sha256:4eea95ef275de7abaef630c9b2c002ffbc01918b726a39f5a4353916ec72d2f3"},
|
||||
{file = "coverage-7.6.10-cp310-cp310-musllinux_1_2_x86_64.whl", hash = "sha256:e2f0280519e42b0a17550072861e0bc8a80a0870de260f9796157d3fca2733c5"},
|
||||
{file = "coverage-7.6.10-cp310-cp310-win32.whl", hash = "sha256:bc67deb76bc3717f22e765ab3e07ee9c7a5e26b9019ca19a3b063d9f4b874244"},
|
||||
{file = "coverage-7.6.10-cp310-cp310-win_amd64.whl", hash = "sha256:0f460286cb94036455e703c66988851d970fdfd8acc2a1122ab7f4f904e4029e"},
|
||||
{file = "coverage-7.6.10-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:ea3c8f04b3e4af80e17bab607c386a830ffc2fb88a5484e1df756478cf70d1d3"},
|
||||
{file = "coverage-7.6.10-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:507a20fc863cae1d5720797761b42d2d87a04b3e5aeb682ef3b7332e90598f43"},
|
||||
{file = "coverage-7.6.10-cp311-cp311-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:d37a84878285b903c0fe21ac8794c6dab58150e9359f1aaebbeddd6412d53132"},
|
||||
{file = "coverage-7.6.10-cp311-cp311-manylinux_2_5_i686.manylinux1_i686.manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:a534738b47b0de1995f85f582d983d94031dffb48ab86c95bdf88dc62212142f"},
|
||||
{file = "coverage-7.6.10-cp311-cp311-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:0d7a2bf79378d8fb8afaa994f91bfd8215134f8631d27eba3e0e2c13546ce994"},
|
||||
{file = "coverage-7.6.10-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:6713ba4b4ebc330f3def51df1d5d38fad60b66720948112f114968feb52d3f99"},
|
||||
{file = "coverage-7.6.10-cp311-cp311-musllinux_1_2_i686.whl", hash = "sha256:ab32947f481f7e8c763fa2c92fd9f44eeb143e7610c4ca9ecd6a36adab4081bd"},
|
||||
{file = "coverage-7.6.10-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:7bbd8c8f1b115b892e34ba66a097b915d3871db7ce0e6b9901f462ff3a975377"},
|
||||
{file = "coverage-7.6.10-cp311-cp311-win32.whl", hash = "sha256:299e91b274c5c9cdb64cbdf1b3e4a8fe538a7a86acdd08fae52301b28ba297f8"},
|
||||
{file = "coverage-7.6.10-cp311-cp311-win_amd64.whl", hash = "sha256:489a01f94aa581dbd961f306e37d75d4ba16104bbfa2b0edb21d29b73be83609"},
|
||||
{file = "coverage-7.6.10-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:27c6e64726b307782fa5cbe531e7647aee385a29b2107cd87ba7c0105a5d3853"},
|
||||
{file = "coverage-7.6.10-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:c56e097019e72c373bae32d946ecf9858fda841e48d82df7e81c63ac25554078"},
|
||||
{file = "coverage-7.6.10-cp312-cp312-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:c7827a5bc7bdb197b9e066cdf650b2887597ad124dd99777332776f7b7c7d0d0"},
|
||||
{file = "coverage-7.6.10-cp312-cp312-manylinux_2_5_i686.manylinux1_i686.manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:204a8238afe787323a8b47d8be4df89772d5c1e4651b9ffa808552bdf20e1d50"},
|
||||
{file = "coverage-7.6.10-cp312-cp312-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:e67926f51821b8e9deb6426ff3164870976fe414d033ad90ea75e7ed0c2e5022"},
|
||||
{file = "coverage-7.6.10-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:e78b270eadb5702938c3dbe9367f878249b5ef9a2fcc5360ac7bff694310d17b"},
|
||||
{file = "coverage-7.6.10-cp312-cp312-musllinux_1_2_i686.whl", hash = "sha256:714f942b9c15c3a7a5fe6876ce30af831c2ad4ce902410b7466b662358c852c0"},
|
||||
{file = "coverage-7.6.10-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:abb02e2f5a3187b2ac4cd46b8ced85a0858230b577ccb2c62c81482ca7d18852"},
|
||||
{file = "coverage-7.6.10-cp312-cp312-win32.whl", hash = "sha256:55b201b97286cf61f5e76063f9e2a1d8d2972fc2fcfd2c1272530172fd28c359"},
|
||||
{file = "coverage-7.6.10-cp312-cp312-win_amd64.whl", hash = "sha256:e4ae5ac5e0d1e4edfc9b4b57b4cbecd5bc266a6915c500f358817a8496739247"},
|
||||
{file = "coverage-7.6.10-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:05fca8ba6a87aabdd2d30d0b6c838b50510b56cdcfc604d40760dae7153b73d9"},
|
||||
{file = "coverage-7.6.10-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:9e80eba8801c386f72e0712a0453431259c45c3249f0009aff537a517b52942b"},
|
||||
{file = "coverage-7.6.10-cp313-cp313-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:a372c89c939d57abe09e08c0578c1d212e7a678135d53aa16eec4430adc5e690"},
|
||||
{file = "coverage-7.6.10-cp313-cp313-manylinux_2_5_i686.manylinux1_i686.manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:ec22b5e7fe7a0fa8509181c4aac1db48f3dd4d3a566131b313d1efc102892c18"},
|
||||
{file = "coverage-7.6.10-cp313-cp313-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:26bcf5c4df41cad1b19c84af71c22cbc9ea9a547fc973f1f2cc9a290002c8b3c"},
|
||||
{file = "coverage-7.6.10-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:4e4630c26b6084c9b3cb53b15bd488f30ceb50b73c35c5ad7871b869cb7365fd"},
|
||||
{file = "coverage-7.6.10-cp313-cp313-musllinux_1_2_i686.whl", hash = "sha256:2396e8116db77789f819d2bc8a7e200232b7a282c66e0ae2d2cd84581a89757e"},
|
||||
{file = "coverage-7.6.10-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:79109c70cc0882e4d2d002fe69a24aa504dec0cc17169b3c7f41a1d341a73694"},
|
||||
{file = "coverage-7.6.10-cp313-cp313-win32.whl", hash = "sha256:9e1747bab246d6ff2c4f28b4d186b205adced9f7bd9dc362051cc37c4a0c7bd6"},
|
||||
{file = "coverage-7.6.10-cp313-cp313-win_amd64.whl", hash = "sha256:254f1a3b1eef5f7ed23ef265eaa89c65c8c5b6b257327c149db1ca9d4a35f25e"},
|
||||
{file = "coverage-7.6.10-cp313-cp313t-macosx_10_13_x86_64.whl", hash = "sha256:2ccf240eb719789cedbb9fd1338055de2761088202a9a0b73032857e53f612fe"},
|
||||
{file = "coverage-7.6.10-cp313-cp313t-macosx_11_0_arm64.whl", hash = "sha256:0c807ca74d5a5e64427c8805de15b9ca140bba13572d6d74e262f46f50b13273"},
|
||||
{file = "coverage-7.6.10-cp313-cp313t-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:2bcfa46d7709b5a7ffe089075799b902020b62e7ee56ebaed2f4bdac04c508d8"},
|
||||
{file = "coverage-7.6.10-cp313-cp313t-manylinux_2_5_i686.manylinux1_i686.manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:4e0de1e902669dccbf80b0415fb6b43d27edca2fbd48c74da378923b05316098"},
|
||||
{file = "coverage-7.6.10-cp313-cp313t-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:3f7b444c42bbc533aaae6b5a2166fd1a797cdb5eb58ee51a92bee1eb94a1e1cb"},
|
||||
{file = "coverage-7.6.10-cp313-cp313t-musllinux_1_2_aarch64.whl", hash = "sha256:b330368cb99ef72fcd2dc3ed260adf67b31499584dc8a20225e85bfe6f6cfed0"},
|
||||
{file = "coverage-7.6.10-cp313-cp313t-musllinux_1_2_i686.whl", hash = "sha256:9a7cfb50515f87f7ed30bc882f68812fd98bc2852957df69f3003d22a2aa0abf"},
|
||||
{file = "coverage-7.6.10-cp313-cp313t-musllinux_1_2_x86_64.whl", hash = "sha256:6f93531882a5f68c28090f901b1d135de61b56331bba82028489bc51bdd818d2"},
|
||||
{file = "coverage-7.6.10-cp313-cp313t-win32.whl", hash = "sha256:89d76815a26197c858f53c7f6a656686ec392b25991f9e409bcef020cd532312"},
|
||||
{file = "coverage-7.6.10-cp313-cp313t-win_amd64.whl", hash = "sha256:54a5f0f43950a36312155dae55c505a76cd7f2b12d26abeebbe7a0b36dbc868d"},
|
||||
{file = "coverage-7.6.10-cp39-cp39-macosx_10_9_x86_64.whl", hash = "sha256:656c82b8a0ead8bba147de9a89bda95064874c91a3ed43a00e687f23cc19d53a"},
|
||||
{file = "coverage-7.6.10-cp39-cp39-macosx_11_0_arm64.whl", hash = "sha256:ccc2b70a7ed475c68ceb548bf69cec1e27305c1c2606a5eb7c3afff56a1b3b27"},
|
||||
{file = "coverage-7.6.10-cp39-cp39-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:a5e37dc41d57ceba70956fa2fc5b63c26dba863c946ace9705f8eca99daecdc4"},
|
||||
{file = "coverage-7.6.10-cp39-cp39-manylinux_2_5_i686.manylinux1_i686.manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:0aa9692b4fdd83a4647eeb7db46410ea1322b5ed94cd1715ef09d1d5922ba87f"},
|
||||
{file = "coverage-7.6.10-cp39-cp39-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:aa744da1820678b475e4ba3dfd994c321c5b13381d1041fe9c608620e6676e25"},
|
||||
{file = "coverage-7.6.10-cp39-cp39-musllinux_1_2_aarch64.whl", hash = "sha256:c0b1818063dc9e9d838c09e3a473c1422f517889436dd980f5d721899e66f315"},
|
||||
{file = "coverage-7.6.10-cp39-cp39-musllinux_1_2_i686.whl", hash = "sha256:59af35558ba08b758aec4d56182b222976330ef8d2feacbb93964f576a7e7a90"},
|
||||
{file = "coverage-7.6.10-cp39-cp39-musllinux_1_2_x86_64.whl", hash = "sha256:7ed2f37cfce1ce101e6dffdfd1c99e729dd2ffc291d02d3e2d0af8b53d13840d"},
|
||||
{file = "coverage-7.6.10-cp39-cp39-win32.whl", hash = "sha256:4bcc276261505d82f0ad426870c3b12cb177752834a633e737ec5ee79bbdff18"},
|
||||
{file = "coverage-7.6.10-cp39-cp39-win_amd64.whl", hash = "sha256:457574f4599d2b00f7f637a0700a6422243b3565509457b2dbd3f50703e11f59"},
|
||||
{file = "coverage-7.6.10-pp39.pp310-none-any.whl", hash = "sha256:fd34e7b3405f0cc7ab03d54a334c17a9e802897580d964bd8c2001f4b9fd488f"},
|
||||
{file = "coverage-7.6.10.tar.gz", hash = "sha256:7fb105327c8f8f0682e29843e2ff96af9dcbe5bab8eeb4b398c6a33a16d80a23"},
|
||||
]
|
||||
|
||||
[package.extras]
|
||||
toml = ["tomli"]
|
||||
|
||||
[[package]]
|
||||
name = "debugpy"
|
||||
version = "1.8.11"
|
||||
version = "1.8.12"
|
||||
description = "An implementation of the Debug Adapter Protocol for Python"
|
||||
optional = false
|
||||
python-versions = ">=3.8"
|
||||
files = [
|
||||
{file = "debugpy-1.8.11-cp310-cp310-macosx_14_0_x86_64.whl", hash = "sha256:2b26fefc4e31ff85593d68b9022e35e8925714a10ab4858fb1b577a8a48cb8cd"},
|
||||
{file = "debugpy-1.8.11-cp310-cp310-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:61bc8b3b265e6949855300e84dc93d02d7a3a637f2aec6d382afd4ceb9120c9f"},
|
||||
{file = "debugpy-1.8.11-cp310-cp310-win32.whl", hash = "sha256:c928bbf47f65288574b78518449edaa46c82572d340e2750889bbf8cd92f3737"},
|
||||
{file = "debugpy-1.8.11-cp310-cp310-win_amd64.whl", hash = "sha256:8da1db4ca4f22583e834dcabdc7832e56fe16275253ee53ba66627b86e304da1"},
|
||||
{file = "debugpy-1.8.11-cp311-cp311-macosx_14_0_universal2.whl", hash = "sha256:85de8474ad53ad546ff1c7c7c89230db215b9b8a02754d41cb5a76f70d0be296"},
|
||||
{file = "debugpy-1.8.11-cp311-cp311-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:8ffc382e4afa4aee367bf413f55ed17bd91b191dcaf979890af239dda435f2a1"},
|
||||
{file = "debugpy-1.8.11-cp311-cp311-win32.whl", hash = "sha256:40499a9979c55f72f4eb2fc38695419546b62594f8af194b879d2a18439c97a9"},
|
||||
{file = "debugpy-1.8.11-cp311-cp311-win_amd64.whl", hash = "sha256:987bce16e86efa86f747d5151c54e91b3c1e36acc03ce1ddb50f9d09d16ded0e"},
|
||||
{file = "debugpy-1.8.11-cp312-cp312-macosx_14_0_universal2.whl", hash = "sha256:84e511a7545d11683d32cdb8f809ef63fc17ea2a00455cc62d0a4dbb4ed1c308"},
|
||||
{file = "debugpy-1.8.11-cp312-cp312-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:ce291a5aca4985d82875d6779f61375e959208cdf09fcec40001e65fb0a54768"},
|
||||
{file = "debugpy-1.8.11-cp312-cp312-win32.whl", hash = "sha256:28e45b3f827d3bf2592f3cf7ae63282e859f3259db44ed2b129093ca0ac7940b"},
|
||||
{file = "debugpy-1.8.11-cp312-cp312-win_amd64.whl", hash = "sha256:44b1b8e6253bceada11f714acf4309ffb98bfa9ac55e4fce14f9e5d4484287a1"},
|
||||
{file = "debugpy-1.8.11-cp313-cp313-macosx_14_0_universal2.whl", hash = "sha256:8988f7163e4381b0da7696f37eec7aca19deb02e500245df68a7159739bbd0d3"},
|
||||
{file = "debugpy-1.8.11-cp313-cp313-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:6c1f6a173d1140e557347419767d2b14ac1c9cd847e0b4c5444c7f3144697e4e"},
|
||||
{file = "debugpy-1.8.11-cp313-cp313-win32.whl", hash = "sha256:bb3b15e25891f38da3ca0740271e63ab9db61f41d4d8541745cfc1824252cb28"},
|
||||
{file = "debugpy-1.8.11-cp313-cp313-win_amd64.whl", hash = "sha256:d8768edcbeb34da9e11bcb8b5c2e0958d25218df7a6e56adf415ef262cd7b6d1"},
|
||||
{file = "debugpy-1.8.11-cp38-cp38-macosx_14_0_x86_64.whl", hash = "sha256:ad7efe588c8f5cf940f40c3de0cd683cc5b76819446abaa50dc0829a30c094db"},
|
||||
{file = "debugpy-1.8.11-cp38-cp38-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:189058d03a40103a57144752652b3ab08ff02b7595d0ce1f651b9acc3a3a35a0"},
|
||||
{file = "debugpy-1.8.11-cp38-cp38-win32.whl", hash = "sha256:32db46ba45849daed7ccf3f2e26f7a386867b077f39b2a974bb5c4c2c3b0a280"},
|
||||
{file = "debugpy-1.8.11-cp38-cp38-win_amd64.whl", hash = "sha256:116bf8342062246ca749013df4f6ea106f23bc159305843491f64672a55af2e5"},
|
||||
{file = "debugpy-1.8.11-cp39-cp39-macosx_14_0_x86_64.whl", hash = "sha256:654130ca6ad5de73d978057eaf9e582244ff72d4574b3e106fb8d3d2a0d32458"},
|
||||
{file = "debugpy-1.8.11-cp39-cp39-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:23dc34c5e03b0212fa3c49a874df2b8b1b8fda95160bd79c01eb3ab51ea8d851"},
|
||||
{file = "debugpy-1.8.11-cp39-cp39-win32.whl", hash = "sha256:52d8a3166c9f2815bfae05f386114b0b2d274456980d41f320299a8d9a5615a7"},
|
||||
{file = "debugpy-1.8.11-cp39-cp39-win_amd64.whl", hash = "sha256:52c3cf9ecda273a19cc092961ee34eb9ba8687d67ba34cc7b79a521c1c64c4c0"},
|
||||
{file = "debugpy-1.8.11-py2.py3-none-any.whl", hash = "sha256:0e22f846f4211383e6a416d04b4c13ed174d24cc5d43f5fd52e7821d0ebc8920"},
|
||||
{file = "debugpy-1.8.11.tar.gz", hash = "sha256:6ad2688b69235c43b020e04fecccdf6a96c8943ca9c2fb340b8adc103c655e57"},
|
||||
{file = "debugpy-1.8.12-cp310-cp310-macosx_14_0_x86_64.whl", hash = "sha256:a2ba7ffe58efeae5b8fad1165357edfe01464f9aef25e814e891ec690e7dd82a"},
|
||||
{file = "debugpy-1.8.12-cp310-cp310-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:cbbd4149c4fc5e7d508ece083e78c17442ee13b0e69bfa6bd63003e486770f45"},
|
||||
{file = "debugpy-1.8.12-cp310-cp310-win32.whl", hash = "sha256:b202f591204023b3ce62ff9a47baa555dc00bb092219abf5caf0e3718ac20e7c"},
|
||||
{file = "debugpy-1.8.12-cp310-cp310-win_amd64.whl", hash = "sha256:9649eced17a98ce816756ce50433b2dd85dfa7bc92ceb60579d68c053f98dff9"},
|
||||
{file = "debugpy-1.8.12-cp311-cp311-macosx_14_0_universal2.whl", hash = "sha256:36f4829839ef0afdfdd208bb54f4c3d0eea86106d719811681a8627ae2e53dd5"},
|
||||
{file = "debugpy-1.8.12-cp311-cp311-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:a28ed481d530e3138553be60991d2d61103ce6da254e51547b79549675f539b7"},
|
||||
{file = "debugpy-1.8.12-cp311-cp311-win32.whl", hash = "sha256:4ad9a94d8f5c9b954e0e3b137cc64ef3f579d0df3c3698fe9c3734ee397e4abb"},
|
||||
{file = "debugpy-1.8.12-cp311-cp311-win_amd64.whl", hash = "sha256:4703575b78dd697b294f8c65588dc86874ed787b7348c65da70cfc885efdf1e1"},
|
||||
{file = "debugpy-1.8.12-cp312-cp312-macosx_14_0_universal2.whl", hash = "sha256:7e94b643b19e8feb5215fa508aee531387494bf668b2eca27fa769ea11d9f498"},
|
||||
{file = "debugpy-1.8.12-cp312-cp312-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:086b32e233e89a2740c1615c2f775c34ae951508b28b308681dbbb87bba97d06"},
|
||||
{file = "debugpy-1.8.12-cp312-cp312-win32.whl", hash = "sha256:2ae5df899732a6051b49ea2632a9ea67f929604fd2b036613a9f12bc3163b92d"},
|
||||
{file = "debugpy-1.8.12-cp312-cp312-win_amd64.whl", hash = "sha256:39dfbb6fa09f12fae32639e3286112fc35ae976114f1f3d37375f3130a820969"},
|
||||
{file = "debugpy-1.8.12-cp313-cp313-macosx_14_0_universal2.whl", hash = "sha256:696d8ae4dff4cbd06bf6b10d671e088b66669f110c7c4e18a44c43cf75ce966f"},
|
||||
{file = "debugpy-1.8.12-cp313-cp313-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:898fba72b81a654e74412a67c7e0a81e89723cfe2a3ea6fcd3feaa3395138ca9"},
|
||||
{file = "debugpy-1.8.12-cp313-cp313-win32.whl", hash = "sha256:22a11c493c70413a01ed03f01c3c3a2fc4478fc6ee186e340487b2edcd6f4180"},
|
||||
{file = "debugpy-1.8.12-cp313-cp313-win_amd64.whl", hash = "sha256:fdb3c6d342825ea10b90e43d7f20f01535a72b3a1997850c0c3cefa5c27a4a2c"},
|
||||
{file = "debugpy-1.8.12-cp38-cp38-macosx_14_0_x86_64.whl", hash = "sha256:b0232cd42506d0c94f9328aaf0d1d0785f90f87ae72d9759df7e5051be039738"},
|
||||
{file = "debugpy-1.8.12-cp38-cp38-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:9af40506a59450f1315168d47a970db1a65aaab5df3833ac389d2899a5d63b3f"},
|
||||
{file = "debugpy-1.8.12-cp38-cp38-win32.whl", hash = "sha256:5cc45235fefac57f52680902b7d197fb2f3650112379a6fa9aa1b1c1d3ed3f02"},
|
||||
{file = "debugpy-1.8.12-cp38-cp38-win_amd64.whl", hash = "sha256:557cc55b51ab2f3371e238804ffc8510b6ef087673303890f57a24195d096e61"},
|
||||
{file = "debugpy-1.8.12-cp39-cp39-macosx_14_0_x86_64.whl", hash = "sha256:b5c6c967d02fee30e157ab5227706f965d5c37679c687b1e7bbc5d9e7128bd41"},
|
||||
{file = "debugpy-1.8.12-cp39-cp39-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:88a77f422f31f170c4b7e9ca58eae2a6c8e04da54121900651dfa8e66c29901a"},
|
||||
{file = "debugpy-1.8.12-cp39-cp39-win32.whl", hash = "sha256:a4042edef80364239f5b7b5764e55fd3ffd40c32cf6753da9bda4ff0ac466018"},
|
||||
{file = "debugpy-1.8.12-cp39-cp39-win_amd64.whl", hash = "sha256:f30b03b0f27608a0b26c75f0bb8a880c752c0e0b01090551b9d87c7d783e2069"},
|
||||
{file = "debugpy-1.8.12-py2.py3-none-any.whl", hash = "sha256:274b6a2040349b5c9864e475284bce5bb062e63dce368a394b8cc865ae3b00c6"},
|
||||
{file = "debugpy-1.8.12.tar.gz", hash = "sha256:646530b04f45c830ceae8e491ca1c9320a2d2f0efea3141487c82130aba70dce"},
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -638,13 +712,13 @@ files = [
|
||||
|
||||
[[package]]
|
||||
name = "executing"
|
||||
version = "2.1.0"
|
||||
version = "2.2.0"
|
||||
description = "Get the currently executing AST node of a frame, and other information"
|
||||
optional = false
|
||||
python-versions = ">=3.8"
|
||||
files = [
|
||||
{file = "executing-2.1.0-py2.py3-none-any.whl", hash = "sha256:8d63781349375b5ebccc3142f4b30350c0cd9c79f921cde38be2be4637e98eaf"},
|
||||
{file = "executing-2.1.0.tar.gz", hash = "sha256:8ea27ddd260da8150fa5a708269c4a10e76161e2496ec3e587da9e3c0fe4b9ab"},
|
||||
{file = "executing-2.2.0-py2.py3-none-any.whl", hash = "sha256:11387150cad388d62750327a53d3339fad4888b39a6fe233c3afbb54ecffd3aa"},
|
||||
{file = "executing-2.2.0.tar.gz", hash = "sha256:5d108c028108fe2551d1a7b2e8b713341e2cb4fc0aa7dcf966fa4327a5226755"},
|
||||
]
|
||||
|
||||
[package.extras]
|
||||
@@ -666,18 +740,18 @@ devel = ["colorama", "json-spec", "jsonschema", "pylint", "pytest", "pytest-benc
|
||||
|
||||
[[package]]
|
||||
name = "filelock"
|
||||
version = "3.16.1"
|
||||
version = "3.17.0"
|
||||
description = "A platform independent file lock."
|
||||
optional = false
|
||||
python-versions = ">=3.8"
|
||||
python-versions = ">=3.9"
|
||||
files = [
|
||||
{file = "filelock-3.16.1-py3-none-any.whl", hash = "sha256:2082e5703d51fbf98ea75855d9d5527e33d8ff23099bec374a134febee6946b0"},
|
||||
{file = "filelock-3.16.1.tar.gz", hash = "sha256:c249fbfcd5db47e5e2d6d62198e565475ee65e4831e2561c8e313fa7eb961435"},
|
||||
{file = "filelock-3.17.0-py3-none-any.whl", hash = "sha256:533dc2f7ba78dc2f0f531fc6c4940addf7b70a481e269a5a3b93be94ffbe8338"},
|
||||
{file = "filelock-3.17.0.tar.gz", hash = "sha256:ee4e77401ef576ebb38cd7f13b9b28893194acc20a8e68e18730ba9c0e54660e"},
|
||||
]
|
||||
|
||||
[package.extras]
|
||||
docs = ["furo (>=2024.8.6)", "sphinx (>=8.0.2)", "sphinx-autodoc-typehints (>=2.4.1)"]
|
||||
testing = ["covdefaults (>=2.3)", "coverage (>=7.6.1)", "diff-cover (>=9.2)", "pytest (>=8.3.3)", "pytest-asyncio (>=0.24)", "pytest-cov (>=5)", "pytest-mock (>=3.14)", "pytest-timeout (>=2.3.1)", "virtualenv (>=20.26.4)"]
|
||||
docs = ["furo (>=2024.8.6)", "sphinx (>=8.1.3)", "sphinx-autodoc-typehints (>=3)"]
|
||||
testing = ["covdefaults (>=2.3)", "coverage (>=7.6.10)", "diff-cover (>=9.2.1)", "pytest (>=8.3.4)", "pytest-asyncio (>=0.25.2)", "pytest-cov (>=6)", "pytest-mock (>=3.14)", "pytest-timeout (>=2.3.1)", "virtualenv (>=20.28.1)"]
|
||||
typing = ["typing-extensions (>=4.12.2)"]
|
||||
|
||||
[[package]]
|
||||
@@ -1718,66 +1792,66 @@ test = ["pytest", "pytest-console-scripts", "pytest-jupyter", "pytest-tornasync"
|
||||
|
||||
[[package]]
|
||||
name = "numpy"
|
||||
version = "2.2.1"
|
||||
version = "2.2.2"
|
||||
description = "Fundamental package for array computing in Python"
|
||||
optional = false
|
||||
python-versions = ">=3.10"
|
||||
files = [
|
||||
{file = "numpy-2.2.1-cp310-cp310-macosx_10_9_x86_64.whl", hash = "sha256:5edb4e4caf751c1518e6a26a83501fda79bff41cc59dac48d70e6d65d4ec4440"},
|
||||
{file = "numpy-2.2.1-cp310-cp310-macosx_11_0_arm64.whl", hash = "sha256:aa3017c40d513ccac9621a2364f939d39e550c542eb2a894b4c8da92b38896ab"},
|
||||
{file = "numpy-2.2.1-cp310-cp310-macosx_14_0_arm64.whl", hash = "sha256:61048b4a49b1c93fe13426e04e04fdf5a03f456616f6e98c7576144677598675"},
|
||||
{file = "numpy-2.2.1-cp310-cp310-macosx_14_0_x86_64.whl", hash = "sha256:7671dc19c7019103ca44e8d94917eba8534c76133523ca8406822efdd19c9308"},
|
||||
{file = "numpy-2.2.1-cp310-cp310-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:4250888bcb96617e00bfa28ac24850a83c9f3a16db471eca2ee1f1714df0f957"},
|
||||
{file = "numpy-2.2.1-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:a7746f235c47abc72b102d3bce9977714c2444bdfaea7888d241b4c4bb6a78bf"},
|
||||
{file = "numpy-2.2.1-cp310-cp310-musllinux_1_2_aarch64.whl", hash = "sha256:059e6a747ae84fce488c3ee397cee7e5f905fd1bda5fb18c66bc41807ff119b2"},
|
||||
{file = "numpy-2.2.1-cp310-cp310-musllinux_1_2_x86_64.whl", hash = "sha256:f62aa6ee4eb43b024b0e5a01cf65a0bb078ef8c395e8713c6e8a12a697144528"},
|
||||
{file = "numpy-2.2.1-cp310-cp310-win32.whl", hash = "sha256:48fd472630715e1c1c89bf1feab55c29098cb403cc184b4859f9c86d4fcb6a95"},
|
||||
{file = "numpy-2.2.1-cp310-cp310-win_amd64.whl", hash = "sha256:b541032178a718c165a49638d28272b771053f628382d5e9d1c93df23ff58dbf"},
|
||||
{file = "numpy-2.2.1-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:40f9e544c1c56ba8f1cf7686a8c9b5bb249e665d40d626a23899ba6d5d9e1484"},
|
||||
{file = "numpy-2.2.1-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:f9b57eaa3b0cd8db52049ed0330747b0364e899e8a606a624813452b8203d5f7"},
|
||||
{file = "numpy-2.2.1-cp311-cp311-macosx_14_0_arm64.whl", hash = "sha256:bc8a37ad5b22c08e2dbd27df2b3ef7e5c0864235805b1e718a235bcb200cf1cb"},
|
||||
{file = "numpy-2.2.1-cp311-cp311-macosx_14_0_x86_64.whl", hash = "sha256:9036d6365d13b6cbe8f27a0eaf73ddcc070cae584e5ff94bb45e3e9d729feab5"},
|
||||
{file = "numpy-2.2.1-cp311-cp311-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:51faf345324db860b515d3f364eaa93d0e0551a88d6218a7d61286554d190d73"},
|
||||
{file = "numpy-2.2.1-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:38efc1e56b73cc9b182fe55e56e63b044dd26a72128fd2fbd502f75555d92591"},
|
||||
{file = "numpy-2.2.1-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:31b89fa67a8042e96715c68e071a1200c4e172f93b0fbe01a14c0ff3ff820fc8"},
|
||||
{file = "numpy-2.2.1-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:4c86e2a209199ead7ee0af65e1d9992d1dce7e1f63c4b9a616500f93820658d0"},
|
||||
{file = "numpy-2.2.1-cp311-cp311-win32.whl", hash = "sha256:b34d87e8a3090ea626003f87f9392b3929a7bbf4104a05b6667348b6bd4bf1cd"},
|
||||
{file = "numpy-2.2.1-cp311-cp311-win_amd64.whl", hash = "sha256:360137f8fb1b753c5cde3ac388597ad680eccbbbb3865ab65efea062c4a1fd16"},
|
||||
{file = "numpy-2.2.1-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:694f9e921a0c8f252980e85bce61ebbd07ed2b7d4fa72d0e4246f2f8aa6642ab"},
|
||||
{file = "numpy-2.2.1-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:3683a8d166f2692664262fd4900f207791d005fb088d7fdb973cc8d663626faa"},
|
||||
{file = "numpy-2.2.1-cp312-cp312-macosx_14_0_arm64.whl", hash = "sha256:780077d95eafc2ccc3ced969db22377b3864e5b9a0ea5eb347cc93b3ea900315"},
|
||||
{file = "numpy-2.2.1-cp312-cp312-macosx_14_0_x86_64.whl", hash = "sha256:55ba24ebe208344aa7a00e4482f65742969a039c2acfcb910bc6fcd776eb4355"},
|
||||
{file = "numpy-2.2.1-cp312-cp312-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:9b1d07b53b78bf84a96898c1bc139ad7f10fda7423f5fd158fd0f47ec5e01ac7"},
|
||||
{file = "numpy-2.2.1-cp312-cp312-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:5062dc1a4e32a10dc2b8b13cedd58988261416e811c1dc4dbdea4f57eea61b0d"},
|
||||
{file = "numpy-2.2.1-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:fce4f615f8ca31b2e61aa0eb5865a21e14f5629515c9151850aa936c02a1ee51"},
|
||||
{file = "numpy-2.2.1-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:67d4cda6fa6ffa073b08c8372aa5fa767ceb10c9a0587c707505a6d426f4e046"},
|
||||
{file = "numpy-2.2.1-cp312-cp312-win32.whl", hash = "sha256:32cb94448be47c500d2c7a95f93e2f21a01f1fd05dd2beea1ccd049bb6001cd2"},
|
||||
{file = "numpy-2.2.1-cp312-cp312-win_amd64.whl", hash = "sha256:ba5511d8f31c033a5fcbda22dd5c813630af98c70b2661f2d2c654ae3cdfcfc8"},
|
||||
{file = "numpy-2.2.1-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:f1d09e520217618e76396377c81fba6f290d5f926f50c35f3a5f72b01a0da780"},
|
||||
{file = "numpy-2.2.1-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:3ecc47cd7f6ea0336042be87d9e7da378e5c7e9b3c8ad0f7c966f714fc10d821"},
|
||||
{file = "numpy-2.2.1-cp313-cp313-macosx_14_0_arm64.whl", hash = "sha256:f419290bc8968a46c4933158c91a0012b7a99bb2e465d5ef5293879742f8797e"},
|
||||
{file = "numpy-2.2.1-cp313-cp313-macosx_14_0_x86_64.whl", hash = "sha256:5b6c390bfaef8c45a260554888966618328d30e72173697e5cabe6b285fb2348"},
|
||||
{file = "numpy-2.2.1-cp313-cp313-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:526fc406ab991a340744aad7e25251dd47a6720a685fa3331e5c59fef5282a59"},
|
||||
{file = "numpy-2.2.1-cp313-cp313-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:f74e6fdeb9a265624ec3a3918430205dff1df7e95a230779746a6af78bc615af"},
|
||||
{file = "numpy-2.2.1-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:53c09385ff0b72ba79d8715683c1168c12e0b6e84fb0372e97553d1ea91efe51"},
|
||||
{file = "numpy-2.2.1-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:f3eac17d9ec51be534685ba877b6ab5edc3ab7ec95c8f163e5d7b39859524716"},
|
||||
{file = "numpy-2.2.1-cp313-cp313-win32.whl", hash = "sha256:9ad014faa93dbb52c80d8f4d3dcf855865c876c9660cb9bd7553843dd03a4b1e"},
|
||||
{file = "numpy-2.2.1-cp313-cp313-win_amd64.whl", hash = "sha256:164a829b6aacf79ca47ba4814b130c4020b202522a93d7bff2202bfb33b61c60"},
|
||||
{file = "numpy-2.2.1-cp313-cp313t-macosx_10_13_x86_64.whl", hash = "sha256:4dfda918a13cc4f81e9118dea249e192ab167a0bb1966272d5503e39234d694e"},
|
||||
{file = "numpy-2.2.1-cp313-cp313t-macosx_11_0_arm64.whl", hash = "sha256:733585f9f4b62e9b3528dd1070ec4f52b8acf64215b60a845fa13ebd73cd0712"},
|
||||
{file = "numpy-2.2.1-cp313-cp313t-macosx_14_0_arm64.whl", hash = "sha256:89b16a18e7bba224ce5114db863e7029803c179979e1af6ad6a6b11f70545008"},
|
||||
{file = "numpy-2.2.1-cp313-cp313t-macosx_14_0_x86_64.whl", hash = "sha256:676f4eebf6b2d430300f1f4f4c2461685f8269f94c89698d832cdf9277f30b84"},
|
||||
{file = "numpy-2.2.1-cp313-cp313t-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:27f5cdf9f493b35f7e41e8368e7d7b4bbafaf9660cba53fb21d2cd174ec09631"},
|
||||
{file = "numpy-2.2.1-cp313-cp313t-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:c1ad395cf254c4fbb5b2132fee391f361a6e8c1adbd28f2cd8e79308a615fe9d"},
|
||||
{file = "numpy-2.2.1-cp313-cp313t-musllinux_1_2_aarch64.whl", hash = "sha256:08ef779aed40dbc52729d6ffe7dd51df85796a702afbf68a4f4e41fafdc8bda5"},
|
||||
{file = "numpy-2.2.1-cp313-cp313t-musllinux_1_2_x86_64.whl", hash = "sha256:26c9c4382b19fcfbbed3238a14abf7ff223890ea1936b8890f058e7ba35e8d71"},
|
||||
{file = "numpy-2.2.1-cp313-cp313t-win32.whl", hash = "sha256:93cf4e045bae74c90ca833cba583c14b62cb4ba2cba0abd2b141ab52548247e2"},
|
||||
{file = "numpy-2.2.1-cp313-cp313t-win_amd64.whl", hash = "sha256:bff7d8ec20f5f42607599f9994770fa65d76edca264a87b5e4ea5629bce12268"},
|
||||
{file = "numpy-2.2.1-pp310-pypy310_pp73-macosx_10_15_x86_64.whl", hash = "sha256:7ba9cc93a91d86365a5d270dee221fdc04fb68d7478e6bf6af650de78a8339e3"},
|
||||
{file = "numpy-2.2.1-pp310-pypy310_pp73-macosx_14_0_x86_64.whl", hash = "sha256:3d03883435a19794e41f147612a77a8f56d4e52822337844fff3d4040a142964"},
|
||||
{file = "numpy-2.2.1-pp310-pypy310_pp73-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:4511d9e6071452b944207c8ce46ad2f897307910b402ea5fa975da32e0102800"},
|
||||
{file = "numpy-2.2.1-pp310-pypy310_pp73-win_amd64.whl", hash = "sha256:5c5cc0cbabe9452038ed984d05ac87910f89370b9242371bd9079cb4af61811e"},
|
||||
{file = "numpy-2.2.1.tar.gz", hash = "sha256:45681fd7128c8ad1c379f0ca0776a8b0c6583d2f69889ddac01559dfe4390918"},
|
||||
{file = "numpy-2.2.2-cp310-cp310-macosx_10_9_x86_64.whl", hash = "sha256:7079129b64cb78bdc8d611d1fd7e8002c0a2565da6a47c4df8062349fee90e3e"},
|
||||
{file = "numpy-2.2.2-cp310-cp310-macosx_11_0_arm64.whl", hash = "sha256:2ec6c689c61df613b783aeb21f945c4cbe6c51c28cb70aae8430577ab39f163e"},
|
||||
{file = "numpy-2.2.2-cp310-cp310-macosx_14_0_arm64.whl", hash = "sha256:40c7ff5da22cd391944a28c6a9c638a5eef77fcf71d6e3a79e1d9d9e82752715"},
|
||||
{file = "numpy-2.2.2-cp310-cp310-macosx_14_0_x86_64.whl", hash = "sha256:995f9e8181723852ca458e22de5d9b7d3ba4da3f11cc1cb113f093b271d7965a"},
|
||||
{file = "numpy-2.2.2-cp310-cp310-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:b78ea78450fd96a498f50ee096f69c75379af5138f7881a51355ab0e11286c97"},
|
||||
{file = "numpy-2.2.2-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:3fbe72d347fbc59f94124125e73fc4976a06927ebc503ec5afbfb35f193cd957"},
|
||||
{file = "numpy-2.2.2-cp310-cp310-musllinux_1_2_aarch64.whl", hash = "sha256:8e6da5cffbbe571f93588f562ed130ea63ee206d12851b60819512dd3e1ba50d"},
|
||||
{file = "numpy-2.2.2-cp310-cp310-musllinux_1_2_x86_64.whl", hash = "sha256:09d6a2032faf25e8d0cadde7fd6145118ac55d2740132c1d845f98721b5ebcfd"},
|
||||
{file = "numpy-2.2.2-cp310-cp310-win32.whl", hash = "sha256:159ff6ee4c4a36a23fe01b7c3d07bd8c14cc433d9720f977fcd52c13c0098160"},
|
||||
{file = "numpy-2.2.2-cp310-cp310-win_amd64.whl", hash = "sha256:64bd6e1762cd7f0986a740fee4dff927b9ec2c5e4d9a28d056eb17d332158014"},
|
||||
{file = "numpy-2.2.2-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:642199e98af1bd2b6aeb8ecf726972d238c9877b0f6e8221ee5ab945ec8a2189"},
|
||||
{file = "numpy-2.2.2-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:6d9fc9d812c81e6168b6d405bf00b8d6739a7f72ef22a9214c4241e0dc70b323"},
|
||||
{file = "numpy-2.2.2-cp311-cp311-macosx_14_0_arm64.whl", hash = "sha256:c7d1fd447e33ee20c1f33f2c8e6634211124a9aabde3c617687d8b739aa69eac"},
|
||||
{file = "numpy-2.2.2-cp311-cp311-macosx_14_0_x86_64.whl", hash = "sha256:451e854cfae0febe723077bd0cf0a4302a5d84ff25f0bfece8f29206c7bed02e"},
|
||||
{file = "numpy-2.2.2-cp311-cp311-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:bd249bc894af67cbd8bad2c22e7cbcd46cf87ddfca1f1289d1e7e54868cc785c"},
|
||||
{file = "numpy-2.2.2-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:02935e2c3c0c6cbe9c7955a8efa8908dd4221d7755644c59d1bba28b94fd334f"},
|
||||
{file = "numpy-2.2.2-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:a972cec723e0563aa0823ee2ab1df0cb196ed0778f173b381c871a03719d4826"},
|
||||
{file = "numpy-2.2.2-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:d6d6a0910c3b4368d89dde073e630882cdb266755565155bc33520283b2d9df8"},
|
||||
{file = "numpy-2.2.2-cp311-cp311-win32.whl", hash = "sha256:860fd59990c37c3ef913c3ae390b3929d005243acca1a86facb0773e2d8d9e50"},
|
||||
{file = "numpy-2.2.2-cp311-cp311-win_amd64.whl", hash = "sha256:da1eeb460ecce8d5b8608826595c777728cdf28ce7b5a5a8c8ac8d949beadcf2"},
|
||||
{file = "numpy-2.2.2-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:ac9bea18d6d58a995fac1b2cb4488e17eceeac413af014b1dd26170b766d8467"},
|
||||
{file = "numpy-2.2.2-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:23ae9f0c2d889b7b2d88a3791f6c09e2ef827c2446f1c4a3e3e76328ee4afd9a"},
|
||||
{file = "numpy-2.2.2-cp312-cp312-macosx_14_0_arm64.whl", hash = "sha256:3074634ea4d6df66be04f6728ee1d173cfded75d002c75fac79503a880bf3825"},
|
||||
{file = "numpy-2.2.2-cp312-cp312-macosx_14_0_x86_64.whl", hash = "sha256:8ec0636d3f7d68520afc6ac2dc4b8341ddb725039de042faf0e311599f54eb37"},
|
||||
{file = "numpy-2.2.2-cp312-cp312-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:2ffbb1acd69fdf8e89dd60ef6182ca90a743620957afb7066385a7bbe88dc748"},
|
||||
{file = "numpy-2.2.2-cp312-cp312-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:0349b025e15ea9d05c3d63f9657707a4e1d471128a3b1d876c095f328f8ff7f0"},
|
||||
{file = "numpy-2.2.2-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:463247edcee4a5537841d5350bc87fe8e92d7dd0e8c71c995d2c6eecb8208278"},
|
||||
{file = "numpy-2.2.2-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:9dd47ff0cb2a656ad69c38da850df3454da88ee9a6fde0ba79acceee0e79daba"},
|
||||
{file = "numpy-2.2.2-cp312-cp312-win32.whl", hash = "sha256:4525b88c11906d5ab1b0ec1f290996c0020dd318af8b49acaa46f198b1ffc283"},
|
||||
{file = "numpy-2.2.2-cp312-cp312-win_amd64.whl", hash = "sha256:5acea83b801e98541619af398cc0109ff48016955cc0818f478ee9ef1c5c3dcb"},
|
||||
{file = "numpy-2.2.2-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:b208cfd4f5fe34e1535c08983a1a6803fdbc7a1e86cf13dd0c61de0b51a0aadc"},
|
||||
{file = "numpy-2.2.2-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:d0bbe7dd86dca64854f4b6ce2ea5c60b51e36dfd597300057cf473d3615f2369"},
|
||||
{file = "numpy-2.2.2-cp313-cp313-macosx_14_0_arm64.whl", hash = "sha256:22ea3bb552ade325530e72a0c557cdf2dea8914d3a5e1fecf58fa5dbcc6f43cd"},
|
||||
{file = "numpy-2.2.2-cp313-cp313-macosx_14_0_x86_64.whl", hash = "sha256:128c41c085cab8a85dc29e66ed88c05613dccf6bc28b3866cd16050a2f5448be"},
|
||||
{file = "numpy-2.2.2-cp313-cp313-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:250c16b277e3b809ac20d1f590716597481061b514223c7badb7a0f9993c7f84"},
|
||||
{file = "numpy-2.2.2-cp313-cp313-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:e0c8854b09bc4de7b041148d8550d3bd712b5c21ff6a8ed308085f190235d7ff"},
|
||||
{file = "numpy-2.2.2-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:b6fb9c32a91ec32a689ec6410def76443e3c750e7cfc3fb2206b985ffb2b85f0"},
|
||||
{file = "numpy-2.2.2-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:57b4012e04cc12b78590a334907e01b3a85efb2107df2b8733ff1ed05fce71de"},
|
||||
{file = "numpy-2.2.2-cp313-cp313-win32.whl", hash = "sha256:4dbd80e453bd34bd003b16bd802fac70ad76bd463f81f0c518d1245b1c55e3d9"},
|
||||
{file = "numpy-2.2.2-cp313-cp313-win_amd64.whl", hash = "sha256:5a8c863ceacae696aff37d1fd636121f1a512117652e5dfb86031c8d84836369"},
|
||||
{file = "numpy-2.2.2-cp313-cp313t-macosx_10_13_x86_64.whl", hash = "sha256:b3482cb7b3325faa5f6bc179649406058253d91ceda359c104dac0ad320e1391"},
|
||||
{file = "numpy-2.2.2-cp313-cp313t-macosx_11_0_arm64.whl", hash = "sha256:9491100aba630910489c1d0158034e1c9a6546f0b1340f716d522dc103788e39"},
|
||||
{file = "numpy-2.2.2-cp313-cp313t-macosx_14_0_arm64.whl", hash = "sha256:41184c416143defa34cc8eb9d070b0a5ba4f13a0fa96a709e20584638254b317"},
|
||||
{file = "numpy-2.2.2-cp313-cp313t-macosx_14_0_x86_64.whl", hash = "sha256:7dca87ca328f5ea7dafc907c5ec100d187911f94825f8700caac0b3f4c384b49"},
|
||||
{file = "numpy-2.2.2-cp313-cp313t-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:0bc61b307655d1a7f9f4b043628b9f2b721e80839914ede634e3d485913e1fb2"},
|
||||
{file = "numpy-2.2.2-cp313-cp313t-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:9fad446ad0bc886855ddf5909cbf8cb5d0faa637aaa6277fb4b19ade134ab3c7"},
|
||||
{file = "numpy-2.2.2-cp313-cp313t-musllinux_1_2_aarch64.whl", hash = "sha256:149d1113ac15005652e8d0d3f6fd599360e1a708a4f98e43c9c77834a28238cb"},
|
||||
{file = "numpy-2.2.2-cp313-cp313t-musllinux_1_2_x86_64.whl", hash = "sha256:106397dbbb1896f99e044efc90360d098b3335060375c26aa89c0d8a97c5f648"},
|
||||
{file = "numpy-2.2.2-cp313-cp313t-win32.whl", hash = "sha256:0eec19f8af947a61e968d5429f0bd92fec46d92b0008d0a6685b40d6adf8a4f4"},
|
||||
{file = "numpy-2.2.2-cp313-cp313t-win_amd64.whl", hash = "sha256:97b974d3ba0fb4612b77ed35d7627490e8e3dff56ab41454d9e8b23448940576"},
|
||||
{file = "numpy-2.2.2-pp310-pypy310_pp73-macosx_10_15_x86_64.whl", hash = "sha256:b0531f0b0e07643eb089df4c509d30d72c9ef40defa53e41363eca8a8cc61495"},
|
||||
{file = "numpy-2.2.2-pp310-pypy310_pp73-macosx_14_0_x86_64.whl", hash = "sha256:e9e82dcb3f2ebbc8cb5ce1102d5f1c5ed236bf8a11730fb45ba82e2841ec21df"},
|
||||
{file = "numpy-2.2.2-pp310-pypy310_pp73-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:e0d4142eb40ca6f94539e4db929410f2a46052a0fe7a2c1c59f6179c39938d2a"},
|
||||
{file = "numpy-2.2.2-pp310-pypy310_pp73-win_amd64.whl", hash = "sha256:356ca982c188acbfa6af0d694284d8cf20e95b1c3d0aefa8929376fea9146f60"},
|
||||
{file = "numpy-2.2.2.tar.gz", hash = "sha256:ed6906f61834d687738d25988ae117683705636936cc605be0bb208b23df4d8f"},
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -1996,13 +2070,13 @@ twisted = ["twisted"]
|
||||
|
||||
[[package]]
|
||||
name = "prompt-toolkit"
|
||||
version = "3.0.48"
|
||||
version = "3.0.50"
|
||||
description = "Library for building powerful interactive command lines in Python"
|
||||
optional = false
|
||||
python-versions = ">=3.7.0"
|
||||
python-versions = ">=3.8.0"
|
||||
files = [
|
||||
{file = "prompt_toolkit-3.0.48-py3-none-any.whl", hash = "sha256:f49a827f90062e411f1ce1f854f2aedb3c23353244f8108b89283587397ac10e"},
|
||||
{file = "prompt_toolkit-3.0.48.tar.gz", hash = "sha256:d6623ab0477a80df74e646bdbc93621143f5caf104206aa29294d53de1a03d90"},
|
||||
{file = "prompt_toolkit-3.0.50-py3-none-any.whl", hash = "sha256:9b6427eb19e479d98acff65196a307c555eb567989e6d88ebbb1b509d9779198"},
|
||||
{file = "prompt_toolkit-3.0.50.tar.gz", hash = "sha256:544748f3860a2623ca5cd6d2795e7a14f3d0e1c3c9728359013f79877fc89bab"},
|
||||
]
|
||||
|
||||
[package.dependencies]
|
||||
@@ -2240,6 +2314,24 @@ pluggy = ">=1.5,<2"
|
||||
[package.extras]
|
||||
dev = ["argcomplete", "attrs (>=19.2)", "hypothesis (>=3.56)", "mock", "pygments (>=2.7.2)", "requests", "setuptools", "xmlschema"]
|
||||
|
||||
[[package]]
|
||||
name = "pytest-cov"
|
||||
version = "6.0.0"
|
||||
description = "Pytest plugin for measuring coverage."
|
||||
optional = false
|
||||
python-versions = ">=3.9"
|
||||
files = [
|
||||
{file = "pytest-cov-6.0.0.tar.gz", hash = "sha256:fde0b595ca248bb8e2d76f020b465f3b107c9632e6a1d1705f17834c89dcadc0"},
|
||||
{file = "pytest_cov-6.0.0-py3-none-any.whl", hash = "sha256:eee6f1b9e61008bd34975a4d5bab25801eb31898b032dd55addc93e96fcaaa35"},
|
||||
]
|
||||
|
||||
[package.dependencies]
|
||||
coverage = {version = ">=7.5", extras = ["toml"]}
|
||||
pytest = ">=4.6"
|
||||
|
||||
[package.extras]
|
||||
testing = ["fields", "hunter", "process-tests", "pytest-xdist", "virtualenv"]
|
||||
|
||||
[[package]]
|
||||
name = "pytest-mock"
|
||||
version = "3.14.0"
|
||||
@@ -2648,18 +2740,19 @@ all = ["numpy"]
|
||||
|
||||
[[package]]
|
||||
name = "referencing"
|
||||
version = "0.35.1"
|
||||
version = "0.36.1"
|
||||
description = "JSON Referencing + Python"
|
||||
optional = false
|
||||
python-versions = ">=3.8"
|
||||
python-versions = ">=3.9"
|
||||
files = [
|
||||
{file = "referencing-0.35.1-py3-none-any.whl", hash = "sha256:eda6d3234d62814d1c64e305c1331c9a3a6132da475ab6382eaa997b21ee75de"},
|
||||
{file = "referencing-0.35.1.tar.gz", hash = "sha256:25b42124a6c8b632a425174f24087783efb348a6f1e0008e63cd4466fedf703c"},
|
||||
{file = "referencing-0.36.1-py3-none-any.whl", hash = "sha256:363d9c65f080d0d70bc41c721dce3c7f3e77fc09f269cd5c8813da18069a6794"},
|
||||
{file = "referencing-0.36.1.tar.gz", hash = "sha256:ca2e6492769e3602957e9b831b94211599d2aade9477f5d44110d2530cf9aade"},
|
||||
]
|
||||
|
||||
[package.dependencies]
|
||||
attrs = ">=22.2.0"
|
||||
rpds-py = ">=0.7.0"
|
||||
typing-extensions = {version = ">=4.4.0", markers = "python_version < \"3.13\""}
|
||||
|
||||
[[package]]
|
||||
name = "requests"
|
||||
@@ -2821,20 +2914,20 @@ files = [
|
||||
|
||||
[[package]]
|
||||
name = "s3transfer"
|
||||
version = "0.10.4"
|
||||
version = "0.11.1"
|
||||
description = "An Amazon S3 Transfer Manager"
|
||||
optional = false
|
||||
python-versions = ">=3.8"
|
||||
files = [
|
||||
{file = "s3transfer-0.10.4-py3-none-any.whl", hash = "sha256:244a76a24355363a68164241438de1b72f8781664920260c48465896b712a41e"},
|
||||
{file = "s3transfer-0.10.4.tar.gz", hash = "sha256:29edc09801743c21eb5ecbc617a152df41d3c287f67b615f73e5f750583666a7"},
|
||||
{file = "s3transfer-0.11.1-py3-none-any.whl", hash = "sha256:8fa0aa48177be1f3425176dfe1ab85dcd3d962df603c3dbfc585e6bf857ef0ff"},
|
||||
{file = "s3transfer-0.11.1.tar.gz", hash = "sha256:3f25c900a367c8b7f7d8f9c34edc87e300bde424f779dc9f0a8ae4f9df9264f6"},
|
||||
]
|
||||
|
||||
[package.dependencies]
|
||||
botocore = ">=1.33.2,<2.0a.0"
|
||||
botocore = ">=1.36.0,<2.0a.0"
|
||||
|
||||
[package.extras]
|
||||
crt = ["botocore[crt] (>=1.33.2,<2.0a.0)"]
|
||||
crt = ["botocore[crt] (>=1.36.0,<2.0a.0)"]
|
||||
|
||||
[[package]]
|
||||
name = "send2trash"
|
||||
@@ -3075,13 +3168,13 @@ files = [
|
||||
|
||||
[[package]]
|
||||
name = "tzdata"
|
||||
version = "2024.2"
|
||||
version = "2025.1"
|
||||
description = "Provider of IANA time zone data"
|
||||
optional = false
|
||||
python-versions = ">=2"
|
||||
files = [
|
||||
{file = "tzdata-2024.2-py2.py3-none-any.whl", hash = "sha256:a48093786cdcde33cad18c2555e8532f34422074448fbc874186f0abd79565cd"},
|
||||
{file = "tzdata-2024.2.tar.gz", hash = "sha256:7d85cc416e9382e69095b7bdf4afd9e3880418a2413feec7069d533d6b4e31cc"},
|
||||
{file = "tzdata-2025.1-py2.py3-none-any.whl", hash = "sha256:7e127113816800496f027041c570f50bcd464a020098a3b6b199517772303639"},
|
||||
{file = "tzdata-2025.1.tar.gz", hash = "sha256:24894909e88cdb28bd1636c6887801df64cb485bd593f2fd83ef29075a81d694"},
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -3178,4 +3271,4 @@ files = [
|
||||
[metadata]
|
||||
lock-version = "2.0"
|
||||
python-versions = "^3.12"
|
||||
content-hash = "977386812e414d6b13d5237092e9da24d33cb5400a34b6a241898379eed1fba9"
|
||||
content-hash = "f95476763e9ff04194b62402a3835367059c0af94bdb0b706eab6709d5a02848"
|
||||
|
||||
@@ -16,7 +16,6 @@ psutil = "^6.1.0"
|
||||
rapidfuzz = "^3.10.1"
|
||||
pyxlsb = "^1.0.10"
|
||||
openpyxl = "^3.1.5"
|
||||
pytest-mock = "^3.14.0"
|
||||
|
||||
[tool.poetry.group.dev.dependencies]
|
||||
black = "^24.10.0"
|
||||
@@ -27,6 +26,7 @@ isort = "^5.13.2"
|
||||
[tool.poetry.group.test.dependencies]
|
||||
pytest = "^8.3.3"
|
||||
pytest-mock = "^3.14.0"
|
||||
pytest-cov = "^6.0.0"
|
||||
|
||||
[build-system]
|
||||
requires = ["poetry-core"]
|
||||
|
||||
@@ -97,8 +97,8 @@ def run_b_prompts(file_object):
|
||||
################## PREPROCESS ##################
|
||||
contract_text = preprocess.clean_text(contract_text)
|
||||
text_dict, top_sheet_dict, num_pages = preprocess.split_text(contract_text)
|
||||
text_dict = preprocess.clean_tables(text_dict, filename) # TODO: Optimize later to not run a bunch of extra prompts for the extra table pages
|
||||
exhibit_pages, exhibit_chunk_mapping = preprocess.one_to_n_exhibit_chunking(text_dict, filename)
|
||||
text_dict = preprocess.clean_tables(text_dict, filename)
|
||||
print(f"B Preprocessing Complete - {filename}")
|
||||
|
||||
################## RUN BOTTOM UP PROMPTS ##################
|
||||
|
||||
@@ -1,7 +1,5 @@
|
||||
import csv
|
||||
|
||||
import src.utils.table_utils as table_utils
|
||||
from src import config, keywords, preprocessing_funcs
|
||||
from src import keywords, preprocessing_funcs
|
||||
|
||||
|
||||
def clean_text(contract_text):
|
||||
@@ -68,19 +66,30 @@ def one_to_n_exhibit_chunking(text_dict, filename):
|
||||
return exhibit_pages, exhibit_chunk_mapping
|
||||
|
||||
|
||||
def clean_tables(text_dict, filename):
|
||||
def clean_tables(text_dict, filename, row_limit=5):
|
||||
"""
|
||||
In current state, this function is a wrapper for align_and_format_tables. Call any new table-related preprocessing funcs here
|
||||
Cleans and processes tables within a given text dictionary.
|
||||
|
||||
Parameters:
|
||||
text_dict (dict): A dictionary where keys are page numbers and values are the text on those pages.
|
||||
filename (str): The name of the file being processed
|
||||
This function performs the following steps:
|
||||
1. Groups and merges tables from the text dictionary.
|
||||
2. Splits the grouped tables into pages based on the specified row limit.
|
||||
3. Reinserts the new tables back into the text dictionary.
|
||||
4. Aligns and formats the tables.
|
||||
5. Splits the tables into separate subpages.
|
||||
|
||||
Args:
|
||||
text_dict (dict): A dictionary containing text data with tables to be processed.
|
||||
filename (str): The name of the file being processed.
|
||||
row_limit (int, optional): The maximum number of rows per table page. Defaults to 5.
|
||||
|
||||
Returns:
|
||||
dict: text_dict, with the tables aligned and formatted
|
||||
dict: text_dict, with the tables aligned, formatted, and split into multiple subpages according to the row limit.
|
||||
"""
|
||||
text_dict = table_utils.align_and_format_tables(text_dict, filename)
|
||||
return text_dict
|
||||
groups, page_tables_map = table_utils.group_and_merge_tables(text_dict)
|
||||
group_table_split_info = table_utils.split_table_to_pages(groups, page_tables_map, row_limit=row_limit)
|
||||
modified_text_dict = table_utils.reinsert_new_tables_to_text(group_table_split_info, text_dict)
|
||||
text_dict = table_utils.align_and_format_tables(modified_text_dict, filename)
|
||||
return table_utils.split_tables_to_separate_subpages(text_dict)
|
||||
|
||||
|
||||
def one_to_one_smart_chunking(text_dict, contract_text, keyword_mappings=keywords.GROUPED_KEYWORD_MAPPINGS):
|
||||
|
||||
@@ -25,6 +25,7 @@ def remove_page_indicators(contract_text: str) -> str:
|
||||
return cleaned_text
|
||||
|
||||
|
||||
# TODO: write unit tests
|
||||
def split_text(text: str) -> dict[str, str]:
|
||||
"""Split text on pages by the string `Start of Page No. = '
|
||||
|
||||
@@ -34,19 +35,14 @@ def split_text(text: str) -> dict[str, str]:
|
||||
Returns:
|
||||
dict[str, str]: A dictionary, keyed by the string page number and valued by the page text.
|
||||
"""
|
||||
|
||||
if isinstance(text, str):
|
||||
if not text:
|
||||
return {}
|
||||
|
||||
temp_list = text.split("Start of Page No. = ")
|
||||
text_list = re.split(r"Start of Page No. = [0-9]+\n", text)
|
||||
temp_list = text.split("Start of Page No. = ")
|
||||
text_list = re.split(r"Start of Page No. = [0-9]+\n", text)
|
||||
|
||||
text_dict = {}
|
||||
for i in range(len(text_list)):
|
||||
text_dict[temp_list[i].split()[0]] = text_list[i] # splits on whitespace characters, which includes spaces, tabs, and newline characters
|
||||
text_dict = {}
|
||||
for i in range(len(text_list)):
|
||||
text_dict[temp_list[i].split()[0]] = text_list[i] # splits on whitespace characters, which includes spaces, tabs, and newline characters
|
||||
|
||||
return {k: v for k, v in text_dict.items() if k != "Document"}
|
||||
return {k: v for k, v in text_dict.items() if k != "Document"}
|
||||
|
||||
|
||||
def clean_law_symbols(contract_text):
|
||||
@@ -304,7 +300,7 @@ def get_exhibit_pages(text_dict, filename):
|
||||
for page_num, page in text_dict.items():
|
||||
prompt = preprocessing_prompts.EXHIBIT_CHECK(page[0:100])
|
||||
claude_answer_raw = llm_utils.invoke_claude(
|
||||
prompt, config.MODEL_ID_CLAUDE3_HAIKU, filename, max_tokens=10
|
||||
prompt, config.MODEL_ID_CLAUDE3_HAIKU, filename, max_tokens=10 # TODO: low priority, try increasing max_tokens and maybe pass multiple pages in to reduce overall calls
|
||||
)
|
||||
claude_answer_extracted = string_utils.extract_text_from_delimiters(
|
||||
claude_answer_raw, Delimiter.PIPE
|
||||
|
||||
@@ -6,7 +6,7 @@ Analyze the following contract excerpt and determine if it indicates the start o
|
||||
1. Look specifically for the words 'Exhibit', 'Article', 'Amendment', 'Schedule', 'Attachment', or 'Addendum'. These words indicate the start of a new section.
|
||||
2. If the above words are found in the context of a longer sentence or statement, this does not by itself indicate the start of a section. Section starts will be formatted like a title or header.
|
||||
3. The phrase 'Additional Provisions' does not indicate the start of a section.
|
||||
4. Return Y if the excerpt indicates the start of a section, N if not. Enclose your final answer in |pipes|
|
||||
4. Return Y if the excerpt indicates the start of a section, N if not. Please justify your answer, but enclose your final answer in |pipes|
|
||||
|
||||
Here is the text to analyze:
|
||||
|
||||
@@ -43,17 +43,20 @@ Alignment Instructions:
|
||||
Align the tables in the text using the Table Header text found after -------Table Start--------. If there is no header, align based on context clues in the text such as "the table below", etc.
|
||||
If there are multiple tables on the page, the first table should be placed in the first valid spot, the second in the second, etc.
|
||||
If the Table Header is None, place the updated table at the top of the page.
|
||||
Do not replicate the table header text twice. Do not remove the -------Table Start-------- or -------Table End-------- lines.
|
||||
|
||||
Formatting Instructions:
|
||||
Here is a template format to use:
|
||||
Reformat the table from json format to a human-readable fixed-width format. Use the following tempate format:
|
||||
|
||||
Original JSON: "-------Table Start-------- Table Header {{'Column 1' : ['Value 1', 'Value 2', 'Value 3'], 'Column 2' : ['Value 4', 'Value 5', 'Value 6']}}-------Table End--------"
|
||||
|
||||
Output Format:
|
||||
-------Table Start--------
|
||||
Table Header
|
||||
Column 1 : Value 1 | Column 2 : Value 4
|
||||
Column 1 : Value 2 | Column 2 : Value 5
|
||||
Column 1 : Value 3 | Column 2 : Value 6
|
||||
-------Table End--------
|
||||
|
||||
Ensure that all text other than the tables is exactly the same
|
||||
|
||||
|
||||
@@ -223,27 +223,64 @@ def primary_string_to_dict(string_dict, filename):
|
||||
data.append(dict_)
|
||||
return data
|
||||
|
||||
reimbursement_strings = [ # These are for `method='keyword'`
|
||||
"%",
|
||||
"$",
|
||||
"percent",
|
||||
"compensation schedule",
|
||||
"reimbursement schedule",
|
||||
]
|
||||
|
||||
def contains_reimbursement(text, page="1"): # string_funcs.py
|
||||
if isinstance(text, dict):
|
||||
return page.isdigit() and (
|
||||
"%" in text[page]
|
||||
or "$" in text[page]
|
||||
or "percent " in text[page]
|
||||
or "compensation schedule" in text[page].lower()
|
||||
or "reimbursement schedule" in text[page].lower()
|
||||
)
|
||||
elif isinstance(text, str):
|
||||
return (
|
||||
"%" in text
|
||||
or "$" in text
|
||||
or "percent " in text
|
||||
or "compensation schedule" in text.lower()
|
||||
or "reimbursement schedule" in text.lower()
|
||||
)
|
||||
# Maybe drop compensation / reimbursement schedules
|
||||
# contains and count should do a number followed by a percent or a number and a dollar
|
||||
# Also could be like "one hundred percent" or "one hundred dollars"
|
||||
# limit reimbursement strings to the first three
|
||||
|
||||
# Then for counting AND containing, use the regex
|
||||
|
||||
# maybe have `contains_reimbursement_keywords` and `contains_reimbursement_regex` functions
|
||||
# method='keyword' or method='regex'
|
||||
|
||||
reimb_regex = r"(?<![$%])(?:\$\d+|\d+[$%])(?![$%])"
|
||||
|
||||
def contains_reimbursement(text, page="1", method="keyword"): # string_funcs.py
|
||||
"""
|
||||
Checks if the given text contains any reimbursement-related keywords or patterns.
|
||||
|
||||
Args:
|
||||
text (str or dict): The text to check. If a dictionary is provided, it should have page numbers as keys.
|
||||
page (str, optional): The page number to check in the dictionary. Defaults to "1".
|
||||
method (str, optional): The method to use for checking. Options are "keyword"
|
||||
and "regex". Defaults to "keyword".
|
||||
|
||||
Returns:
|
||||
bool: True if any reimbursement-related keyword or pattern is found, False otherwise.
|
||||
"""
|
||||
if method == "keyword":
|
||||
if isinstance(text, dict):
|
||||
text_to_check = text.get(page, "").lower()
|
||||
elif isinstance(text, str):
|
||||
text_to_check = text.lower()
|
||||
else:
|
||||
print("contains_reimbursement - Invalid data type")
|
||||
return False
|
||||
return any(keyword in text_to_check for keyword in reimbursement_strings)
|
||||
elif method == "regex":
|
||||
return bool(re.search(reimb_regex, text))
|
||||
else:
|
||||
print("contains_reimbursement - Invalid data type")
|
||||
raise ValueError("Invalid method. Choose 'keyword' or 'regex'.")
|
||||
|
||||
def count_reimbursements_in_exhibit(exhibit_text: str) -> int: #JUST by regex
|
||||
"""Counts reimbursements in an exhibit text. Reimbursements are detected by a regex
|
||||
search as defined by `reimb_regex`.
|
||||
|
||||
Args:
|
||||
exhibit_text (str): Input exhibit text
|
||||
|
||||
Returns:
|
||||
int: Number of reimbursements detected
|
||||
"""
|
||||
return len(re.findall(reimb_regex, exhibit_text))
|
||||
|
||||
def is_empty(value): # string_funcs.py
|
||||
if pd.isna(value):
|
||||
@@ -251,7 +288,7 @@ def is_empty(value): # string_funcs.py
|
||||
else:
|
||||
empty_values = [None, "", "N/A", "NA", "null", "none", "NaN", np.nan, "nan"]
|
||||
return value in empty_values
|
||||
|
||||
|
||||
|
||||
def get_exhibit_chunk(text_dict: dict,
|
||||
exhibit_chunk_mapping: dict,
|
||||
|
||||
@@ -1,15 +1,531 @@
|
||||
from src.prompts import preprocessing_prompts
|
||||
import ast
|
||||
import copy
|
||||
import math
|
||||
import re
|
||||
from collections import defaultdict
|
||||
from typing import Any, Dict, List, Optional, Pattern, Tuple, Union
|
||||
|
||||
import pandas as pd
|
||||
import src.utils.llm_utils as llm_utils
|
||||
import src.utils.string_utils as string_utils
|
||||
from src import config
|
||||
from src.prompts import preprocessing_prompts
|
||||
|
||||
# Table regex patterns
|
||||
START_PAGE_PATTERN: Pattern[str] = re.compile(
|
||||
r"^(?!None\n).+\n\{.*?\}", re.DOTALL
|
||||
) # Regex for start page - shouldn't have literal `None`
|
||||
CONTINUATION_PAGE_PATTERN: Pattern[str] = re.compile(
|
||||
r"^None\n\{.*?\}", re.DOTALL
|
||||
) # Regex for continuation page - has literal `None` present at the start
|
||||
START_PAGE_NUM_PATTERN: str = r"\n\d+\n?"
|
||||
CONTINUATION_PAGE_NUM_PATTERN: str = r"\d+\n"
|
||||
DICTIONARY_PATTERN: str = r"\{.*\}"
|
||||
|
||||
# Markers for table start and end
|
||||
START_MARKER: str = "-------Table Start--------"
|
||||
END_MARKER: str = "-------Table End--------"
|
||||
|
||||
|
||||
def align_and_format_tables(text_dict, filename):
|
||||
for page_num, page_text in text_dict.items():
|
||||
if "Table Start" in page_text and string_utils.contains_reimbursement(page_text):
|
||||
# Right now we are only aligning/formatting if the page contained a reimbursement. Should we do it for all tables?
|
||||
# Decision as of 1/24/25: no, we only care about table formatting for reimbursement
|
||||
# In a possible future case where there are many NPIs/TINs, we may want to format all tables
|
||||
# But for now, our regexes catch NPIs/TINs.
|
||||
if "Table Start" in page_text and string_utils.contains_reimbursement(
|
||||
page_text
|
||||
):
|
||||
prompt = preprocessing_prompts.ALIGN_AND_FORMAT_TABLES(page_text)
|
||||
aligned_page = llm_utils.invoke_claude(
|
||||
prompt, config.MODEL_ID_CLAUDE35_SONNET, filename, 8192
|
||||
)
|
||||
text_dict[page_num] = aligned_page
|
||||
return text_dict
|
||||
|
||||
|
||||
def count_tables(exhibit_text: str) -> int:
|
||||
"""Count number of tables found in an exhibit, through use of a
|
||||
regex that catches the table start and end strings.
|
||||
|
||||
Args:
|
||||
exhibit_text (str): Input page text
|
||||
|
||||
Returns:
|
||||
int: Number of tables
|
||||
"""
|
||||
pattern = r"-Table Start-" # Source text contains "-------Table Start--------" but in case hyphen count is unreliable, just looking for one
|
||||
matches = re.findall(pattern, exhibit_text)
|
||||
return len(matches)
|
||||
|
||||
|
||||
def extract_first_table_from_page(text: str) -> str:
|
||||
"""
|
||||
Extracts the first table from a given page of text based on predefined start and end markers.
|
||||
|
||||
Args:
|
||||
text (str): The text content of the page from which the table needs to be extracted.
|
||||
|
||||
Returns:
|
||||
str: The extracted table content as a string. If the start or end markers are not found, returns an empty string.
|
||||
"""
|
||||
if START_MARKER not in text or END_MARKER not in text:
|
||||
return ""
|
||||
|
||||
start_index: int = text.find(START_MARKER) + len(START_MARKER)
|
||||
end_index: int = text.find(END_MARKER)
|
||||
filtered_page: str = text[start_index:end_index].strip()
|
||||
|
||||
return filtered_page
|
||||
|
||||
|
||||
def extract_snippets_between_markers(text: str) -> List[str]:
|
||||
"""
|
||||
Extracts all snippets from the input text that are enclosed between START_MARKER and END_MARKER.
|
||||
|
||||
Args:
|
||||
text (str): The input text from which to extract the snippets.
|
||||
|
||||
Returns:
|
||||
list: A list of substrings found between START_MARKER and END_MARKER in the input text.
|
||||
"""
|
||||
return re.findall(rf"{START_MARKER}(.+?){END_MARKER}", text, re.DOTALL)
|
||||
|
||||
|
||||
def get_str_dictionaries_from_text(
|
||||
text: str, only_first: bool = True
|
||||
) -> Union[List[str], str]:
|
||||
"""
|
||||
Extracts dictionary-like strings from the given text using a predefined pattern.
|
||||
|
||||
Args:
|
||||
text (str): The input text from which to extract dictionary-like strings.
|
||||
only_first (bool, optional): If True, returns only the first match. If False, returns all matches. Defaults to True.
|
||||
|
||||
Returns:
|
||||
list or str: A list of all matched dictionary-like strings if only_first is False, otherwise the first matched string.
|
||||
"""
|
||||
matches = re.findall(DICTIONARY_PATTERN, text)
|
||||
if not matches:
|
||||
return ""
|
||||
return matches[0] if only_first else matches
|
||||
|
||||
|
||||
def page_contains_multiple_tables(text: str) -> bool:
|
||||
"""
|
||||
Checks if the given text contains multiple tables.
|
||||
|
||||
Args:
|
||||
text (str): The text to be checked for multiple tables.
|
||||
|
||||
Returns:
|
||||
bool: True if the text contains more than one table, False otherwise.
|
||||
"""
|
||||
return len(re.findall(rf"{START_MARKER}", text)) > 1
|
||||
|
||||
|
||||
def get_page_metadata(page: str) -> Union[List[str], List[Tuple]]:
|
||||
"""
|
||||
Extracts and returns metadata from a given page.
|
||||
|
||||
If the page contains a single table, it returns the common metadata for the table.
|
||||
If the page contains multiple tables, it returns a list of metadata for each table (ordered by position in the page).
|
||||
If the page contains no tables (or contains a malformed table), it returns an empty list.
|
||||
|
||||
Args:
|
||||
page (str): The content of the page from which metadata is to be extracted.
|
||||
|
||||
Returns:
|
||||
List[str]: A list containing metadata. If the page contains a single table,
|
||||
the list contains one element with the common metadata. If the
|
||||
page contains multiple tables, the list contains tuples where
|
||||
each tuple consists of the metadata and the matched dictionary pattern (the corresponding table).
|
||||
"""
|
||||
# if the page does not contain a table (or contains a malformed table), return an empty list
|
||||
if not re.search(START_MARKER, page) or not re.search(END_MARKER, page):
|
||||
print("Malformed table in get_page_metadata!")
|
||||
return []
|
||||
|
||||
|
||||
# return the common metadata for the table (or the group of tables starting with this one)
|
||||
if not page_contains_multiple_tables(page):
|
||||
group_metadata = page[: page.find(START_MARKER)].strip()
|
||||
group_metadata = re.sub(
|
||||
START_PAGE_NUM_PATTERN, "\n", group_metadata
|
||||
) # Remove page number marker
|
||||
return [group_metadata]
|
||||
else: # if the page contains multiple tables
|
||||
all_tables = extract_snippets_between_markers(page)
|
||||
table_metadata_map = []
|
||||
|
||||
for table in all_tables:
|
||||
match = re.search(DICTIONARY_PATTERN, table, re.DOTALL)
|
||||
if match:
|
||||
metadata = table.replace(match.group(0), "").strip()
|
||||
table_metadata_map.append((metadata, match.group(0)))
|
||||
|
||||
return table_metadata_map
|
||||
|
||||
|
||||
def convert_str_to_dict(text: str) -> Dict[str, Any]:
|
||||
"""
|
||||
Converts a string representation of a dictionary to an actual dictionary.
|
||||
|
||||
This function searches for a dictionary pattern within the given text and
|
||||
converts the matched string to a dictionary using `ast.literal_eval`.
|
||||
|
||||
Args:
|
||||
text (str): The input string that potentially contains a dictionary.
|
||||
|
||||
Returns:
|
||||
dict: The dictionary extracted from the input string. If no dictionary
|
||||
pattern is found, an empty dictionary is returned.
|
||||
"""
|
||||
match = re.search(DICTIONARY_PATTERN, text)
|
||||
if match:
|
||||
dict_str: str = match.group(0)
|
||||
try:
|
||||
return ast.literal_eval(dict_str)
|
||||
except (SyntaxError, ValueError):
|
||||
print(f"Invalid dictionary format: {dict_str}")
|
||||
return {}
|
||||
|
||||
|
||||
def group_pages_containing_tables(
|
||||
text_dict: Dict[str, str]
|
||||
) -> Dict[str, Dict[str, Any]]:
|
||||
#TODO: add test cases for this function
|
||||
"""
|
||||
Groups pages containing tables into start and continuation groups based on patterns.
|
||||
|
||||
Args:
|
||||
text_dict (Dict[str, str]): A dictionary where keys are page numbers (as strings) and values are the page contents.
|
||||
|
||||
Returns:
|
||||
Dict[str, Dict[str, Any]]: A dictionary where keys are the start page numbers of groups, and values are dictionaries containing:
|
||||
- "metadata": Metadata of the start page.
|
||||
- "group": List of page numbers in the group.
|
||||
|
||||
The function processes each page in the input dictionary, identifies if the page contains a table that matches a start or continuation pattern,
|
||||
and groups the pages accordingly. If a page does not match any pattern, it is not included in any group.
|
||||
"""
|
||||
groups: Dict[str, Dict[str, Any]] = defaultdict(dict)
|
||||
current_group: List[str] = []
|
||||
group_metadata: Optional[str] = None
|
||||
for page_num, page_content in text_dict.items():
|
||||
# We can have one or more tables present on a page. The first table (ONLY) on the page decides if the page will be a start or continuation page.
|
||||
|
||||
first_table_on_page = extract_first_table_from_page(page_content)
|
||||
|
||||
if re.search(
|
||||
START_PAGE_PATTERN, first_table_on_page
|
||||
): # Check if the page matches the start pattern
|
||||
if current_group: # if a group is already established previously
|
||||
group_start_page: str = current_group[0]
|
||||
groups[group_start_page] = {
|
||||
"metadata": group_metadata,
|
||||
"group": current_group,
|
||||
}
|
||||
current_group = (
|
||||
[]
|
||||
) # reset the current group, because we've found the start pattern
|
||||
group_metadata = None
|
||||
current_group = [page_num]
|
||||
group_metadata = get_page_metadata(page_content)
|
||||
elif re.search(
|
||||
CONTINUATION_PAGE_PATTERN, first_table_on_page
|
||||
): # Check if the page matches the continuation pattern
|
||||
if current_group:
|
||||
current_group.append(page_num)
|
||||
else: # If the first page we're processing is a continuation page, we start a new group (edge case)
|
||||
current_group = [page_num]
|
||||
group_metadata = get_page_metadata(page_content)
|
||||
else: # If the page doesn't match any pattern
|
||||
if current_group: # if we're in the middle of a group, close it
|
||||
group_start_page = current_group[0]
|
||||
groups[group_start_page] = {
|
||||
"metadata": group_metadata,
|
||||
"group": current_group,
|
||||
}
|
||||
current_group = [] # reset the current group
|
||||
group_metadata = None
|
||||
if current_group: # save off the current group if we're at the end of the text
|
||||
group_start_page = current_group[0]
|
||||
groups[group_start_page] = {"metadata": group_metadata, "group": current_group}
|
||||
return groups
|
||||
|
||||
|
||||
def insert_metadata(metadata: str, table_content: str) -> str:
|
||||
"""
|
||||
Inserts metadata at the beginning of the table content.
|
||||
|
||||
Args:
|
||||
metadata (str): The metadata to be inserted.
|
||||
table_content (str): The content of the table.
|
||||
|
||||
Returns:
|
||||
str: The combined string with metadata followed by the table content.
|
||||
"""
|
||||
return f"{metadata}\n{table_content}"
|
||||
|
||||
|
||||
def insert_column_headers(
|
||||
data_dict: dict[str, list[str]], columns: list
|
||||
) -> dict[str, list[str]]:
|
||||
"""
|
||||
Inserts column headers into a dictionary of data. It takes the existing column headers and inserts them as the first row of the data.
|
||||
|
||||
Args:
|
||||
data_dict (dict): The dictionary containing the data to be modified.
|
||||
columns (list): A list of column headers to be inserted.
|
||||
|
||||
Returns:
|
||||
dict: A new dictionary with the column headers inserted. If the input dictionary is empty, returns an empty dictionary.
|
||||
"""
|
||||
if not data_dict:
|
||||
print("Dictionary not found in the text.")
|
||||
return {}
|
||||
record = list(data_dict.items())
|
||||
modified_dict = {
|
||||
col_name: [col_items[0]] + col_items[1]
|
||||
for col_name, col_items in zip(columns, record)
|
||||
}
|
||||
|
||||
return modified_dict
|
||||
|
||||
|
||||
def correct_column_headers_and_merge(
|
||||
text_dict: dict[str, str], groups: dict[str, dict[str, Any]]
|
||||
) -> dict[str, List[pd.DataFrame]]:
|
||||
"""
|
||||
Corrects column headers and merges tables from multiple pages.
|
||||
|
||||
Args:
|
||||
text_dict (dict[str, str]): A dictionary where keys are page numbers and values are the text content of those pages.
|
||||
groups (dict): A dictionary where keys are the starting page numbers of groups and values are dictionaries containing group data.
|
||||
|
||||
Returns:
|
||||
dict: A dictionary where keys are the starting page numbers of groups and values are lists of DataFrames representing the merged tables.
|
||||
"""
|
||||
|
||||
page_tables_map = {}
|
||||
|
||||
columns_to_use: List[str] = [] # for mypy
|
||||
|
||||
for group_start_page, group_data in groups.items(): # Loop through groups
|
||||
dfs_list = []
|
||||
columns = None
|
||||
|
||||
for page in group_data["group"]: # Loop through pages in the group
|
||||
page_content = text_dict[page]
|
||||
|
||||
str_dicts = get_str_dictionaries_from_text(page_content, only_first=False)
|
||||
for str_dict in str_dicts:
|
||||
data_dict = convert_str_to_dict(str_dict)
|
||||
# if the page contains multiple tables, rewrite the latest column headers each time
|
||||
# Otherwise, use the column headers from the first table in the group
|
||||
columns_to_use = data_dict.keys() if page == group_start_page else columns_to_use
|
||||
|
||||
col_corrected_dict = insert_column_headers(data_dict, columns_to_use)
|
||||
|
||||
if (
|
||||
page == group_start_page
|
||||
): # on the first table, we've already changed the keys to be the first row, so we need to drop that row
|
||||
df = pd.DataFrame(col_corrected_dict).drop(index=0)
|
||||
else:
|
||||
df = pd.DataFrame(col_corrected_dict)
|
||||
|
||||
df["page_num"] = page
|
||||
|
||||
dfs_list.append(df)
|
||||
|
||||
tables = dfs_list
|
||||
page_tables_map[group_start_page] = (
|
||||
tables # List of dataframes corresponding to the tables in the group
|
||||
)
|
||||
return page_tables_map
|
||||
|
||||
|
||||
def group_and_merge_tables(text_dict: dict[str, str]) -> tuple[dict, dict]:
|
||||
"""
|
||||
Groups pages containing tables and merges the tables after correcting column headers.
|
||||
|
||||
Args:
|
||||
text_dict (dict[str, str]): A dictionary where keys are page identifiers and values are the text content of the pages.
|
||||
|
||||
Returns:
|
||||
tuple[dict, dict]: A tuple containing:
|
||||
- groups (dict): A dictionary where keys are group identifiers and values are lists of page identifiers that belong to each group.
|
||||
- page_tables_map (dict): A dictionary where keys are page identifiers and values are the merged tables after correcting column headers.
|
||||
"""
|
||||
groups = group_pages_containing_tables(text_dict)
|
||||
|
||||
page_tables_map = correct_column_headers_and_merge(
|
||||
text_dict, groups
|
||||
) # extract the tables from all pages, correct the col headers, put the individual tables in the df, then merge those dfs
|
||||
|
||||
return groups, page_tables_map
|
||||
|
||||
|
||||
def split_table_to_pages(
|
||||
groups: dict, page_tables_map: dict, row_limit: int
|
||||
) -> dict:
|
||||
"""
|
||||
Splits tables into pages based on the provided groups and page tables map.
|
||||
Args:
|
||||
groups (dict): A dictionary where keys are the starting page numbers of groups and values are dictionaries containing group data.
|
||||
page_tables_map (dict): A dictionary mapping page numbers to lists of DataFrames representing tables on those pages.
|
||||
row_limit (int, optional): The number of rows per table split. Defaults to 5.
|
||||
Returns:
|
||||
dict: A dictionary where keys are the starting page numbers of groups and values are dictionaries mapping page numbers to lists of related tables.
|
||||
"""
|
||||
group_table_split_info = {}
|
||||
|
||||
for group_start_page, group_data in groups.items(): # Loop through groups
|
||||
dfs_list = page_tables_map[group_start_page] # get list of tables for the group
|
||||
metadata_list = group_data["metadata"]
|
||||
|
||||
if all(
|
||||
[isinstance(metadata, tuple) for metadata in metadata_list]
|
||||
): # if the metadata is a list of tuples, this page has multiple tables
|
||||
metadata_list = [
|
||||
metadata[0] for metadata in metadata_list
|
||||
] # Just extract the metadata strings (the second element of the tuple in the metadata is a text dictionary snippet)
|
||||
|
||||
# if the group contains multiple pages with common metadata, duplicate the metadata for each page
|
||||
# This handles cases:
|
||||
# - when the group contains multiple pages with common metadata, this will work
|
||||
# - when the page contains multiple tables which start AND end on the single page, this will work
|
||||
# An unhandled edge case, when the page contains multiple tables in this scenario:
|
||||
# - when the LAST table of a page is the beginning of a table continued onto the next page, we won't get the metadata right
|
||||
|
||||
# If the group has a single page with multiple tables
|
||||
if len(group_data["group"]) > 1: # if the group contains multiple pages
|
||||
metadata_list = [
|
||||
metadata_list[0] for _ in range(len(dfs_list))
|
||||
] # duplicate the metadata for each page
|
||||
|
||||
if (
|
||||
len(group_data["group"]) == 1
|
||||
): # its a single page group with len(dfs_list) number of tables - eg [6]->[6,6,6,6] if the group contains just page "6" and page "6" has 4 tables
|
||||
group_data["group"] = [group_data["group"][0] for _ in range(len(dfs_list))]
|
||||
|
||||
chunked_tables_page_mapping = defaultdict(list)
|
||||
|
||||
# loop through each page, metadata, and dataframe in the group
|
||||
for page_num, metadata, df in zip(group_data["group"], metadata_list, dfs_list):
|
||||
num_rows = df.shape[0]
|
||||
num_tables = math.ceil(num_rows / row_limit) # number of output tables
|
||||
|
||||
related_tables = []
|
||||
|
||||
for i in range(num_tables):
|
||||
start_row = i * row_limit
|
||||
end_row = min((i + 1) * row_limit, df.shape[0])
|
||||
table_df = df.iloc[start_row:end_row, :]
|
||||
|
||||
str_df = str(table_df.drop(columns=["page_num"]).to_dict(orient="list"))
|
||||
|
||||
table_dict = insert_metadata(metadata, str_df)
|
||||
|
||||
assert ( # TODO: convert to `raise` instead of `assert`
|
||||
table_df["page_num"].nunique() == 1
|
||||
), "All rows in a table should belong to the same page"
|
||||
|
||||
page_num = table_df["page_num"].unique()[0]
|
||||
|
||||
related_tables.append(
|
||||
table_dict
|
||||
) # these tables are splits of a larger table from one page
|
||||
|
||||
chunked_tables_page_mapping[page_num].append(related_tables)
|
||||
|
||||
group_table_split_info[group_start_page] = chunked_tables_page_mapping
|
||||
|
||||
return group_table_split_info
|
||||
|
||||
|
||||
def replace_tables_in_page(
|
||||
page_num: int, modified_text_dict: dict, new_tables_list: list
|
||||
) -> None:
|
||||
"""
|
||||
Replaces existing tables within pages to modified (corrected) tables.
|
||||
Args:
|
||||
page_num (int): The page number where the reinsertion should occur.
|
||||
modified_text_dict (dict): A dictionary containing the text of each page, with page numbers as keys.
|
||||
new_tables_list (list): A list of lists, where each inner list contains tables to be inserted.
|
||||
Returns:
|
||||
None: This function modifies the `modified_text_dict` in place.
|
||||
"""
|
||||
|
||||
text = modified_text_dict[page_num]
|
||||
|
||||
formatted_tables = []
|
||||
|
||||
for (
|
||||
table_list
|
||||
) in new_tables_list: # Format new tables for insertion back into pages
|
||||
lst = []
|
||||
|
||||
for table in table_list:
|
||||
lst.append(f"{START_MARKER}\n{table}\n{END_MARKER}")
|
||||
|
||||
concatenated_tables = "\n".join(lst)
|
||||
|
||||
formatted_tables.append(concatenated_tables)
|
||||
|
||||
detect_tables_pattern = rf"({START_MARKER}.+?{END_MARKER})"
|
||||
|
||||
tables_to_replace = re.findall(detect_tables_pattern, text, re.DOTALL)
|
||||
|
||||
for original_table, replacement_table in zip(tables_to_replace, formatted_tables):
|
||||
text = re.sub(
|
||||
re.escape(original_table), replacement_table, text, count=1, flags=re.DOTALL
|
||||
)
|
||||
|
||||
modified_text_dict[page_num] = text
|
||||
|
||||
|
||||
def reinsert_new_tables_to_text(group_table_split_info: dict, text_dict: dict) -> dict:
|
||||
"""
|
||||
Take what's in groups, apply `replace_tables_in_page` to each group, and then reinsert the tables back into the text.
|
||||
This is really a wrapper function for `replace_tables_in_page` which goes through each page and applies the function.
|
||||
|
||||
Args:
|
||||
group_table_split_info (dict): A dictionary where the keys are the starting page numbers of groups and the values
|
||||
are dictionaries mapping page numbers to lists of split tables.
|
||||
|
||||
Returns:
|
||||
dict: A modified version of the original text dictionary with the tables reinserted.
|
||||
"""
|
||||
modified_text_dict = copy.deepcopy(text_dict)
|
||||
|
||||
for group_start_page, chunked_tables_page_mapping in group_table_split_info.items():
|
||||
|
||||
for page_num in chunked_tables_page_mapping.keys():
|
||||
split_tables_list = chunked_tables_page_mapping[page_num]
|
||||
# perform the reinsertion (inplace)
|
||||
replace_tables_in_page(page_num, modified_text_dict, split_tables_list)
|
||||
|
||||
return modified_text_dict
|
||||
|
||||
def split_tables_to_separate_subpages(text_dict: dict[str, str]) -> dict[str, str]:
|
||||
"""
|
||||
Splits the text content of each page in the input dictionary into separate subpages based on a table end marker.
|
||||
|
||||
Args:
|
||||
text_dict (dict[str, str]): A dictionary where keys are page identifiers and values are the text content of those pages.
|
||||
|
||||
Returns:
|
||||
dict[str, str]: A new dictionary where keys are modified page identifiers (including subpage indices) and values are the split text content. Empty strings are removed from the result.
|
||||
"""
|
||||
new_text_dict = {}
|
||||
for page in text_dict.keys():
|
||||
# split the text into separate pages based on the table end marker
|
||||
if END_MARKER in text_dict[page]:
|
||||
new_text_list = re.split(fr"(?<={END_MARKER})", text_dict[page]) # this regex will split on the end marker, and keep the end marker in the text
|
||||
for i, text in enumerate(new_text_list):
|
||||
new_text_dict[f"{page}.{i}"] = text
|
||||
else:
|
||||
new_text_dict[page] = text_dict[page]
|
||||
return {k : v for k, v in new_text_dict.items() if v.strip()} # remove empty strings
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import pytest
|
||||
from src.preprocessing_funcs import split_text
|
||||
from src.preprocessing_funcs import (
|
||||
remove_page_indicators,
|
||||
split_text,
|
||||
@@ -21,31 +22,6 @@ class TestPreprocessingFuncs:
|
||||
def test_clean_newlines(self, input_text, expected_output):
|
||||
assert remove_page_indicators(input_text) == expected_output
|
||||
|
||||
# Test cases for split_text
|
||||
@pytest.mark.parametrize("input_text, expected_output", [
|
||||
# Splits on page markers
|
||||
(
|
||||
"Title Start of Page No. = 1\nContent1\nStart of Page No. = 2\nContent2",
|
||||
{"Title" : "Title ", "1": "Content1\n", "2": "Content2"}
|
||||
),
|
||||
# Ignores "Document" key
|
||||
(
|
||||
"Document\nStart of Page No. = 1\nContent1",
|
||||
{"1": "Content1"}
|
||||
),
|
||||
# Handles empty input
|
||||
("", {}),
|
||||
# Handles no page markers
|
||||
("This is a single page.", {"This": "This is a single page."}),
|
||||
# Handles multiple page markers
|
||||
(
|
||||
"Introduction\nStart of Page No. = 1\nContent1\nStart of Page No. = 2\nContent2\nStart of Page No. = 3\nContent3",
|
||||
{'Introduction': 'Introduction\n', '1': 'Content1\n', '2': 'Content2\n', '3': 'Content3'}
|
||||
),
|
||||
])
|
||||
def test_split_text(self, input_text, expected_output):
|
||||
assert split_text(input_text) == expected_output
|
||||
|
||||
# Test cases for clean_law_symbols
|
||||
@pytest.mark.parametrize("input_text, expected_output", [
|
||||
# Replaces double dollar signs
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
import pytest
|
||||
import src.utils as utils
|
||||
from src.utils.string_utils import extract_text_from_delimiters, json_parsing_search, secondary_string_to_dict, primary_string_to_dict, contains_reimbursement, is_empty
|
||||
from src.utils.string_utils import extract_text_from_delimiters, json_parsing_search, secondary_string_to_dict, primary_string_to_dict, contains_reimbursement, is_empty, count_reimbursements_in_exhibit
|
||||
import src.utils.llm_utils as llm_utils
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
@@ -175,7 +175,7 @@ class TestStringUtils:
|
||||
text = ["This is a list, not a dictionary or string."]
|
||||
page = "1"
|
||||
result = contains_reimbursement(text, page)
|
||||
assert result is None
|
||||
assert result is False
|
||||
captured = capsys.readouterr()
|
||||
assert "contains_reimbursement - Invalid data type" in captured.out
|
||||
|
||||
@@ -194,3 +194,17 @@ class TestStringUtils:
|
||||
def test_is_empty(self, value, expected):
|
||||
result = is_empty(value)
|
||||
assert result == expected
|
||||
|
||||
class TestCountReimbursementsInExhibit:
|
||||
@pytest.mark.parametrize("exhibit_text, expected_count", [
|
||||
("This exhibit includes a 10% reimbursement and a $100 reimbursement.", 2),
|
||||
("No reimbursements mentioned here.", 0),
|
||||
("Reimbursement of 50% and another reimbursement of $200.", 2),
|
||||
("100 percent reimbursement and fifty dollars reimbursement.", 0),
|
||||
("", 0),
|
||||
("Reimbursement: 20% and $300.", 2),
|
||||
("Reimbursement: 20% and $300. Another 10% reimbursement.", 3),
|
||||
])
|
||||
def test_count_reimbursements_in_exhibit(self, exhibit_text, expected_count):
|
||||
result = count_reimbursements_in_exhibit(exhibit_text)
|
||||
assert result == expected_count
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
import pytest
|
||||
from src.prompts import preprocessing_prompts
|
||||
import src.utils as utils
|
||||
from src.utils.table_utils import align_and_format_tables
|
||||
from src import (config)
|
||||
from src import config
|
||||
import copy
|
||||
from src.utils.table_utils import *
|
||||
|
||||
class TestAlignAndFormatTables:
|
||||
@pytest.fixture
|
||||
@@ -112,4 +112,618 @@ class TestAlignAndFormatTables:
|
||||
)
|
||||
for call in mock_invoke_claude.call_args_list:
|
||||
prompt, _, _, _ = call[0] # Unpack the arguments
|
||||
assert input_text_dict["2"] not in prompt
|
||||
assert input_text_dict["2"] not in prompt
|
||||
|
||||
class TestCountTables:
|
||||
def test_count_tables_single_table(self):
|
||||
"""Test counting tables when there is a single table."""
|
||||
exhibit_text = (
|
||||
"-------Table Start-------- Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}-------Table End--------"
|
||||
)
|
||||
result = count_tables(exhibit_text)
|
||||
assert result == 1
|
||||
|
||||
def test_count_tables_multiple_tables(self):
|
||||
"""Test counting tables when there are multiple tables."""
|
||||
exhibit_text = (
|
||||
"-------Table Start-------- Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}-------Table End--------\n"
|
||||
"-------Table Start-------- Another Table Header {'Column A': ['Value A1', 'Value A2'], 'Column B': ['Value B1', 'Value B2']}-------Table End--------"
|
||||
)
|
||||
result = count_tables(exhibit_text)
|
||||
assert result == 2
|
||||
|
||||
def test_count_tables_no_table(self):
|
||||
"""Test counting tables when there are no tables."""
|
||||
exhibit_text = (
|
||||
"This page does not contain a table.\n"
|
||||
"Neither does this page."
|
||||
)
|
||||
result = count_tables(exhibit_text)
|
||||
assert result == 0
|
||||
|
||||
def test_count_tables_mixed_content(self):
|
||||
"""Test counting tables when there is a mix of pages with and without tables."""
|
||||
exhibit_text = (
|
||||
"-------Table Start-------- Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}-------Table End--------\n"
|
||||
"This page does not contain a table.\n"
|
||||
"-------Table Start-------- Another Table Header {'Column A': ['Value A1', 'Value A2'], 'Column B': ['Value B1', 'Value B2']}-------Table End--------"
|
||||
)
|
||||
result = count_tables(exhibit_text)
|
||||
assert result == 2
|
||||
|
||||
class TestExtractFirstTableFromPage:
|
||||
def test_extract_first_table_from_page_with_table(self):
|
||||
"""Test extracting the first table when the page contains a table."""
|
||||
text = (
|
||||
"Some text before the table.\n"
|
||||
"-------Table Start--------\n"
|
||||
"Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n"
|
||||
"-------Table End--------\n"
|
||||
"Some text after the table."
|
||||
)
|
||||
result = extract_first_table_from_page(text)
|
||||
expected = "Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}"
|
||||
assert result == expected
|
||||
|
||||
def test_extract_first_table_from_page_no_table(self):
|
||||
"""Test extracting the first table when the page does not contain a table."""
|
||||
text = "This page does not contain a table."
|
||||
result = extract_first_table_from_page(text)
|
||||
assert result == ""
|
||||
|
||||
def test_extract_first_table_from_page_multiple_tables(self):
|
||||
"""Test extracting the first table when the page contains multiple tables."""
|
||||
text = (
|
||||
"Some text before the tables.\n"
|
||||
"-------Table Start--------\n"
|
||||
"First Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n"
|
||||
"-------Table End--------\n"
|
||||
"Some text between the tables.\n"
|
||||
"-------Table Start--------\n"
|
||||
"Second Table Header {'Column A': ['Value A1', 'Value A2'], 'Column B': ['Value B1', 'Value B2']}\n"
|
||||
"-------Table End--------\n"
|
||||
"Some text after the tables."
|
||||
)
|
||||
result = extract_first_table_from_page(text)
|
||||
expected = "First Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}"
|
||||
assert result == expected
|
||||
|
||||
def test_extract_first_table_from_page_no_end_marker(self):
|
||||
"""Test extracting the first table when the end marker is missing."""
|
||||
text = (
|
||||
"Some text before the table.\n"
|
||||
"-------Table Start--------\n"
|
||||
"Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n"
|
||||
"Some text after the table."
|
||||
)
|
||||
result = extract_first_table_from_page(text)
|
||||
assert result == ""
|
||||
|
||||
def test_extract_first_table_from_page_no_start_marker(self):
|
||||
"""Test extracting the first table when the start marker is missing."""
|
||||
text = (
|
||||
"Some text before the table.\n"
|
||||
"Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n"
|
||||
"-------Table End--------\n"
|
||||
"Some text after the table."
|
||||
)
|
||||
result = extract_first_table_from_page(text)
|
||||
assert result == ""
|
||||
|
||||
class TestExtractSnippetsBetweenMarkers:
|
||||
def test_extract_snippets_between_markers_single_snippet(self):
|
||||
"""Test extracting snippets when there is a single snippet."""
|
||||
text = (
|
||||
"Some text before the table.\n"
|
||||
"-------Table Start--------\n"
|
||||
"Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n"
|
||||
"-------Table End--------\n"
|
||||
"Some text after the table."
|
||||
)
|
||||
result = extract_snippets_between_markers(text)
|
||||
expected = ["\nTable Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n"]
|
||||
assert result == expected
|
||||
|
||||
def test_extract_snippets_between_markers_multiple_snippets(self):
|
||||
"""Test extracting snippets when there are multiple snippets."""
|
||||
text = (
|
||||
"Some text before the tables.\n"
|
||||
"-------Table Start--------\n"
|
||||
"First Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n"
|
||||
"-------Table End--------\n"
|
||||
"Some text between the tables.\n"
|
||||
"-------Table Start--------\n"
|
||||
"Second Table Header {'Column A': ['Value A1', 'Value A2'], 'Column B': ['Value B1', 'Value B2']}\n"
|
||||
"-------Table End--------\n"
|
||||
"Some text after the tables."
|
||||
)
|
||||
result = extract_snippets_between_markers(text)
|
||||
expected = [
|
||||
"\nFirst Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n",
|
||||
"\nSecond Table Header {'Column A': ['Value A1', 'Value A2'], 'Column B': ['Value B1', 'Value B2']}\n"
|
||||
]
|
||||
assert result == expected
|
||||
|
||||
def test_extract_snippets_between_markers_no_snippets(self):
|
||||
"""Test extracting snippets when there are no snippets."""
|
||||
text = "This page does not contain a table."
|
||||
result = extract_snippets_between_markers(text)
|
||||
assert result == []
|
||||
|
||||
def test_extract_snippets_between_markers_no_end_marker(self):
|
||||
"""Test extracting snippets when the end marker is missing."""
|
||||
text = (
|
||||
"Some text before the table.\n"
|
||||
"-------Table Start--------\n"
|
||||
"Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n"
|
||||
"Some text after the table."
|
||||
)
|
||||
result = extract_snippets_between_markers(text)
|
||||
assert result == []
|
||||
|
||||
def test_extract_snippets_between_markers_no_start_marker(self):
|
||||
"""Test extracting snippets when the start marker is missing."""
|
||||
text = (
|
||||
"Some text before the table.\n"
|
||||
"Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n"
|
||||
"-------Table End--------\n"
|
||||
"Some text after the table."
|
||||
)
|
||||
result = extract_snippets_between_markers(text)
|
||||
assert result == []
|
||||
|
||||
|
||||
class TestGetStrDictionariesFromText:
|
||||
def test_get_str_dictionaries_from_text_single_match(self):
|
||||
"""Test extracting a single dictionary-like string when only_first is True."""
|
||||
text = (
|
||||
"Some text before the dictionary.\n"
|
||||
"{'key1': 'value1', 'key2': 'value2'}\n"
|
||||
"Some text after the dictionary."
|
||||
)
|
||||
result = get_str_dictionaries_from_text(text, only_first=True)
|
||||
expected = "{'key1': 'value1', 'key2': 'value2'}"
|
||||
assert result == expected
|
||||
|
||||
def test_get_str_dictionaries_from_text_multiple_matches(self):
|
||||
"""Test extracting all dictionary-like strings when only_first is False."""
|
||||
text = (
|
||||
"Some text before the dictionaries.\n"
|
||||
"{'key1': 'value1', 'key2': 'value2'}\n"
|
||||
"Some text between the dictionaries.\n"
|
||||
"{'keyA': 'valueA', 'keyB': 'valueB'}\n"
|
||||
"Some text after the dictionaries."
|
||||
)
|
||||
result = get_str_dictionaries_from_text(text, only_first=False)
|
||||
expected = [
|
||||
"{'key1': 'value1', 'key2': 'value2'}",
|
||||
"{'keyA': 'valueA', 'keyB': 'valueB'}"
|
||||
]
|
||||
assert result == expected
|
||||
|
||||
def test_get_str_dictionaries_from_text_no_match(self):
|
||||
"""Test extracting dictionary-like strings when there are no matches."""
|
||||
text = "This text does not contain any dictionary-like strings."
|
||||
result = get_str_dictionaries_from_text(text, only_first=True)
|
||||
assert result == ""
|
||||
|
||||
def test_get_str_dictionaries_from_text_empty_text(self):
|
||||
"""Test extracting dictionary-like strings from an empty text."""
|
||||
text = ""
|
||||
result = get_str_dictionaries_from_text(text, only_first=True)
|
||||
assert result == ""
|
||||
|
||||
def test_get_str_dictionaries_from_text_single_match_only_first_false(self):
|
||||
"""Test extracting dictionary-like strings when there is a single match and only_first is False."""
|
||||
text = (
|
||||
"Some text before the dictionary.\n"
|
||||
"{'key1': 'value1', 'key2': 'value2'}\n"
|
||||
"Some text after the dictionary."
|
||||
)
|
||||
result = get_str_dictionaries_from_text(text, only_first=False)
|
||||
expected = ["{'key1': 'value1', 'key2': 'value2'}"]
|
||||
assert result == expected
|
||||
|
||||
|
||||
class TestPageContainsMultipleTables:
|
||||
def test_page_contains_multiple_tables_single_table(self):
|
||||
"""Test when the page contains a single table."""
|
||||
text = (
|
||||
"Some text before the table.\n"
|
||||
"-------Table Start--------\n"
|
||||
"Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n"
|
||||
"-------Table End--------\n"
|
||||
"Some text after the table."
|
||||
)
|
||||
result = page_contains_multiple_tables(text)
|
||||
assert result is False
|
||||
|
||||
def test_page_contains_multiple_tables_multiple_tables(self):
|
||||
"""Test when the page contains multiple tables."""
|
||||
text = (
|
||||
"Some text before the tables.\n"
|
||||
"-------Table Start--------\n"
|
||||
"First Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n"
|
||||
"-------Table End--------\n"
|
||||
"Some text between the tables.\n"
|
||||
"-------Table Start--------\n"
|
||||
"Second Table Header {'Column A': ['Value A1', 'Value A2'], 'Column B': ['Value B1', 'Value B2']}\n"
|
||||
"-------Table End--------\n"
|
||||
"Some text after the tables."
|
||||
)
|
||||
result = page_contains_multiple_tables(text)
|
||||
assert result is True
|
||||
|
||||
def test_page_contains_multiple_tables_no_table(self):
|
||||
"""Test when the page does not contain any tables."""
|
||||
text = "This page does not contain a table."
|
||||
result = page_contains_multiple_tables(text)
|
||||
assert result is False
|
||||
|
||||
def test_page_contains_multiple_tables_empty_text(self):
|
||||
"""Test when the text is empty."""
|
||||
text = ""
|
||||
result = page_contains_multiple_tables(text)
|
||||
assert result is False
|
||||
|
||||
def test_page_contains_multiple_tables_table_without_end_marker(self):
|
||||
"""Test when the page contains a table without an end marker."""
|
||||
text = (
|
||||
"Some text before the table.\n"
|
||||
"-------Table Start--------\n"
|
||||
"Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n"
|
||||
"Some text after the table."
|
||||
)
|
||||
result = page_contains_multiple_tables(text)
|
||||
assert result is False
|
||||
|
||||
|
||||
class TestGetPageMetadata:
|
||||
def test_get_page_metadata_single_table(self):
|
||||
"""Test extracting metadata when the page contains a single table."""
|
||||
page = (
|
||||
"Page metadata before the table.\n"
|
||||
"-------Table Start--------\n"
|
||||
"Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n"
|
||||
"-------Table End--------\n"
|
||||
"Some text after the table."
|
||||
)
|
||||
result = get_page_metadata(page)
|
||||
expected = ["Page metadata before the table."]
|
||||
assert result == expected
|
||||
|
||||
def test_get_page_metadata_multiple_tables(self):
|
||||
"""Test extracting metadata when the page contains multiple tables."""
|
||||
page = (
|
||||
"Page metadata before the tables.\n"
|
||||
"-------Table Start--------\n"
|
||||
"First Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n"
|
||||
"-------Table End--------\n"
|
||||
"Some text between the tables.\n"
|
||||
"-------Table Start--------\n"
|
||||
"Second Table Header {'Column A': ['Value A1', 'Value A2'], 'Column B': ['Value B1', 'Value B2']}\n"
|
||||
"-------Table End--------\n"
|
||||
"Some text after the tables."
|
||||
)
|
||||
result = get_page_metadata(page)
|
||||
expected = [
|
||||
("First Table Header", "{'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}"),
|
||||
("Second Table Header", "{'Column A': ['Value A1', 'Value A2'], 'Column B': ['Value B1', 'Value B2']}")
|
||||
]
|
||||
assert result == expected
|
||||
|
||||
def test_get_page_metadata_no_table(self):
|
||||
"""Test extracting metadata when the page does not contain any tables."""
|
||||
page = "This page does not contain a table."
|
||||
result = get_page_metadata(page)
|
||||
expected = []
|
||||
assert result == expected
|
||||
|
||||
def test_get_page_metadata_empty_text(self):
|
||||
"""Test extracting metadata from an empty text."""
|
||||
page = ""
|
||||
result = get_page_metadata(page)
|
||||
expected = []
|
||||
assert result == expected
|
||||
|
||||
def test_get_page_metadata_table_without_end_marker(self):
|
||||
"""Test extracting metadata when the page contains a table without an end marker."""
|
||||
page = (
|
||||
"Page metadata before the table.\n"
|
||||
"-------Table Start--------\n"
|
||||
"Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n"
|
||||
"Some text after the table."
|
||||
)
|
||||
result = get_page_metadata(page)
|
||||
expected = []
|
||||
assert result == expected
|
||||
|
||||
def test_get_page_metadata_table_without_start_marker(self):
|
||||
"""Test extracting metadata when the page contains a table without a start marker."""
|
||||
page = (
|
||||
"Page metadata before the table.\n"
|
||||
"Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n"
|
||||
"-------Table End--------\n"
|
||||
"Some text after the table."
|
||||
)
|
||||
result = get_page_metadata(page)
|
||||
expected = []
|
||||
assert result == expected
|
||||
|
||||
class TestConvertStrToDict:
|
||||
def test_convert_str_to_dict_valid_dict(self):
|
||||
"""Test converting a valid dictionary-like string."""
|
||||
text = "{'key1': 'value1', 'key2': 'value2'}"
|
||||
result = convert_str_to_dict(text)
|
||||
expected = {'key1': 'value1', 'key2': 'value2'}
|
||||
assert result == expected
|
||||
|
||||
def test_convert_str_to_dict_no_dict(self):
|
||||
"""Test converting a string that does not contain a dictionary."""
|
||||
text = "This text does not contain any dictionary-like strings."
|
||||
result = convert_str_to_dict(text)
|
||||
expected = {}
|
||||
assert result == expected
|
||||
|
||||
def test_convert_str_to_dict_empty_text(self):
|
||||
"""Test converting an empty string."""
|
||||
text = ""
|
||||
result = convert_str_to_dict(text)
|
||||
expected = {}
|
||||
assert result == expected
|
||||
|
||||
def test_convert_str_to_dict_partial_dict(self):
|
||||
"""Test converting a string that contains a partial dictionary."""
|
||||
text = "Some text before the dictionary {'key1': 'value1', 'key2': 'value2'"
|
||||
result = convert_str_to_dict(text)
|
||||
expected = {}
|
||||
assert result == expected
|
||||
|
||||
def test_convert_str_to_dict_multiple_dicts(self):
|
||||
"""Test converting a string that contains multiple dictionaries."""
|
||||
text = (
|
||||
"Some text before the dictionaries.\n"
|
||||
"{'key1': 'value1', 'key2': 'value2'}\n"
|
||||
"Some text between the dictionaries.\n"
|
||||
"{'keyA': 'valueA', 'keyB': 'valueB'}\n"
|
||||
"Some text after the dictionaries."
|
||||
)
|
||||
result = convert_str_to_dict(text)
|
||||
expected = {'key1': 'value1', 'key2': 'value2'}
|
||||
assert result == expected
|
||||
|
||||
def test_convert_str_to_dict_invalid_dict(self):
|
||||
"""Test converting a string that contains an invalid dictionary."""
|
||||
text = "{'key1': 'value1', 'key2': value2}" # Missing quotes around value2
|
||||
result = convert_str_to_dict(text)
|
||||
expected = {}
|
||||
assert result == expected
|
||||
|
||||
|
||||
class TestInsertMetadata:
|
||||
def test_insert_metadata_with_valid_data(self):
|
||||
"""Test inserting metadata with valid data."""
|
||||
metadata = "Page metadata"
|
||||
table_content = "Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}"
|
||||
result = insert_metadata(metadata, table_content)
|
||||
expected = "Page metadata\nTable Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}"
|
||||
assert result == expected
|
||||
|
||||
def test_insert_metadata_with_empty_metadata(self):
|
||||
"""Test inserting metadata when metadata is empty."""
|
||||
metadata = ""
|
||||
table_content = "Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}"
|
||||
result = insert_metadata(metadata, table_content)
|
||||
expected = "\nTable Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}"
|
||||
assert result == expected
|
||||
|
||||
def test_insert_metadata_with_empty_table_content(self):
|
||||
"""Test inserting metadata when table content is empty."""
|
||||
metadata = "Page metadata"
|
||||
table_content = ""
|
||||
result = insert_metadata(metadata, table_content)
|
||||
expected = "Page metadata\n"
|
||||
assert result == expected
|
||||
|
||||
def test_insert_metadata_with_both_empty(self):
|
||||
"""Test inserting metadata when both metadata and table content are empty."""
|
||||
metadata = ""
|
||||
table_content = ""
|
||||
result = insert_metadata(metadata, table_content)
|
||||
expected = "\n"
|
||||
assert result == expected
|
||||
|
||||
def test_insert_metadata_with_special_characters(self):
|
||||
"""Test inserting metadata when metadata and table content contain special characters."""
|
||||
metadata = "Page metadata with special characters: !@#$%^&*()"
|
||||
table_content = "Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']} with special characters: <>?/\\|"
|
||||
result = insert_metadata(metadata, table_content)
|
||||
expected = "Page metadata with special characters: !@#$%^&*()\nTable Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']} with special characters: <>?/\\|"
|
||||
assert result == expected
|
||||
|
||||
|
||||
class TestInsertColumnHeaders:
|
||||
def test_insert_column_headers_with_valid_data(self):
|
||||
"""Test inserting column headers with valid data."""
|
||||
data_dict = {
|
||||
"Column 1": ["Value 1", "Value 2"],
|
||||
"Column 2": ["Value 3", "Value 4"]
|
||||
}
|
||||
columns = ["Column 1", "Column 2"]
|
||||
result = insert_column_headers(data_dict, columns)
|
||||
expected = {
|
||||
"Column 1": ["Column 1", "Value 1", "Value 2"],
|
||||
"Column 2": ["Column 2", "Value 3", "Value 4"]
|
||||
}
|
||||
assert result == expected
|
||||
|
||||
def test_insert_column_headers_with_empty_data_dict(self):
|
||||
"""Test inserting column headers when data_dict is empty."""
|
||||
data_dict = {}
|
||||
columns = ["Column 1", "Column 2"]
|
||||
result = insert_column_headers(data_dict, columns)
|
||||
expected = {}
|
||||
assert result == expected
|
||||
|
||||
def test_insert_column_headers_with_mismatched_columns(self):
|
||||
"""Test inserting column headers when columns list does not match data_dict keys."""
|
||||
data_dict = {
|
||||
"Column 1": ["Value 1", "Value 2"],
|
||||
"Column 2": ["Value 3", "Value 4"]
|
||||
}
|
||||
columns = ["Column A", "Column B"]
|
||||
result = insert_column_headers(data_dict, columns)
|
||||
expected = {
|
||||
"Column A": ["Column 1", "Value 1", "Value 2"],
|
||||
"Column B": ["Column 2", "Value 3", "Value 4"]
|
||||
}
|
||||
assert result == expected
|
||||
|
||||
def test_insert_column_headers_with_extra_columns(self):
|
||||
"""Test inserting column headers when columns list has extra headers."""
|
||||
data_dict = {
|
||||
"Column 1": ["Value 1", "Value 2"],
|
||||
"Column 2": ["Value 3", "Value 4"]
|
||||
}
|
||||
columns = ["Column 1", "Column 2", "Column 3"]
|
||||
result = insert_column_headers(data_dict, columns)
|
||||
expected = {
|
||||
"Column 1": ["Column 1", "Value 1", "Value 2"],
|
||||
"Column 2": ["Column 2", "Value 3", "Value 4"]
|
||||
}
|
||||
assert result == expected
|
||||
|
||||
def test_insert_column_headers_with_missing_columns(self):
|
||||
"""Test inserting column headers when columns list has missing headers."""
|
||||
data_dict = {
|
||||
"Column 1": ["Value 1", "Value 2"],
|
||||
"Column 2": ["Value 3", "Value 4"]
|
||||
}
|
||||
columns = ["Column 1"]
|
||||
result = insert_column_headers(data_dict, columns)
|
||||
expected = {
|
||||
"Column 1": ["Column 1", "Value 1", "Value 2"]
|
||||
}
|
||||
assert result == expected
|
||||
|
||||
def test_insert_column_headers_with_empty_columns(self):
|
||||
"""Test inserting column headers when columns list is empty."""
|
||||
data_dict = {
|
||||
"Column 1": ["Value 1", "Value 2"],
|
||||
"Column 2": ["Value 3", "Value 4"]
|
||||
}
|
||||
columns = []
|
||||
result = insert_column_headers(data_dict, columns)
|
||||
expected = {}
|
||||
assert result == expected
|
||||
|
||||
|
||||
class TestSplitTablesToSeparateSubpages:
|
||||
def test_split_tables_to_separate_subpages_single_table(self):
|
||||
"""Test splitting pages with a single table."""
|
||||
text_dict = {
|
||||
"1": "Some text before the table.\n"
|
||||
"-------Table Start--------\n"
|
||||
"Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n"
|
||||
"-------Table End--------\n"
|
||||
"Some text after the table."
|
||||
}
|
||||
result = split_tables_to_separate_subpages(text_dict)
|
||||
expected = {
|
||||
"1.0": "Some text before the table.\n"
|
||||
"-------Table Start--------\n"
|
||||
"Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n"
|
||||
"-------Table End--------",
|
||||
"1.1": "\nSome text after the table."
|
||||
}
|
||||
assert result == expected
|
||||
|
||||
def test_split_tables_to_separate_subpages_multiple_tables(self):
|
||||
"""Test splitting pages with multiple tables."""
|
||||
text_dict = {
|
||||
"1": "Some text before the tables.\n"
|
||||
"-------Table Start--------\n"
|
||||
"First Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n"
|
||||
"-------Table End--------\n"
|
||||
"Some text between the tables.\n"
|
||||
"-------Table Start--------\n"
|
||||
"Second Table Header {'Column A': ['Value A1', 'Value A2'], 'Column B': ['Value B1', 'Value B2']}\n"
|
||||
"-------Table End--------\n"
|
||||
"Some text after the tables."
|
||||
}
|
||||
result = split_tables_to_separate_subpages(text_dict)
|
||||
expected = {
|
||||
"1.0": "Some text before the tables.\n"
|
||||
"-------Table Start--------\n"
|
||||
"First Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n"
|
||||
"-------Table End--------",
|
||||
"1.1": "\nSome text between the tables.\n"
|
||||
"-------Table Start--------\n"
|
||||
"Second Table Header {'Column A': ['Value A1', 'Value A2'], 'Column B': ['Value B1', 'Value B2']}\n"
|
||||
"-------Table End--------",
|
||||
"1.2": "\nSome text after the tables."
|
||||
}
|
||||
assert result == expected
|
||||
|
||||
def test_split_tables_to_separate_subpages_no_table(self):
|
||||
"""Test splitting pages with no tables."""
|
||||
text_dict = {
|
||||
"1": "This page does not contain a table."
|
||||
}
|
||||
result = split_tables_to_separate_subpages(text_dict)
|
||||
expected = {
|
||||
"1": "This page does not contain a table."
|
||||
}
|
||||
assert result == expected
|
||||
|
||||
def test_split_tables_to_separate_subpages_empty_text(self):
|
||||
"""Test splitting pages with empty text."""
|
||||
text_dict = {
|
||||
"1": ""
|
||||
}
|
||||
result = split_tables_to_separate_subpages(text_dict)
|
||||
expected = {}
|
||||
assert result == expected
|
||||
|
||||
def test_split_tables_to_separate_subpages_mixed_content(self):
|
||||
"""Test splitting pages with mixed content."""
|
||||
text_dict = {
|
||||
"1": "Some text before the table.\n"
|
||||
"-------Table Start--------\n"
|
||||
"Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n"
|
||||
"-------Table End--------\n"
|
||||
"Some text after the table.",
|
||||
"2": "This page does not contain a table.",
|
||||
"3": "Some text before the tables.\n"
|
||||
"-------Table Start--------\n"
|
||||
"First Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n"
|
||||
"-------Table End--------\n"
|
||||
"Some text between the tables.\n"
|
||||
"-------Table Start--------\n"
|
||||
"Second Table Header {'Column A': ['Value A1', 'Value A2'], 'Column B': ['Value B1', 'Value B2']}\n"
|
||||
"-------Table End--------\n"
|
||||
"Some text after the tables."
|
||||
}
|
||||
result = split_tables_to_separate_subpages(text_dict)
|
||||
expected = {
|
||||
"1.0": "Some text before the table.\n"
|
||||
"-------Table Start--------\n"
|
||||
"Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n"
|
||||
"-------Table End--------",
|
||||
"1.1": "\nSome text after the table.",
|
||||
"2": "This page does not contain a table.",
|
||||
"3.0": "Some text before the tables.\n"
|
||||
"-------Table Start--------\n"
|
||||
"First Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n"
|
||||
"-------Table End--------",
|
||||
"3.1": "\nSome text between the tables.\n"
|
||||
"-------Table Start--------\n"
|
||||
"Second Table Header {'Column A': ['Value A1', 'Value A2'], 'Column B': ['Value B1', 'Value B2']}\n"
|
||||
"-------Table End--------",
|
||||
"3.2": "\nSome text after the tables."
|
||||
}
|
||||
assert result == expected
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user