diff --git a/fieldExtraction/poetry.lock b/fieldExtraction/poetry.lock index 16679a4..00bb450 100644 --- a/fieldExtraction/poetry.lock +++ b/fieldExtraction/poetry.lock @@ -288,32 +288,32 @@ css = ["tinycss2 (>=1.1.0,<1.5)"] [[package]] name = "boto3" -version = "1.35.97" +version = "1.36.4" description = "The AWS SDK for Python" optional = false python-versions = ">=3.8" files = [ - {file = "boto3-1.35.97-py3-none-any.whl", hash = "sha256:8e49416216a6e3a62c2a0c44fba4dd2852c85472e7b702516605b1363867d220"}, - {file = "boto3-1.35.97.tar.gz", hash = "sha256:7d398f66a11e67777c189d1f58c0a75d9d60f98d0ee51b8817e828930bf19e4e"}, + {file = "boto3-1.36.4-py3-none-any.whl", hash = "sha256:9f8f699e75ec63fcc98c4dd7290997c7c06c68d3ac8161ad4735fe71f5fe945c"}, + {file = "boto3-1.36.4.tar.gz", hash = "sha256:eeceeb74ef8b65634d358c27aa074917f4449dc828f79301f1075232618eb502"}, ] [package.dependencies] -botocore = ">=1.35.97,<1.36.0" +botocore = ">=1.36.4,<1.37.0" jmespath = ">=0.7.1,<2.0.0" -s3transfer = ">=0.10.0,<0.11.0" +s3transfer = ">=0.11.0,<0.12.0" [package.extras] crt = ["botocore[crt] (>=1.21.0,<2.0a0)"] [[package]] name = "botocore" -version = "1.35.97" +version = "1.36.4" description = "Low-level, data-driven core of boto 3." optional = false python-versions = ">=3.8" files = [ - {file = "botocore-1.35.97-py3-none-any.whl", hash = "sha256:fed4f156b1a9b8ece53738f702ba5851b8c6216b4952de326547f349cc494f14"}, - {file = "botocore-1.35.97.tar.gz", hash = "sha256:88f2fab29192ffe2f2115d5bafbbd823ff4b6eb2774296e03ec8b5b0fe074f61"}, + {file = "botocore-1.36.4-py3-none-any.whl", hash = "sha256:3f183aa7bb0c1ba02171143a05f28a4438abdf89dd6b8c0a7727040375a90520"}, + {file = "botocore-1.36.4.tar.gz", hash = "sha256:ef54f5e3316040b6ff775941e6ed052c3230dda0079d17d9f9e3c757375f2027"}, ] [package.dependencies] @@ -322,7 +322,7 @@ python-dateutil = ">=2.1,<3.0.0" urllib3 = {version = ">=1.25.4,<2.2.0 || >2.2.0,<3", markers = "python_version >= \"3.10\""} [package.extras] -crt = ["awscrt (==0.22.0)"] +crt = ["awscrt (==0.23.4)"] [[package]] name = "certifi" @@ -557,39 +557,113 @@ traitlets = ">=4" [package.extras] test = ["pytest"] +[[package]] +name = "coverage" +version = "7.6.10" +description = "Code coverage measurement for Python" +optional = false +python-versions = ">=3.9" +files = [ + {file = "coverage-7.6.10-cp310-cp310-macosx_10_9_x86_64.whl", hash = "sha256:5c912978f7fbf47ef99cec50c4401340436d200d41d714c7a4766f377c5b7b78"}, + {file = "coverage-7.6.10-cp310-cp310-macosx_11_0_arm64.whl", hash = "sha256:a01ec4af7dfeb96ff0078ad9a48810bb0cc8abcb0115180c6013a6b26237626c"}, + {file = "coverage-7.6.10-cp310-cp310-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:a3b204c11e2b2d883946fe1d97f89403aa1811df28ce0447439178cc7463448a"}, + {file = "coverage-7.6.10-cp310-cp310-manylinux_2_5_i686.manylinux1_i686.manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:32ee6d8491fcfc82652a37109f69dee9a830e9379166cb73c16d8dc5c2915165"}, + {file = "coverage-7.6.10-cp310-cp310-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:675cefc4c06e3b4c876b85bfb7c59c5e2218167bbd4da5075cbe3b5790a28988"}, + {file = "coverage-7.6.10-cp310-cp310-musllinux_1_2_aarch64.whl", hash = "sha256:f4f620668dbc6f5e909a0946a877310fb3d57aea8198bde792aae369ee1c23b5"}, + {file = "coverage-7.6.10-cp310-cp310-musllinux_1_2_i686.whl", hash = "sha256:4eea95ef275de7abaef630c9b2c002ffbc01918b726a39f5a4353916ec72d2f3"}, + {file = "coverage-7.6.10-cp310-cp310-musllinux_1_2_x86_64.whl", hash = "sha256:e2f0280519e42b0a17550072861e0bc8a80a0870de260f9796157d3fca2733c5"}, + {file = "coverage-7.6.10-cp310-cp310-win32.whl", hash = "sha256:bc67deb76bc3717f22e765ab3e07ee9c7a5e26b9019ca19a3b063d9f4b874244"}, + {file = "coverage-7.6.10-cp310-cp310-win_amd64.whl", hash = "sha256:0f460286cb94036455e703c66988851d970fdfd8acc2a1122ab7f4f904e4029e"}, + {file = "coverage-7.6.10-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:ea3c8f04b3e4af80e17bab607c386a830ffc2fb88a5484e1df756478cf70d1d3"}, + {file = "coverage-7.6.10-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:507a20fc863cae1d5720797761b42d2d87a04b3e5aeb682ef3b7332e90598f43"}, + {file = "coverage-7.6.10-cp311-cp311-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:d37a84878285b903c0fe21ac8794c6dab58150e9359f1aaebbeddd6412d53132"}, + {file = "coverage-7.6.10-cp311-cp311-manylinux_2_5_i686.manylinux1_i686.manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:a534738b47b0de1995f85f582d983d94031dffb48ab86c95bdf88dc62212142f"}, + {file = "coverage-7.6.10-cp311-cp311-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:0d7a2bf79378d8fb8afaa994f91bfd8215134f8631d27eba3e0e2c13546ce994"}, + {file = "coverage-7.6.10-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:6713ba4b4ebc330f3def51df1d5d38fad60b66720948112f114968feb52d3f99"}, + {file = "coverage-7.6.10-cp311-cp311-musllinux_1_2_i686.whl", hash = "sha256:ab32947f481f7e8c763fa2c92fd9f44eeb143e7610c4ca9ecd6a36adab4081bd"}, + {file = "coverage-7.6.10-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:7bbd8c8f1b115b892e34ba66a097b915d3871db7ce0e6b9901f462ff3a975377"}, + {file = "coverage-7.6.10-cp311-cp311-win32.whl", hash = "sha256:299e91b274c5c9cdb64cbdf1b3e4a8fe538a7a86acdd08fae52301b28ba297f8"}, + {file = "coverage-7.6.10-cp311-cp311-win_amd64.whl", hash = "sha256:489a01f94aa581dbd961f306e37d75d4ba16104bbfa2b0edb21d29b73be83609"}, + {file = "coverage-7.6.10-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:27c6e64726b307782fa5cbe531e7647aee385a29b2107cd87ba7c0105a5d3853"}, + {file = "coverage-7.6.10-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:c56e097019e72c373bae32d946ecf9858fda841e48d82df7e81c63ac25554078"}, + {file = "coverage-7.6.10-cp312-cp312-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:c7827a5bc7bdb197b9e066cdf650b2887597ad124dd99777332776f7b7c7d0d0"}, + {file = "coverage-7.6.10-cp312-cp312-manylinux_2_5_i686.manylinux1_i686.manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:204a8238afe787323a8b47d8be4df89772d5c1e4651b9ffa808552bdf20e1d50"}, + {file = "coverage-7.6.10-cp312-cp312-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:e67926f51821b8e9deb6426ff3164870976fe414d033ad90ea75e7ed0c2e5022"}, + {file = "coverage-7.6.10-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:e78b270eadb5702938c3dbe9367f878249b5ef9a2fcc5360ac7bff694310d17b"}, + {file = "coverage-7.6.10-cp312-cp312-musllinux_1_2_i686.whl", hash = "sha256:714f942b9c15c3a7a5fe6876ce30af831c2ad4ce902410b7466b662358c852c0"}, + {file = "coverage-7.6.10-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:abb02e2f5a3187b2ac4cd46b8ced85a0858230b577ccb2c62c81482ca7d18852"}, + {file = "coverage-7.6.10-cp312-cp312-win32.whl", hash = "sha256:55b201b97286cf61f5e76063f9e2a1d8d2972fc2fcfd2c1272530172fd28c359"}, + {file = "coverage-7.6.10-cp312-cp312-win_amd64.whl", hash = "sha256:e4ae5ac5e0d1e4edfc9b4b57b4cbecd5bc266a6915c500f358817a8496739247"}, + {file = "coverage-7.6.10-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:05fca8ba6a87aabdd2d30d0b6c838b50510b56cdcfc604d40760dae7153b73d9"}, + {file = "coverage-7.6.10-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:9e80eba8801c386f72e0712a0453431259c45c3249f0009aff537a517b52942b"}, + {file = "coverage-7.6.10-cp313-cp313-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:a372c89c939d57abe09e08c0578c1d212e7a678135d53aa16eec4430adc5e690"}, + {file = "coverage-7.6.10-cp313-cp313-manylinux_2_5_i686.manylinux1_i686.manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:ec22b5e7fe7a0fa8509181c4aac1db48f3dd4d3a566131b313d1efc102892c18"}, + {file = "coverage-7.6.10-cp313-cp313-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:26bcf5c4df41cad1b19c84af71c22cbc9ea9a547fc973f1f2cc9a290002c8b3c"}, + {file = "coverage-7.6.10-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:4e4630c26b6084c9b3cb53b15bd488f30ceb50b73c35c5ad7871b869cb7365fd"}, + {file = "coverage-7.6.10-cp313-cp313-musllinux_1_2_i686.whl", hash = "sha256:2396e8116db77789f819d2bc8a7e200232b7a282c66e0ae2d2cd84581a89757e"}, + {file = "coverage-7.6.10-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:79109c70cc0882e4d2d002fe69a24aa504dec0cc17169b3c7f41a1d341a73694"}, + {file = "coverage-7.6.10-cp313-cp313-win32.whl", hash = "sha256:9e1747bab246d6ff2c4f28b4d186b205adced9f7bd9dc362051cc37c4a0c7bd6"}, + {file = "coverage-7.6.10-cp313-cp313-win_amd64.whl", hash = "sha256:254f1a3b1eef5f7ed23ef265eaa89c65c8c5b6b257327c149db1ca9d4a35f25e"}, + {file = "coverage-7.6.10-cp313-cp313t-macosx_10_13_x86_64.whl", hash = "sha256:2ccf240eb719789cedbb9fd1338055de2761088202a9a0b73032857e53f612fe"}, + {file = "coverage-7.6.10-cp313-cp313t-macosx_11_0_arm64.whl", hash = "sha256:0c807ca74d5a5e64427c8805de15b9ca140bba13572d6d74e262f46f50b13273"}, + {file = "coverage-7.6.10-cp313-cp313t-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:2bcfa46d7709b5a7ffe089075799b902020b62e7ee56ebaed2f4bdac04c508d8"}, + {file = "coverage-7.6.10-cp313-cp313t-manylinux_2_5_i686.manylinux1_i686.manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:4e0de1e902669dccbf80b0415fb6b43d27edca2fbd48c74da378923b05316098"}, + {file = "coverage-7.6.10-cp313-cp313t-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:3f7b444c42bbc533aaae6b5a2166fd1a797cdb5eb58ee51a92bee1eb94a1e1cb"}, + {file = "coverage-7.6.10-cp313-cp313t-musllinux_1_2_aarch64.whl", hash = "sha256:b330368cb99ef72fcd2dc3ed260adf67b31499584dc8a20225e85bfe6f6cfed0"}, + {file = "coverage-7.6.10-cp313-cp313t-musllinux_1_2_i686.whl", hash = "sha256:9a7cfb50515f87f7ed30bc882f68812fd98bc2852957df69f3003d22a2aa0abf"}, + {file = "coverage-7.6.10-cp313-cp313t-musllinux_1_2_x86_64.whl", hash = "sha256:6f93531882a5f68c28090f901b1d135de61b56331bba82028489bc51bdd818d2"}, + {file = "coverage-7.6.10-cp313-cp313t-win32.whl", hash = "sha256:89d76815a26197c858f53c7f6a656686ec392b25991f9e409bcef020cd532312"}, + {file = "coverage-7.6.10-cp313-cp313t-win_amd64.whl", hash = "sha256:54a5f0f43950a36312155dae55c505a76cd7f2b12d26abeebbe7a0b36dbc868d"}, + {file = "coverage-7.6.10-cp39-cp39-macosx_10_9_x86_64.whl", hash = "sha256:656c82b8a0ead8bba147de9a89bda95064874c91a3ed43a00e687f23cc19d53a"}, + {file = "coverage-7.6.10-cp39-cp39-macosx_11_0_arm64.whl", hash = "sha256:ccc2b70a7ed475c68ceb548bf69cec1e27305c1c2606a5eb7c3afff56a1b3b27"}, + {file = "coverage-7.6.10-cp39-cp39-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:a5e37dc41d57ceba70956fa2fc5b63c26dba863c946ace9705f8eca99daecdc4"}, + {file = "coverage-7.6.10-cp39-cp39-manylinux_2_5_i686.manylinux1_i686.manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:0aa9692b4fdd83a4647eeb7db46410ea1322b5ed94cd1715ef09d1d5922ba87f"}, + {file = "coverage-7.6.10-cp39-cp39-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:aa744da1820678b475e4ba3dfd994c321c5b13381d1041fe9c608620e6676e25"}, + {file = "coverage-7.6.10-cp39-cp39-musllinux_1_2_aarch64.whl", hash = "sha256:c0b1818063dc9e9d838c09e3a473c1422f517889436dd980f5d721899e66f315"}, + {file = "coverage-7.6.10-cp39-cp39-musllinux_1_2_i686.whl", hash = "sha256:59af35558ba08b758aec4d56182b222976330ef8d2feacbb93964f576a7e7a90"}, + {file = "coverage-7.6.10-cp39-cp39-musllinux_1_2_x86_64.whl", hash = "sha256:7ed2f37cfce1ce101e6dffdfd1c99e729dd2ffc291d02d3e2d0af8b53d13840d"}, + {file = "coverage-7.6.10-cp39-cp39-win32.whl", hash = "sha256:4bcc276261505d82f0ad426870c3b12cb177752834a633e737ec5ee79bbdff18"}, + {file = "coverage-7.6.10-cp39-cp39-win_amd64.whl", hash = "sha256:457574f4599d2b00f7f637a0700a6422243b3565509457b2dbd3f50703e11f59"}, + {file = "coverage-7.6.10-pp39.pp310-none-any.whl", hash = "sha256:fd34e7b3405f0cc7ab03d54a334c17a9e802897580d964bd8c2001f4b9fd488f"}, + {file = "coverage-7.6.10.tar.gz", hash = "sha256:7fb105327c8f8f0682e29843e2ff96af9dcbe5bab8eeb4b398c6a33a16d80a23"}, +] + +[package.extras] +toml = ["tomli"] + [[package]] name = "debugpy" -version = "1.8.11" +version = "1.8.12" description = "An implementation of the Debug Adapter Protocol for Python" optional = false python-versions = ">=3.8" files = [ - {file = "debugpy-1.8.11-cp310-cp310-macosx_14_0_x86_64.whl", hash = "sha256:2b26fefc4e31ff85593d68b9022e35e8925714a10ab4858fb1b577a8a48cb8cd"}, - {file = "debugpy-1.8.11-cp310-cp310-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:61bc8b3b265e6949855300e84dc93d02d7a3a637f2aec6d382afd4ceb9120c9f"}, - {file = "debugpy-1.8.11-cp310-cp310-win32.whl", hash = "sha256:c928bbf47f65288574b78518449edaa46c82572d340e2750889bbf8cd92f3737"}, - {file = "debugpy-1.8.11-cp310-cp310-win_amd64.whl", hash = "sha256:8da1db4ca4f22583e834dcabdc7832e56fe16275253ee53ba66627b86e304da1"}, - {file = "debugpy-1.8.11-cp311-cp311-macosx_14_0_universal2.whl", hash = "sha256:85de8474ad53ad546ff1c7c7c89230db215b9b8a02754d41cb5a76f70d0be296"}, - {file = "debugpy-1.8.11-cp311-cp311-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:8ffc382e4afa4aee367bf413f55ed17bd91b191dcaf979890af239dda435f2a1"}, - {file = "debugpy-1.8.11-cp311-cp311-win32.whl", hash = "sha256:40499a9979c55f72f4eb2fc38695419546b62594f8af194b879d2a18439c97a9"}, - {file = "debugpy-1.8.11-cp311-cp311-win_amd64.whl", hash = "sha256:987bce16e86efa86f747d5151c54e91b3c1e36acc03ce1ddb50f9d09d16ded0e"}, - {file = "debugpy-1.8.11-cp312-cp312-macosx_14_0_universal2.whl", hash = "sha256:84e511a7545d11683d32cdb8f809ef63fc17ea2a00455cc62d0a4dbb4ed1c308"}, - {file = "debugpy-1.8.11-cp312-cp312-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:ce291a5aca4985d82875d6779f61375e959208cdf09fcec40001e65fb0a54768"}, - {file = "debugpy-1.8.11-cp312-cp312-win32.whl", hash = "sha256:28e45b3f827d3bf2592f3cf7ae63282e859f3259db44ed2b129093ca0ac7940b"}, - {file = "debugpy-1.8.11-cp312-cp312-win_amd64.whl", hash = "sha256:44b1b8e6253bceada11f714acf4309ffb98bfa9ac55e4fce14f9e5d4484287a1"}, - {file = "debugpy-1.8.11-cp313-cp313-macosx_14_0_universal2.whl", hash = "sha256:8988f7163e4381b0da7696f37eec7aca19deb02e500245df68a7159739bbd0d3"}, - {file = "debugpy-1.8.11-cp313-cp313-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:6c1f6a173d1140e557347419767d2b14ac1c9cd847e0b4c5444c7f3144697e4e"}, - {file = "debugpy-1.8.11-cp313-cp313-win32.whl", hash = "sha256:bb3b15e25891f38da3ca0740271e63ab9db61f41d4d8541745cfc1824252cb28"}, - {file = "debugpy-1.8.11-cp313-cp313-win_amd64.whl", hash = "sha256:d8768edcbeb34da9e11bcb8b5c2e0958d25218df7a6e56adf415ef262cd7b6d1"}, - {file = "debugpy-1.8.11-cp38-cp38-macosx_14_0_x86_64.whl", hash = "sha256:ad7efe588c8f5cf940f40c3de0cd683cc5b76819446abaa50dc0829a30c094db"}, - {file = "debugpy-1.8.11-cp38-cp38-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:189058d03a40103a57144752652b3ab08ff02b7595d0ce1f651b9acc3a3a35a0"}, - {file = "debugpy-1.8.11-cp38-cp38-win32.whl", hash = "sha256:32db46ba45849daed7ccf3f2e26f7a386867b077f39b2a974bb5c4c2c3b0a280"}, - {file = "debugpy-1.8.11-cp38-cp38-win_amd64.whl", hash = "sha256:116bf8342062246ca749013df4f6ea106f23bc159305843491f64672a55af2e5"}, - {file = "debugpy-1.8.11-cp39-cp39-macosx_14_0_x86_64.whl", hash = "sha256:654130ca6ad5de73d978057eaf9e582244ff72d4574b3e106fb8d3d2a0d32458"}, - {file = "debugpy-1.8.11-cp39-cp39-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:23dc34c5e03b0212fa3c49a874df2b8b1b8fda95160bd79c01eb3ab51ea8d851"}, - {file = "debugpy-1.8.11-cp39-cp39-win32.whl", hash = "sha256:52d8a3166c9f2815bfae05f386114b0b2d274456980d41f320299a8d9a5615a7"}, - {file = "debugpy-1.8.11-cp39-cp39-win_amd64.whl", hash = "sha256:52c3cf9ecda273a19cc092961ee34eb9ba8687d67ba34cc7b79a521c1c64c4c0"}, - {file = "debugpy-1.8.11-py2.py3-none-any.whl", hash = "sha256:0e22f846f4211383e6a416d04b4c13ed174d24cc5d43f5fd52e7821d0ebc8920"}, - {file = "debugpy-1.8.11.tar.gz", hash = "sha256:6ad2688b69235c43b020e04fecccdf6a96c8943ca9c2fb340b8adc103c655e57"}, + {file = "debugpy-1.8.12-cp310-cp310-macosx_14_0_x86_64.whl", hash = "sha256:a2ba7ffe58efeae5b8fad1165357edfe01464f9aef25e814e891ec690e7dd82a"}, + {file = "debugpy-1.8.12-cp310-cp310-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:cbbd4149c4fc5e7d508ece083e78c17442ee13b0e69bfa6bd63003e486770f45"}, + {file = "debugpy-1.8.12-cp310-cp310-win32.whl", hash = "sha256:b202f591204023b3ce62ff9a47baa555dc00bb092219abf5caf0e3718ac20e7c"}, + {file = "debugpy-1.8.12-cp310-cp310-win_amd64.whl", hash = "sha256:9649eced17a98ce816756ce50433b2dd85dfa7bc92ceb60579d68c053f98dff9"}, + {file = "debugpy-1.8.12-cp311-cp311-macosx_14_0_universal2.whl", hash = "sha256:36f4829839ef0afdfdd208bb54f4c3d0eea86106d719811681a8627ae2e53dd5"}, + {file = "debugpy-1.8.12-cp311-cp311-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:a28ed481d530e3138553be60991d2d61103ce6da254e51547b79549675f539b7"}, + {file = "debugpy-1.8.12-cp311-cp311-win32.whl", hash = "sha256:4ad9a94d8f5c9b954e0e3b137cc64ef3f579d0df3c3698fe9c3734ee397e4abb"}, + {file = "debugpy-1.8.12-cp311-cp311-win_amd64.whl", hash = "sha256:4703575b78dd697b294f8c65588dc86874ed787b7348c65da70cfc885efdf1e1"}, + {file = "debugpy-1.8.12-cp312-cp312-macosx_14_0_universal2.whl", hash = "sha256:7e94b643b19e8feb5215fa508aee531387494bf668b2eca27fa769ea11d9f498"}, + {file = "debugpy-1.8.12-cp312-cp312-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:086b32e233e89a2740c1615c2f775c34ae951508b28b308681dbbb87bba97d06"}, + {file = "debugpy-1.8.12-cp312-cp312-win32.whl", hash = "sha256:2ae5df899732a6051b49ea2632a9ea67f929604fd2b036613a9f12bc3163b92d"}, + {file = "debugpy-1.8.12-cp312-cp312-win_amd64.whl", hash = "sha256:39dfbb6fa09f12fae32639e3286112fc35ae976114f1f3d37375f3130a820969"}, + {file = "debugpy-1.8.12-cp313-cp313-macosx_14_0_universal2.whl", hash = "sha256:696d8ae4dff4cbd06bf6b10d671e088b66669f110c7c4e18a44c43cf75ce966f"}, + {file = "debugpy-1.8.12-cp313-cp313-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:898fba72b81a654e74412a67c7e0a81e89723cfe2a3ea6fcd3feaa3395138ca9"}, + {file = "debugpy-1.8.12-cp313-cp313-win32.whl", hash = "sha256:22a11c493c70413a01ed03f01c3c3a2fc4478fc6ee186e340487b2edcd6f4180"}, + {file = "debugpy-1.8.12-cp313-cp313-win_amd64.whl", hash = "sha256:fdb3c6d342825ea10b90e43d7f20f01535a72b3a1997850c0c3cefa5c27a4a2c"}, + {file = "debugpy-1.8.12-cp38-cp38-macosx_14_0_x86_64.whl", hash = "sha256:b0232cd42506d0c94f9328aaf0d1d0785f90f87ae72d9759df7e5051be039738"}, + {file = "debugpy-1.8.12-cp38-cp38-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:9af40506a59450f1315168d47a970db1a65aaab5df3833ac389d2899a5d63b3f"}, + {file = "debugpy-1.8.12-cp38-cp38-win32.whl", hash = "sha256:5cc45235fefac57f52680902b7d197fb2f3650112379a6fa9aa1b1c1d3ed3f02"}, + {file = "debugpy-1.8.12-cp38-cp38-win_amd64.whl", hash = "sha256:557cc55b51ab2f3371e238804ffc8510b6ef087673303890f57a24195d096e61"}, + {file = "debugpy-1.8.12-cp39-cp39-macosx_14_0_x86_64.whl", hash = "sha256:b5c6c967d02fee30e157ab5227706f965d5c37679c687b1e7bbc5d9e7128bd41"}, + {file = "debugpy-1.8.12-cp39-cp39-manylinux_2_5_x86_64.manylinux1_x86_64.manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:88a77f422f31f170c4b7e9ca58eae2a6c8e04da54121900651dfa8e66c29901a"}, + {file = "debugpy-1.8.12-cp39-cp39-win32.whl", hash = "sha256:a4042edef80364239f5b7b5764e55fd3ffd40c32cf6753da9bda4ff0ac466018"}, + {file = "debugpy-1.8.12-cp39-cp39-win_amd64.whl", hash = "sha256:f30b03b0f27608a0b26c75f0bb8a880c752c0e0b01090551b9d87c7d783e2069"}, + {file = "debugpy-1.8.12-py2.py3-none-any.whl", hash = "sha256:274b6a2040349b5c9864e475284bce5bb062e63dce368a394b8cc865ae3b00c6"}, + {file = "debugpy-1.8.12.tar.gz", hash = "sha256:646530b04f45c830ceae8e491ca1c9320a2d2f0efea3141487c82130aba70dce"}, ] [[package]] @@ -638,13 +712,13 @@ files = [ [[package]] name = "executing" -version = "2.1.0" +version = "2.2.0" description = "Get the currently executing AST node of a frame, and other information" optional = false python-versions = ">=3.8" files = [ - {file = "executing-2.1.0-py2.py3-none-any.whl", hash = "sha256:8d63781349375b5ebccc3142f4b30350c0cd9c79f921cde38be2be4637e98eaf"}, - {file = "executing-2.1.0.tar.gz", hash = "sha256:8ea27ddd260da8150fa5a708269c4a10e76161e2496ec3e587da9e3c0fe4b9ab"}, + {file = "executing-2.2.0-py2.py3-none-any.whl", hash = "sha256:11387150cad388d62750327a53d3339fad4888b39a6fe233c3afbb54ecffd3aa"}, + {file = "executing-2.2.0.tar.gz", hash = "sha256:5d108c028108fe2551d1a7b2e8b713341e2cb4fc0aa7dcf966fa4327a5226755"}, ] [package.extras] @@ -666,18 +740,18 @@ devel = ["colorama", "json-spec", "jsonschema", "pylint", "pytest", "pytest-benc [[package]] name = "filelock" -version = "3.16.1" +version = "3.17.0" description = "A platform independent file lock." optional = false -python-versions = ">=3.8" +python-versions = ">=3.9" files = [ - {file = "filelock-3.16.1-py3-none-any.whl", hash = "sha256:2082e5703d51fbf98ea75855d9d5527e33d8ff23099bec374a134febee6946b0"}, - {file = "filelock-3.16.1.tar.gz", hash = "sha256:c249fbfcd5db47e5e2d6d62198e565475ee65e4831e2561c8e313fa7eb961435"}, + {file = "filelock-3.17.0-py3-none-any.whl", hash = "sha256:533dc2f7ba78dc2f0f531fc6c4940addf7b70a481e269a5a3b93be94ffbe8338"}, + {file = "filelock-3.17.0.tar.gz", hash = "sha256:ee4e77401ef576ebb38cd7f13b9b28893194acc20a8e68e18730ba9c0e54660e"}, ] [package.extras] -docs = ["furo (>=2024.8.6)", "sphinx (>=8.0.2)", "sphinx-autodoc-typehints (>=2.4.1)"] -testing = ["covdefaults (>=2.3)", "coverage (>=7.6.1)", "diff-cover (>=9.2)", "pytest (>=8.3.3)", "pytest-asyncio (>=0.24)", "pytest-cov (>=5)", "pytest-mock (>=3.14)", "pytest-timeout (>=2.3.1)", "virtualenv (>=20.26.4)"] +docs = ["furo (>=2024.8.6)", "sphinx (>=8.1.3)", "sphinx-autodoc-typehints (>=3)"] +testing = ["covdefaults (>=2.3)", "coverage (>=7.6.10)", "diff-cover (>=9.2.1)", "pytest (>=8.3.4)", "pytest-asyncio (>=0.25.2)", "pytest-cov (>=6)", "pytest-mock (>=3.14)", "pytest-timeout (>=2.3.1)", "virtualenv (>=20.28.1)"] typing = ["typing-extensions (>=4.12.2)"] [[package]] @@ -1718,66 +1792,66 @@ test = ["pytest", "pytest-console-scripts", "pytest-jupyter", "pytest-tornasync" [[package]] name = "numpy" -version = "2.2.1" +version = "2.2.2" description = "Fundamental package for array computing in Python" optional = false python-versions = ">=3.10" files = [ - {file = "numpy-2.2.1-cp310-cp310-macosx_10_9_x86_64.whl", hash = "sha256:5edb4e4caf751c1518e6a26a83501fda79bff41cc59dac48d70e6d65d4ec4440"}, - {file = "numpy-2.2.1-cp310-cp310-macosx_11_0_arm64.whl", hash = "sha256:aa3017c40d513ccac9621a2364f939d39e550c542eb2a894b4c8da92b38896ab"}, - {file = "numpy-2.2.1-cp310-cp310-macosx_14_0_arm64.whl", hash = "sha256:61048b4a49b1c93fe13426e04e04fdf5a03f456616f6e98c7576144677598675"}, - {file = "numpy-2.2.1-cp310-cp310-macosx_14_0_x86_64.whl", hash = "sha256:7671dc19c7019103ca44e8d94917eba8534c76133523ca8406822efdd19c9308"}, - {file = "numpy-2.2.1-cp310-cp310-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:4250888bcb96617e00bfa28ac24850a83c9f3a16db471eca2ee1f1714df0f957"}, - {file = "numpy-2.2.1-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:a7746f235c47abc72b102d3bce9977714c2444bdfaea7888d241b4c4bb6a78bf"}, - {file = "numpy-2.2.1-cp310-cp310-musllinux_1_2_aarch64.whl", hash = "sha256:059e6a747ae84fce488c3ee397cee7e5f905fd1bda5fb18c66bc41807ff119b2"}, - {file = "numpy-2.2.1-cp310-cp310-musllinux_1_2_x86_64.whl", hash = "sha256:f62aa6ee4eb43b024b0e5a01cf65a0bb078ef8c395e8713c6e8a12a697144528"}, - {file = "numpy-2.2.1-cp310-cp310-win32.whl", hash = "sha256:48fd472630715e1c1c89bf1feab55c29098cb403cc184b4859f9c86d4fcb6a95"}, - {file = "numpy-2.2.1-cp310-cp310-win_amd64.whl", hash = "sha256:b541032178a718c165a49638d28272b771053f628382d5e9d1c93df23ff58dbf"}, - {file = "numpy-2.2.1-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:40f9e544c1c56ba8f1cf7686a8c9b5bb249e665d40d626a23899ba6d5d9e1484"}, - {file = "numpy-2.2.1-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:f9b57eaa3b0cd8db52049ed0330747b0364e899e8a606a624813452b8203d5f7"}, - {file = "numpy-2.2.1-cp311-cp311-macosx_14_0_arm64.whl", hash = "sha256:bc8a37ad5b22c08e2dbd27df2b3ef7e5c0864235805b1e718a235bcb200cf1cb"}, - {file = "numpy-2.2.1-cp311-cp311-macosx_14_0_x86_64.whl", hash = "sha256:9036d6365d13b6cbe8f27a0eaf73ddcc070cae584e5ff94bb45e3e9d729feab5"}, - {file = "numpy-2.2.1-cp311-cp311-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:51faf345324db860b515d3f364eaa93d0e0551a88d6218a7d61286554d190d73"}, - {file = "numpy-2.2.1-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:38efc1e56b73cc9b182fe55e56e63b044dd26a72128fd2fbd502f75555d92591"}, - {file = "numpy-2.2.1-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:31b89fa67a8042e96715c68e071a1200c4e172f93b0fbe01a14c0ff3ff820fc8"}, - {file = "numpy-2.2.1-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:4c86e2a209199ead7ee0af65e1d9992d1dce7e1f63c4b9a616500f93820658d0"}, - {file = "numpy-2.2.1-cp311-cp311-win32.whl", hash = "sha256:b34d87e8a3090ea626003f87f9392b3929a7bbf4104a05b6667348b6bd4bf1cd"}, - {file = "numpy-2.2.1-cp311-cp311-win_amd64.whl", hash = "sha256:360137f8fb1b753c5cde3ac388597ad680eccbbbb3865ab65efea062c4a1fd16"}, - {file = "numpy-2.2.1-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:694f9e921a0c8f252980e85bce61ebbd07ed2b7d4fa72d0e4246f2f8aa6642ab"}, - {file = "numpy-2.2.1-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:3683a8d166f2692664262fd4900f207791d005fb088d7fdb973cc8d663626faa"}, - {file = "numpy-2.2.1-cp312-cp312-macosx_14_0_arm64.whl", hash = "sha256:780077d95eafc2ccc3ced969db22377b3864e5b9a0ea5eb347cc93b3ea900315"}, - {file = "numpy-2.2.1-cp312-cp312-macosx_14_0_x86_64.whl", hash = "sha256:55ba24ebe208344aa7a00e4482f65742969a039c2acfcb910bc6fcd776eb4355"}, - {file = "numpy-2.2.1-cp312-cp312-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:9b1d07b53b78bf84a96898c1bc139ad7f10fda7423f5fd158fd0f47ec5e01ac7"}, - {file = "numpy-2.2.1-cp312-cp312-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:5062dc1a4e32a10dc2b8b13cedd58988261416e811c1dc4dbdea4f57eea61b0d"}, - {file = "numpy-2.2.1-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:fce4f615f8ca31b2e61aa0eb5865a21e14f5629515c9151850aa936c02a1ee51"}, - {file = "numpy-2.2.1-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:67d4cda6fa6ffa073b08c8372aa5fa767ceb10c9a0587c707505a6d426f4e046"}, - {file = "numpy-2.2.1-cp312-cp312-win32.whl", hash = "sha256:32cb94448be47c500d2c7a95f93e2f21a01f1fd05dd2beea1ccd049bb6001cd2"}, - {file = "numpy-2.2.1-cp312-cp312-win_amd64.whl", hash = "sha256:ba5511d8f31c033a5fcbda22dd5c813630af98c70b2661f2d2c654ae3cdfcfc8"}, - {file = "numpy-2.2.1-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:f1d09e520217618e76396377c81fba6f290d5f926f50c35f3a5f72b01a0da780"}, - {file = "numpy-2.2.1-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:3ecc47cd7f6ea0336042be87d9e7da378e5c7e9b3c8ad0f7c966f714fc10d821"}, - {file = "numpy-2.2.1-cp313-cp313-macosx_14_0_arm64.whl", hash = "sha256:f419290bc8968a46c4933158c91a0012b7a99bb2e465d5ef5293879742f8797e"}, - {file = "numpy-2.2.1-cp313-cp313-macosx_14_0_x86_64.whl", hash = "sha256:5b6c390bfaef8c45a260554888966618328d30e72173697e5cabe6b285fb2348"}, - {file = "numpy-2.2.1-cp313-cp313-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:526fc406ab991a340744aad7e25251dd47a6720a685fa3331e5c59fef5282a59"}, - {file = "numpy-2.2.1-cp313-cp313-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:f74e6fdeb9a265624ec3a3918430205dff1df7e95a230779746a6af78bc615af"}, - {file = "numpy-2.2.1-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:53c09385ff0b72ba79d8715683c1168c12e0b6e84fb0372e97553d1ea91efe51"}, - {file = "numpy-2.2.1-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:f3eac17d9ec51be534685ba877b6ab5edc3ab7ec95c8f163e5d7b39859524716"}, - {file = "numpy-2.2.1-cp313-cp313-win32.whl", hash = "sha256:9ad014faa93dbb52c80d8f4d3dcf855865c876c9660cb9bd7553843dd03a4b1e"}, - {file = "numpy-2.2.1-cp313-cp313-win_amd64.whl", hash = "sha256:164a829b6aacf79ca47ba4814b130c4020b202522a93d7bff2202bfb33b61c60"}, - {file = "numpy-2.2.1-cp313-cp313t-macosx_10_13_x86_64.whl", hash = "sha256:4dfda918a13cc4f81e9118dea249e192ab167a0bb1966272d5503e39234d694e"}, - {file = "numpy-2.2.1-cp313-cp313t-macosx_11_0_arm64.whl", hash = "sha256:733585f9f4b62e9b3528dd1070ec4f52b8acf64215b60a845fa13ebd73cd0712"}, - {file = "numpy-2.2.1-cp313-cp313t-macosx_14_0_arm64.whl", hash = "sha256:89b16a18e7bba224ce5114db863e7029803c179979e1af6ad6a6b11f70545008"}, - {file = "numpy-2.2.1-cp313-cp313t-macosx_14_0_x86_64.whl", hash = "sha256:676f4eebf6b2d430300f1f4f4c2461685f8269f94c89698d832cdf9277f30b84"}, - {file = "numpy-2.2.1-cp313-cp313t-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:27f5cdf9f493b35f7e41e8368e7d7b4bbafaf9660cba53fb21d2cd174ec09631"}, - {file = "numpy-2.2.1-cp313-cp313t-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:c1ad395cf254c4fbb5b2132fee391f361a6e8c1adbd28f2cd8e79308a615fe9d"}, - {file = "numpy-2.2.1-cp313-cp313t-musllinux_1_2_aarch64.whl", hash = "sha256:08ef779aed40dbc52729d6ffe7dd51df85796a702afbf68a4f4e41fafdc8bda5"}, - {file = "numpy-2.2.1-cp313-cp313t-musllinux_1_2_x86_64.whl", hash = "sha256:26c9c4382b19fcfbbed3238a14abf7ff223890ea1936b8890f058e7ba35e8d71"}, - {file = "numpy-2.2.1-cp313-cp313t-win32.whl", hash = "sha256:93cf4e045bae74c90ca833cba583c14b62cb4ba2cba0abd2b141ab52548247e2"}, - {file = "numpy-2.2.1-cp313-cp313t-win_amd64.whl", hash = "sha256:bff7d8ec20f5f42607599f9994770fa65d76edca264a87b5e4ea5629bce12268"}, - {file = "numpy-2.2.1-pp310-pypy310_pp73-macosx_10_15_x86_64.whl", hash = "sha256:7ba9cc93a91d86365a5d270dee221fdc04fb68d7478e6bf6af650de78a8339e3"}, - {file = "numpy-2.2.1-pp310-pypy310_pp73-macosx_14_0_x86_64.whl", hash = "sha256:3d03883435a19794e41f147612a77a8f56d4e52822337844fff3d4040a142964"}, - {file = "numpy-2.2.1-pp310-pypy310_pp73-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:4511d9e6071452b944207c8ce46ad2f897307910b402ea5fa975da32e0102800"}, - {file = "numpy-2.2.1-pp310-pypy310_pp73-win_amd64.whl", hash = "sha256:5c5cc0cbabe9452038ed984d05ac87910f89370b9242371bd9079cb4af61811e"}, - {file = "numpy-2.2.1.tar.gz", hash = "sha256:45681fd7128c8ad1c379f0ca0776a8b0c6583d2f69889ddac01559dfe4390918"}, + {file = "numpy-2.2.2-cp310-cp310-macosx_10_9_x86_64.whl", hash = "sha256:7079129b64cb78bdc8d611d1fd7e8002c0a2565da6a47c4df8062349fee90e3e"}, + {file = "numpy-2.2.2-cp310-cp310-macosx_11_0_arm64.whl", hash = "sha256:2ec6c689c61df613b783aeb21f945c4cbe6c51c28cb70aae8430577ab39f163e"}, + {file = "numpy-2.2.2-cp310-cp310-macosx_14_0_arm64.whl", hash = "sha256:40c7ff5da22cd391944a28c6a9c638a5eef77fcf71d6e3a79e1d9d9e82752715"}, + {file = "numpy-2.2.2-cp310-cp310-macosx_14_0_x86_64.whl", hash = "sha256:995f9e8181723852ca458e22de5d9b7d3ba4da3f11cc1cb113f093b271d7965a"}, + {file = "numpy-2.2.2-cp310-cp310-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:b78ea78450fd96a498f50ee096f69c75379af5138f7881a51355ab0e11286c97"}, + {file = "numpy-2.2.2-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:3fbe72d347fbc59f94124125e73fc4976a06927ebc503ec5afbfb35f193cd957"}, + {file = "numpy-2.2.2-cp310-cp310-musllinux_1_2_aarch64.whl", hash = "sha256:8e6da5cffbbe571f93588f562ed130ea63ee206d12851b60819512dd3e1ba50d"}, + {file = "numpy-2.2.2-cp310-cp310-musllinux_1_2_x86_64.whl", hash = "sha256:09d6a2032faf25e8d0cadde7fd6145118ac55d2740132c1d845f98721b5ebcfd"}, + {file = "numpy-2.2.2-cp310-cp310-win32.whl", hash = "sha256:159ff6ee4c4a36a23fe01b7c3d07bd8c14cc433d9720f977fcd52c13c0098160"}, + {file = "numpy-2.2.2-cp310-cp310-win_amd64.whl", hash = "sha256:64bd6e1762cd7f0986a740fee4dff927b9ec2c5e4d9a28d056eb17d332158014"}, + {file = "numpy-2.2.2-cp311-cp311-macosx_10_9_x86_64.whl", hash = "sha256:642199e98af1bd2b6aeb8ecf726972d238c9877b0f6e8221ee5ab945ec8a2189"}, + {file = "numpy-2.2.2-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:6d9fc9d812c81e6168b6d405bf00b8d6739a7f72ef22a9214c4241e0dc70b323"}, + {file = "numpy-2.2.2-cp311-cp311-macosx_14_0_arm64.whl", hash = "sha256:c7d1fd447e33ee20c1f33f2c8e6634211124a9aabde3c617687d8b739aa69eac"}, + {file = "numpy-2.2.2-cp311-cp311-macosx_14_0_x86_64.whl", hash = "sha256:451e854cfae0febe723077bd0cf0a4302a5d84ff25f0bfece8f29206c7bed02e"}, + {file = "numpy-2.2.2-cp311-cp311-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:bd249bc894af67cbd8bad2c22e7cbcd46cf87ddfca1f1289d1e7e54868cc785c"}, + {file = "numpy-2.2.2-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:02935e2c3c0c6cbe9c7955a8efa8908dd4221d7755644c59d1bba28b94fd334f"}, + {file = "numpy-2.2.2-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:a972cec723e0563aa0823ee2ab1df0cb196ed0778f173b381c871a03719d4826"}, + {file = "numpy-2.2.2-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:d6d6a0910c3b4368d89dde073e630882cdb266755565155bc33520283b2d9df8"}, + {file = "numpy-2.2.2-cp311-cp311-win32.whl", hash = "sha256:860fd59990c37c3ef913c3ae390b3929d005243acca1a86facb0773e2d8d9e50"}, + {file = "numpy-2.2.2-cp311-cp311-win_amd64.whl", hash = "sha256:da1eeb460ecce8d5b8608826595c777728cdf28ce7b5a5a8c8ac8d949beadcf2"}, + {file = "numpy-2.2.2-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:ac9bea18d6d58a995fac1b2cb4488e17eceeac413af014b1dd26170b766d8467"}, + {file = "numpy-2.2.2-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:23ae9f0c2d889b7b2d88a3791f6c09e2ef827c2446f1c4a3e3e76328ee4afd9a"}, + {file = "numpy-2.2.2-cp312-cp312-macosx_14_0_arm64.whl", hash = "sha256:3074634ea4d6df66be04f6728ee1d173cfded75d002c75fac79503a880bf3825"}, + {file = "numpy-2.2.2-cp312-cp312-macosx_14_0_x86_64.whl", hash = "sha256:8ec0636d3f7d68520afc6ac2dc4b8341ddb725039de042faf0e311599f54eb37"}, + {file = "numpy-2.2.2-cp312-cp312-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:2ffbb1acd69fdf8e89dd60ef6182ca90a743620957afb7066385a7bbe88dc748"}, + {file = "numpy-2.2.2-cp312-cp312-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:0349b025e15ea9d05c3d63f9657707a4e1d471128a3b1d876c095f328f8ff7f0"}, + {file = "numpy-2.2.2-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:463247edcee4a5537841d5350bc87fe8e92d7dd0e8c71c995d2c6eecb8208278"}, + {file = "numpy-2.2.2-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:9dd47ff0cb2a656ad69c38da850df3454da88ee9a6fde0ba79acceee0e79daba"}, + {file = "numpy-2.2.2-cp312-cp312-win32.whl", hash = "sha256:4525b88c11906d5ab1b0ec1f290996c0020dd318af8b49acaa46f198b1ffc283"}, + {file = "numpy-2.2.2-cp312-cp312-win_amd64.whl", hash = "sha256:5acea83b801e98541619af398cc0109ff48016955cc0818f478ee9ef1c5c3dcb"}, + {file = "numpy-2.2.2-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:b208cfd4f5fe34e1535c08983a1a6803fdbc7a1e86cf13dd0c61de0b51a0aadc"}, + {file = "numpy-2.2.2-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:d0bbe7dd86dca64854f4b6ce2ea5c60b51e36dfd597300057cf473d3615f2369"}, + {file = "numpy-2.2.2-cp313-cp313-macosx_14_0_arm64.whl", hash = "sha256:22ea3bb552ade325530e72a0c557cdf2dea8914d3a5e1fecf58fa5dbcc6f43cd"}, + {file = "numpy-2.2.2-cp313-cp313-macosx_14_0_x86_64.whl", hash = "sha256:128c41c085cab8a85dc29e66ed88c05613dccf6bc28b3866cd16050a2f5448be"}, + {file = "numpy-2.2.2-cp313-cp313-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:250c16b277e3b809ac20d1f590716597481061b514223c7badb7a0f9993c7f84"}, + {file = "numpy-2.2.2-cp313-cp313-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:e0c8854b09bc4de7b041148d8550d3bd712b5c21ff6a8ed308085f190235d7ff"}, + {file = "numpy-2.2.2-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:b6fb9c32a91ec32a689ec6410def76443e3c750e7cfc3fb2206b985ffb2b85f0"}, + {file = "numpy-2.2.2-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:57b4012e04cc12b78590a334907e01b3a85efb2107df2b8733ff1ed05fce71de"}, + {file = "numpy-2.2.2-cp313-cp313-win32.whl", hash = "sha256:4dbd80e453bd34bd003b16bd802fac70ad76bd463f81f0c518d1245b1c55e3d9"}, + {file = "numpy-2.2.2-cp313-cp313-win_amd64.whl", hash = "sha256:5a8c863ceacae696aff37d1fd636121f1a512117652e5dfb86031c8d84836369"}, + {file = "numpy-2.2.2-cp313-cp313t-macosx_10_13_x86_64.whl", hash = "sha256:b3482cb7b3325faa5f6bc179649406058253d91ceda359c104dac0ad320e1391"}, + {file = "numpy-2.2.2-cp313-cp313t-macosx_11_0_arm64.whl", hash = "sha256:9491100aba630910489c1d0158034e1c9a6546f0b1340f716d522dc103788e39"}, + {file = "numpy-2.2.2-cp313-cp313t-macosx_14_0_arm64.whl", hash = "sha256:41184c416143defa34cc8eb9d070b0a5ba4f13a0fa96a709e20584638254b317"}, + {file = "numpy-2.2.2-cp313-cp313t-macosx_14_0_x86_64.whl", hash = "sha256:7dca87ca328f5ea7dafc907c5ec100d187911f94825f8700caac0b3f4c384b49"}, + {file = "numpy-2.2.2-cp313-cp313t-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:0bc61b307655d1a7f9f4b043628b9f2b721e80839914ede634e3d485913e1fb2"}, + {file = "numpy-2.2.2-cp313-cp313t-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:9fad446ad0bc886855ddf5909cbf8cb5d0faa637aaa6277fb4b19ade134ab3c7"}, + {file = "numpy-2.2.2-cp313-cp313t-musllinux_1_2_aarch64.whl", hash = "sha256:149d1113ac15005652e8d0d3f6fd599360e1a708a4f98e43c9c77834a28238cb"}, + {file = "numpy-2.2.2-cp313-cp313t-musllinux_1_2_x86_64.whl", hash = "sha256:106397dbbb1896f99e044efc90360d098b3335060375c26aa89c0d8a97c5f648"}, + {file = "numpy-2.2.2-cp313-cp313t-win32.whl", hash = "sha256:0eec19f8af947a61e968d5429f0bd92fec46d92b0008d0a6685b40d6adf8a4f4"}, + {file = "numpy-2.2.2-cp313-cp313t-win_amd64.whl", hash = "sha256:97b974d3ba0fb4612b77ed35d7627490e8e3dff56ab41454d9e8b23448940576"}, + {file = "numpy-2.2.2-pp310-pypy310_pp73-macosx_10_15_x86_64.whl", hash = "sha256:b0531f0b0e07643eb089df4c509d30d72c9ef40defa53e41363eca8a8cc61495"}, + {file = "numpy-2.2.2-pp310-pypy310_pp73-macosx_14_0_x86_64.whl", hash = "sha256:e9e82dcb3f2ebbc8cb5ce1102d5f1c5ed236bf8a11730fb45ba82e2841ec21df"}, + {file = "numpy-2.2.2-pp310-pypy310_pp73-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:e0d4142eb40ca6f94539e4db929410f2a46052a0fe7a2c1c59f6179c39938d2a"}, + {file = "numpy-2.2.2-pp310-pypy310_pp73-win_amd64.whl", hash = "sha256:356ca982c188acbfa6af0d694284d8cf20e95b1c3d0aefa8929376fea9146f60"}, + {file = "numpy-2.2.2.tar.gz", hash = "sha256:ed6906f61834d687738d25988ae117683705636936cc605be0bb208b23df4d8f"}, ] [[package]] @@ -1996,13 +2070,13 @@ twisted = ["twisted"] [[package]] name = "prompt-toolkit" -version = "3.0.48" +version = "3.0.50" description = "Library for building powerful interactive command lines in Python" optional = false -python-versions = ">=3.7.0" +python-versions = ">=3.8.0" files = [ - {file = "prompt_toolkit-3.0.48-py3-none-any.whl", hash = "sha256:f49a827f90062e411f1ce1f854f2aedb3c23353244f8108b89283587397ac10e"}, - {file = "prompt_toolkit-3.0.48.tar.gz", hash = "sha256:d6623ab0477a80df74e646bdbc93621143f5caf104206aa29294d53de1a03d90"}, + {file = "prompt_toolkit-3.0.50-py3-none-any.whl", hash = "sha256:9b6427eb19e479d98acff65196a307c555eb567989e6d88ebbb1b509d9779198"}, + {file = "prompt_toolkit-3.0.50.tar.gz", hash = "sha256:544748f3860a2623ca5cd6d2795e7a14f3d0e1c3c9728359013f79877fc89bab"}, ] [package.dependencies] @@ -2240,6 +2314,24 @@ pluggy = ">=1.5,<2" [package.extras] dev = ["argcomplete", "attrs (>=19.2)", "hypothesis (>=3.56)", "mock", "pygments (>=2.7.2)", "requests", "setuptools", "xmlschema"] +[[package]] +name = "pytest-cov" +version = "6.0.0" +description = "Pytest plugin for measuring coverage." +optional = false +python-versions = ">=3.9" +files = [ + {file = "pytest-cov-6.0.0.tar.gz", hash = "sha256:fde0b595ca248bb8e2d76f020b465f3b107c9632e6a1d1705f17834c89dcadc0"}, + {file = "pytest_cov-6.0.0-py3-none-any.whl", hash = "sha256:eee6f1b9e61008bd34975a4d5bab25801eb31898b032dd55addc93e96fcaaa35"}, +] + +[package.dependencies] +coverage = {version = ">=7.5", extras = ["toml"]} +pytest = ">=4.6" + +[package.extras] +testing = ["fields", "hunter", "process-tests", "pytest-xdist", "virtualenv"] + [[package]] name = "pytest-mock" version = "3.14.0" @@ -2648,18 +2740,19 @@ all = ["numpy"] [[package]] name = "referencing" -version = "0.35.1" +version = "0.36.1" description = "JSON Referencing + Python" optional = false -python-versions = ">=3.8" +python-versions = ">=3.9" files = [ - {file = "referencing-0.35.1-py3-none-any.whl", hash = "sha256:eda6d3234d62814d1c64e305c1331c9a3a6132da475ab6382eaa997b21ee75de"}, - {file = "referencing-0.35.1.tar.gz", hash = "sha256:25b42124a6c8b632a425174f24087783efb348a6f1e0008e63cd4466fedf703c"}, + {file = "referencing-0.36.1-py3-none-any.whl", hash = "sha256:363d9c65f080d0d70bc41c721dce3c7f3e77fc09f269cd5c8813da18069a6794"}, + {file = "referencing-0.36.1.tar.gz", hash = "sha256:ca2e6492769e3602957e9b831b94211599d2aade9477f5d44110d2530cf9aade"}, ] [package.dependencies] attrs = ">=22.2.0" rpds-py = ">=0.7.0" +typing-extensions = {version = ">=4.4.0", markers = "python_version < \"3.13\""} [[package]] name = "requests" @@ -2821,20 +2914,20 @@ files = [ [[package]] name = "s3transfer" -version = "0.10.4" +version = "0.11.1" description = "An Amazon S3 Transfer Manager" optional = false python-versions = ">=3.8" files = [ - {file = "s3transfer-0.10.4-py3-none-any.whl", hash = "sha256:244a76a24355363a68164241438de1b72f8781664920260c48465896b712a41e"}, - {file = "s3transfer-0.10.4.tar.gz", hash = "sha256:29edc09801743c21eb5ecbc617a152df41d3c287f67b615f73e5f750583666a7"}, + {file = "s3transfer-0.11.1-py3-none-any.whl", hash = "sha256:8fa0aa48177be1f3425176dfe1ab85dcd3d962df603c3dbfc585e6bf857ef0ff"}, + {file = "s3transfer-0.11.1.tar.gz", hash = "sha256:3f25c900a367c8b7f7d8f9c34edc87e300bde424f779dc9f0a8ae4f9df9264f6"}, ] [package.dependencies] -botocore = ">=1.33.2,<2.0a.0" +botocore = ">=1.36.0,<2.0a.0" [package.extras] -crt = ["botocore[crt] (>=1.33.2,<2.0a.0)"] +crt = ["botocore[crt] (>=1.36.0,<2.0a.0)"] [[package]] name = "send2trash" @@ -3075,13 +3168,13 @@ files = [ [[package]] name = "tzdata" -version = "2024.2" +version = "2025.1" description = "Provider of IANA time zone data" optional = false python-versions = ">=2" files = [ - {file = "tzdata-2024.2-py2.py3-none-any.whl", hash = "sha256:a48093786cdcde33cad18c2555e8532f34422074448fbc874186f0abd79565cd"}, - {file = "tzdata-2024.2.tar.gz", hash = "sha256:7d85cc416e9382e69095b7bdf4afd9e3880418a2413feec7069d533d6b4e31cc"}, + {file = "tzdata-2025.1-py2.py3-none-any.whl", hash = "sha256:7e127113816800496f027041c570f50bcd464a020098a3b6b199517772303639"}, + {file = "tzdata-2025.1.tar.gz", hash = "sha256:24894909e88cdb28bd1636c6887801df64cb485bd593f2fd83ef29075a81d694"}, ] [[package]] @@ -3178,4 +3271,4 @@ files = [ [metadata] lock-version = "2.0" python-versions = "^3.12" -content-hash = "977386812e414d6b13d5237092e9da24d33cb5400a34b6a241898379eed1fba9" +content-hash = "f95476763e9ff04194b62402a3835367059c0af94bdb0b706eab6709d5a02848" diff --git a/fieldExtraction/pyproject.toml b/fieldExtraction/pyproject.toml index 63cf31e..731ec3a 100644 --- a/fieldExtraction/pyproject.toml +++ b/fieldExtraction/pyproject.toml @@ -16,7 +16,6 @@ psutil = "^6.1.0" rapidfuzz = "^3.10.1" pyxlsb = "^1.0.10" openpyxl = "^3.1.5" -pytest-mock = "^3.14.0" [tool.poetry.group.dev.dependencies] black = "^24.10.0" @@ -27,6 +26,7 @@ isort = "^5.13.2" [tool.poetry.group.test.dependencies] pytest = "^8.3.3" pytest-mock = "^3.14.0" +pytest-cov = "^6.0.0" [build-system] requires = ["poetry-core"] diff --git a/fieldExtraction/src/investment/file_processing.py b/fieldExtraction/src/investment/file_processing.py index 7006c3a..b31d5d8 100644 --- a/fieldExtraction/src/investment/file_processing.py +++ b/fieldExtraction/src/investment/file_processing.py @@ -97,8 +97,8 @@ def run_b_prompts(file_object): ################## PREPROCESS ################## contract_text = preprocess.clean_text(contract_text) text_dict, top_sheet_dict, num_pages = preprocess.split_text(contract_text) + text_dict = preprocess.clean_tables(text_dict, filename) # TODO: Optimize later to not run a bunch of extra prompts for the extra table pages exhibit_pages, exhibit_chunk_mapping = preprocess.one_to_n_exhibit_chunking(text_dict, filename) - text_dict = preprocess.clean_tables(text_dict, filename) print(f"B Preprocessing Complete - {filename}") ################## RUN BOTTOM UP PROMPTS ################## diff --git a/fieldExtraction/src/investment/preprocess.py b/fieldExtraction/src/investment/preprocess.py index a568796..c924f3c 100644 --- a/fieldExtraction/src/investment/preprocess.py +++ b/fieldExtraction/src/investment/preprocess.py @@ -1,7 +1,5 @@ -import csv - import src.utils.table_utils as table_utils -from src import config, keywords, preprocessing_funcs +from src import keywords, preprocessing_funcs def clean_text(contract_text): @@ -68,19 +66,30 @@ def one_to_n_exhibit_chunking(text_dict, filename): return exhibit_pages, exhibit_chunk_mapping -def clean_tables(text_dict, filename): +def clean_tables(text_dict, filename, row_limit=5): """ - In current state, this function is a wrapper for align_and_format_tables. Call any new table-related preprocessing funcs here + Cleans and processes tables within a given text dictionary. - Parameters: - text_dict (dict): A dictionary where keys are page numbers and values are the text on those pages. - filename (str): The name of the file being processed + This function performs the following steps: + 1. Groups and merges tables from the text dictionary. + 2. Splits the grouped tables into pages based on the specified row limit. + 3. Reinserts the new tables back into the text dictionary. + 4. Aligns and formats the tables. + 5. Splits the tables into separate subpages. + + Args: + text_dict (dict): A dictionary containing text data with tables to be processed. + filename (str): The name of the file being processed. + row_limit (int, optional): The maximum number of rows per table page. Defaults to 5. Returns: - dict: text_dict, with the tables aligned and formatted + dict: text_dict, with the tables aligned, formatted, and split into multiple subpages according to the row limit. """ - text_dict = table_utils.align_and_format_tables(text_dict, filename) - return text_dict + groups, page_tables_map = table_utils.group_and_merge_tables(text_dict) + group_table_split_info = table_utils.split_table_to_pages(groups, page_tables_map, row_limit=row_limit) + modified_text_dict = table_utils.reinsert_new_tables_to_text(group_table_split_info, text_dict) + text_dict = table_utils.align_and_format_tables(modified_text_dict, filename) + return table_utils.split_tables_to_separate_subpages(text_dict) def one_to_one_smart_chunking(text_dict, contract_text, keyword_mappings=keywords.GROUPED_KEYWORD_MAPPINGS): diff --git a/fieldExtraction/src/preprocessing_funcs.py b/fieldExtraction/src/preprocessing_funcs.py index a1bad4e..2e76fa8 100644 --- a/fieldExtraction/src/preprocessing_funcs.py +++ b/fieldExtraction/src/preprocessing_funcs.py @@ -25,6 +25,7 @@ def remove_page_indicators(contract_text: str) -> str: return cleaned_text +# TODO: write unit tests def split_text(text: str) -> dict[str, str]: """Split text on pages by the string `Start of Page No. = ' @@ -34,19 +35,14 @@ def split_text(text: str) -> dict[str, str]: Returns: dict[str, str]: A dictionary, keyed by the string page number and valued by the page text. """ - - if isinstance(text, str): - if not text: - return {} - - temp_list = text.split("Start of Page No. = ") - text_list = re.split(r"Start of Page No. = [0-9]+\n", text) + temp_list = text.split("Start of Page No. = ") + text_list = re.split(r"Start of Page No. = [0-9]+\n", text) - text_dict = {} - for i in range(len(text_list)): - text_dict[temp_list[i].split()[0]] = text_list[i] # splits on whitespace characters, which includes spaces, tabs, and newline characters + text_dict = {} + for i in range(len(text_list)): + text_dict[temp_list[i].split()[0]] = text_list[i] # splits on whitespace characters, which includes spaces, tabs, and newline characters - return {k: v for k, v in text_dict.items() if k != "Document"} + return {k: v for k, v in text_dict.items() if k != "Document"} def clean_law_symbols(contract_text): @@ -304,7 +300,7 @@ def get_exhibit_pages(text_dict, filename): for page_num, page in text_dict.items(): prompt = preprocessing_prompts.EXHIBIT_CHECK(page[0:100]) claude_answer_raw = llm_utils.invoke_claude( - prompt, config.MODEL_ID_CLAUDE3_HAIKU, filename, max_tokens=10 + prompt, config.MODEL_ID_CLAUDE3_HAIKU, filename, max_tokens=10 # TODO: low priority, try increasing max_tokens and maybe pass multiple pages in to reduce overall calls ) claude_answer_extracted = string_utils.extract_text_from_delimiters( claude_answer_raw, Delimiter.PIPE diff --git a/fieldExtraction/src/prompts/preprocessing_prompts.py b/fieldExtraction/src/prompts/preprocessing_prompts.py index ac2e177..3cbbee7 100644 --- a/fieldExtraction/src/prompts/preprocessing_prompts.py +++ b/fieldExtraction/src/prompts/preprocessing_prompts.py @@ -6,7 +6,7 @@ Analyze the following contract excerpt and determine if it indicates the start o 1. Look specifically for the words 'Exhibit', 'Article', 'Amendment', 'Schedule', 'Attachment', or 'Addendum'. These words indicate the start of a new section. 2. If the above words are found in the context of a longer sentence or statement, this does not by itself indicate the start of a section. Section starts will be formatted like a title or header. 3. The phrase 'Additional Provisions' does not indicate the start of a section. -4. Return Y if the excerpt indicates the start of a section, N if not. Enclose your final answer in |pipes| +4. Return Y if the excerpt indicates the start of a section, N if not. Please justify your answer, but enclose your final answer in |pipes| Here is the text to analyze: @@ -43,17 +43,20 @@ Alignment Instructions: Align the tables in the text using the Table Header text found after -------Table Start--------. If there is no header, align based on context clues in the text such as "the table below", etc. If there are multiple tables on the page, the first table should be placed in the first valid spot, the second in the second, etc. If the Table Header is None, place the updated table at the top of the page. +Do not replicate the table header text twice. Do not remove the -------Table Start-------- or -------Table End-------- lines. Formatting Instructions: -Here is a template format to use: +Reformat the table from json format to a human-readable fixed-width format. Use the following tempate format: Original JSON: "-------Table Start-------- Table Header {{'Column 1' : ['Value 1', 'Value 2', 'Value 3'], 'Column 2' : ['Value 4', 'Value 5', 'Value 6']}}-------Table End--------" Output Format: +-------Table Start-------- Table Header Column 1 : Value 1 | Column 2 : Value 4 Column 1 : Value 2 | Column 2 : Value 5 Column 1 : Value 3 | Column 2 : Value 6 +-------Table End-------- Ensure that all text other than the tables is exactly the same diff --git a/fieldExtraction/src/utils/string_utils.py b/fieldExtraction/src/utils/string_utils.py index 1e479e9..89d3ce6 100644 --- a/fieldExtraction/src/utils/string_utils.py +++ b/fieldExtraction/src/utils/string_utils.py @@ -223,27 +223,64 @@ def primary_string_to_dict(string_dict, filename): data.append(dict_) return data +reimbursement_strings = [ # These are for `method='keyword'` + "%", + "$", + "percent", + "compensation schedule", + "reimbursement schedule", +] -def contains_reimbursement(text, page="1"): # string_funcs.py - if isinstance(text, dict): - return page.isdigit() and ( - "%" in text[page] - or "$" in text[page] - or "percent " in text[page] - or "compensation schedule" in text[page].lower() - or "reimbursement schedule" in text[page].lower() - ) - elif isinstance(text, str): - return ( - "%" in text - or "$" in text - or "percent " in text - or "compensation schedule" in text.lower() - or "reimbursement schedule" in text.lower() - ) +# Maybe drop compensation / reimbursement schedules +# contains and count should do a number followed by a percent or a number and a dollar +# Also could be like "one hundred percent" or "one hundred dollars" +# limit reimbursement strings to the first three + +# Then for counting AND containing, use the regex + +# maybe have `contains_reimbursement_keywords` and `contains_reimbursement_regex` functions +# method='keyword' or method='regex' + +reimb_regex = r"(? int: #JUST by regex + """Counts reimbursements in an exhibit text. Reimbursements are detected by a regex + search as defined by `reimb_regex`. + + Args: + exhibit_text (str): Input exhibit text + + Returns: + int: Number of reimbursements detected + """ + return len(re.findall(reimb_regex, exhibit_text)) def is_empty(value): # string_funcs.py if pd.isna(value): @@ -251,7 +288,7 @@ def is_empty(value): # string_funcs.py else: empty_values = [None, "", "N/A", "NA", "null", "none", "NaN", np.nan, "nan"] return value in empty_values - + def get_exhibit_chunk(text_dict: dict, exhibit_chunk_mapping: dict, diff --git a/fieldExtraction/src/utils/table_utils.py b/fieldExtraction/src/utils/table_utils.py index 36018df..c16a8ff 100644 --- a/fieldExtraction/src/utils/table_utils.py +++ b/fieldExtraction/src/utils/table_utils.py @@ -1,15 +1,531 @@ -from src.prompts import preprocessing_prompts +import ast +import copy +import math +import re +from collections import defaultdict +from typing import Any, Dict, List, Optional, Pattern, Tuple, Union + +import pandas as pd import src.utils.llm_utils as llm_utils import src.utils.string_utils as string_utils from src import config +from src.prompts import preprocessing_prompts + +# Table regex patterns +START_PAGE_PATTERN: Pattern[str] = re.compile( + r"^(?!None\n).+\n\{.*?\}", re.DOTALL +) # Regex for start page - shouldn't have literal `None` +CONTINUATION_PAGE_PATTERN: Pattern[str] = re.compile( + r"^None\n\{.*?\}", re.DOTALL +) # Regex for continuation page - has literal `None` present at the start +START_PAGE_NUM_PATTERN: str = r"\n\d+\n?" +CONTINUATION_PAGE_NUM_PATTERN: str = r"\d+\n" +DICTIONARY_PATTERN: str = r"\{.*\}" + +# Markers for table start and end +START_MARKER: str = "-------Table Start--------" +END_MARKER: str = "-------Table End--------" def align_and_format_tables(text_dict, filename): for page_num, page_text in text_dict.items(): - if "Table Start" in page_text and string_utils.contains_reimbursement(page_text): + # Right now we are only aligning/formatting if the page contained a reimbursement. Should we do it for all tables? + # Decision as of 1/24/25: no, we only care about table formatting for reimbursement + # In a possible future case where there are many NPIs/TINs, we may want to format all tables + # But for now, our regexes catch NPIs/TINs. + if "Table Start" in page_text and string_utils.contains_reimbursement( + page_text + ): prompt = preprocessing_prompts.ALIGN_AND_FORMAT_TABLES(page_text) aligned_page = llm_utils.invoke_claude( prompt, config.MODEL_ID_CLAUDE35_SONNET, filename, 8192 ) text_dict[page_num] = aligned_page return text_dict + + +def count_tables(exhibit_text: str) -> int: + """Count number of tables found in an exhibit, through use of a + regex that catches the table start and end strings. + + Args: + exhibit_text (str): Input page text + + Returns: + int: Number of tables + """ + pattern = r"-Table Start-" # Source text contains "-------Table Start--------" but in case hyphen count is unreliable, just looking for one + matches = re.findall(pattern, exhibit_text) + return len(matches) + + +def extract_first_table_from_page(text: str) -> str: + """ + Extracts the first table from a given page of text based on predefined start and end markers. + + Args: + text (str): The text content of the page from which the table needs to be extracted. + + Returns: + str: The extracted table content as a string. If the start or end markers are not found, returns an empty string. + """ + if START_MARKER not in text or END_MARKER not in text: + return "" + + start_index: int = text.find(START_MARKER) + len(START_MARKER) + end_index: int = text.find(END_MARKER) + filtered_page: str = text[start_index:end_index].strip() + + return filtered_page + + +def extract_snippets_between_markers(text: str) -> List[str]: + """ + Extracts all snippets from the input text that are enclosed between START_MARKER and END_MARKER. + + Args: + text (str): The input text from which to extract the snippets. + + Returns: + list: A list of substrings found between START_MARKER and END_MARKER in the input text. + """ + return re.findall(rf"{START_MARKER}(.+?){END_MARKER}", text, re.DOTALL) + + +def get_str_dictionaries_from_text( + text: str, only_first: bool = True +) -> Union[List[str], str]: + """ + Extracts dictionary-like strings from the given text using a predefined pattern. + + Args: + text (str): The input text from which to extract dictionary-like strings. + only_first (bool, optional): If True, returns only the first match. If False, returns all matches. Defaults to True. + + Returns: + list or str: A list of all matched dictionary-like strings if only_first is False, otherwise the first matched string. + """ + matches = re.findall(DICTIONARY_PATTERN, text) + if not matches: + return "" + return matches[0] if only_first else matches + + +def page_contains_multiple_tables(text: str) -> bool: + """ + Checks if the given text contains multiple tables. + + Args: + text (str): The text to be checked for multiple tables. + + Returns: + bool: True if the text contains more than one table, False otherwise. + """ + return len(re.findall(rf"{START_MARKER}", text)) > 1 + + +def get_page_metadata(page: str) -> Union[List[str], List[Tuple]]: + """ + Extracts and returns metadata from a given page. + + If the page contains a single table, it returns the common metadata for the table. + If the page contains multiple tables, it returns a list of metadata for each table (ordered by position in the page). + If the page contains no tables (or contains a malformed table), it returns an empty list. + + Args: + page (str): The content of the page from which metadata is to be extracted. + + Returns: + List[str]: A list containing metadata. If the page contains a single table, + the list contains one element with the common metadata. If the + page contains multiple tables, the list contains tuples where + each tuple consists of the metadata and the matched dictionary pattern (the corresponding table). + """ + # if the page does not contain a table (or contains a malformed table), return an empty list + if not re.search(START_MARKER, page) or not re.search(END_MARKER, page): + print("Malformed table in get_page_metadata!") + return [] + + + # return the common metadata for the table (or the group of tables starting with this one) + if not page_contains_multiple_tables(page): + group_metadata = page[: page.find(START_MARKER)].strip() + group_metadata = re.sub( + START_PAGE_NUM_PATTERN, "\n", group_metadata + ) # Remove page number marker + return [group_metadata] + else: # if the page contains multiple tables + all_tables = extract_snippets_between_markers(page) + table_metadata_map = [] + + for table in all_tables: + match = re.search(DICTIONARY_PATTERN, table, re.DOTALL) + if match: + metadata = table.replace(match.group(0), "").strip() + table_metadata_map.append((metadata, match.group(0))) + + return table_metadata_map + + +def convert_str_to_dict(text: str) -> Dict[str, Any]: + """ + Converts a string representation of a dictionary to an actual dictionary. + + This function searches for a dictionary pattern within the given text and + converts the matched string to a dictionary using `ast.literal_eval`. + + Args: + text (str): The input string that potentially contains a dictionary. + + Returns: + dict: The dictionary extracted from the input string. If no dictionary + pattern is found, an empty dictionary is returned. + """ + match = re.search(DICTIONARY_PATTERN, text) + if match: + dict_str: str = match.group(0) + try: + return ast.literal_eval(dict_str) + except (SyntaxError, ValueError): + print(f"Invalid dictionary format: {dict_str}") + return {} + + +def group_pages_containing_tables( + text_dict: Dict[str, str] +) -> Dict[str, Dict[str, Any]]: + #TODO: add test cases for this function + """ + Groups pages containing tables into start and continuation groups based on patterns. + + Args: + text_dict (Dict[str, str]): A dictionary where keys are page numbers (as strings) and values are the page contents. + + Returns: + Dict[str, Dict[str, Any]]: A dictionary where keys are the start page numbers of groups, and values are dictionaries containing: + - "metadata": Metadata of the start page. + - "group": List of page numbers in the group. + + The function processes each page in the input dictionary, identifies if the page contains a table that matches a start or continuation pattern, + and groups the pages accordingly. If a page does not match any pattern, it is not included in any group. + """ + groups: Dict[str, Dict[str, Any]] = defaultdict(dict) + current_group: List[str] = [] + group_metadata: Optional[str] = None + for page_num, page_content in text_dict.items(): + # We can have one or more tables present on a page. The first table (ONLY) on the page decides if the page will be a start or continuation page. + + first_table_on_page = extract_first_table_from_page(page_content) + + if re.search( + START_PAGE_PATTERN, first_table_on_page + ): # Check if the page matches the start pattern + if current_group: # if a group is already established previously + group_start_page: str = current_group[0] + groups[group_start_page] = { + "metadata": group_metadata, + "group": current_group, + } + current_group = ( + [] + ) # reset the current group, because we've found the start pattern + group_metadata = None + current_group = [page_num] + group_metadata = get_page_metadata(page_content) + elif re.search( + CONTINUATION_PAGE_PATTERN, first_table_on_page + ): # Check if the page matches the continuation pattern + if current_group: + current_group.append(page_num) + else: # If the first page we're processing is a continuation page, we start a new group (edge case) + current_group = [page_num] + group_metadata = get_page_metadata(page_content) + else: # If the page doesn't match any pattern + if current_group: # if we're in the middle of a group, close it + group_start_page = current_group[0] + groups[group_start_page] = { + "metadata": group_metadata, + "group": current_group, + } + current_group = [] # reset the current group + group_metadata = None + if current_group: # save off the current group if we're at the end of the text + group_start_page = current_group[0] + groups[group_start_page] = {"metadata": group_metadata, "group": current_group} + return groups + + +def insert_metadata(metadata: str, table_content: str) -> str: + """ + Inserts metadata at the beginning of the table content. + + Args: + metadata (str): The metadata to be inserted. + table_content (str): The content of the table. + + Returns: + str: The combined string with metadata followed by the table content. + """ + return f"{metadata}\n{table_content}" + + +def insert_column_headers( + data_dict: dict[str, list[str]], columns: list +) -> dict[str, list[str]]: + """ + Inserts column headers into a dictionary of data. It takes the existing column headers and inserts them as the first row of the data. + + Args: + data_dict (dict): The dictionary containing the data to be modified. + columns (list): A list of column headers to be inserted. + + Returns: + dict: A new dictionary with the column headers inserted. If the input dictionary is empty, returns an empty dictionary. + """ + if not data_dict: + print("Dictionary not found in the text.") + return {} + record = list(data_dict.items()) + modified_dict = { + col_name: [col_items[0]] + col_items[1] + for col_name, col_items in zip(columns, record) + } + + return modified_dict + + +def correct_column_headers_and_merge( + text_dict: dict[str, str], groups: dict[str, dict[str, Any]] +) -> dict[str, List[pd.DataFrame]]: + """ + Corrects column headers and merges tables from multiple pages. + + Args: + text_dict (dict[str, str]): A dictionary where keys are page numbers and values are the text content of those pages. + groups (dict): A dictionary where keys are the starting page numbers of groups and values are dictionaries containing group data. + + Returns: + dict: A dictionary where keys are the starting page numbers of groups and values are lists of DataFrames representing the merged tables. + """ + + page_tables_map = {} + + columns_to_use: List[str] = [] # for mypy + + for group_start_page, group_data in groups.items(): # Loop through groups + dfs_list = [] + columns = None + + for page in group_data["group"]: # Loop through pages in the group + page_content = text_dict[page] + + str_dicts = get_str_dictionaries_from_text(page_content, only_first=False) + for str_dict in str_dicts: + data_dict = convert_str_to_dict(str_dict) + # if the page contains multiple tables, rewrite the latest column headers each time + # Otherwise, use the column headers from the first table in the group + columns_to_use = data_dict.keys() if page == group_start_page else columns_to_use + + col_corrected_dict = insert_column_headers(data_dict, columns_to_use) + + if ( + page == group_start_page + ): # on the first table, we've already changed the keys to be the first row, so we need to drop that row + df = pd.DataFrame(col_corrected_dict).drop(index=0) + else: + df = pd.DataFrame(col_corrected_dict) + + df["page_num"] = page + + dfs_list.append(df) + + tables = dfs_list + page_tables_map[group_start_page] = ( + tables # List of dataframes corresponding to the tables in the group + ) + return page_tables_map + + +def group_and_merge_tables(text_dict: dict[str, str]) -> tuple[dict, dict]: + """ + Groups pages containing tables and merges the tables after correcting column headers. + + Args: + text_dict (dict[str, str]): A dictionary where keys are page identifiers and values are the text content of the pages. + + Returns: + tuple[dict, dict]: A tuple containing: + - groups (dict): A dictionary where keys are group identifiers and values are lists of page identifiers that belong to each group. + - page_tables_map (dict): A dictionary where keys are page identifiers and values are the merged tables after correcting column headers. + """ + groups = group_pages_containing_tables(text_dict) + + page_tables_map = correct_column_headers_and_merge( + text_dict, groups + ) # extract the tables from all pages, correct the col headers, put the individual tables in the df, then merge those dfs + + return groups, page_tables_map + + +def split_table_to_pages( + groups: dict, page_tables_map: dict, row_limit: int +) -> dict: + """ + Splits tables into pages based on the provided groups and page tables map. + Args: + groups (dict): A dictionary where keys are the starting page numbers of groups and values are dictionaries containing group data. + page_tables_map (dict): A dictionary mapping page numbers to lists of DataFrames representing tables on those pages. + row_limit (int, optional): The number of rows per table split. Defaults to 5. + Returns: + dict: A dictionary where keys are the starting page numbers of groups and values are dictionaries mapping page numbers to lists of related tables. + """ + group_table_split_info = {} + + for group_start_page, group_data in groups.items(): # Loop through groups + dfs_list = page_tables_map[group_start_page] # get list of tables for the group + metadata_list = group_data["metadata"] + + if all( + [isinstance(metadata, tuple) for metadata in metadata_list] + ): # if the metadata is a list of tuples, this page has multiple tables + metadata_list = [ + metadata[0] for metadata in metadata_list + ] # Just extract the metadata strings (the second element of the tuple in the metadata is a text dictionary snippet) + + # if the group contains multiple pages with common metadata, duplicate the metadata for each page + # This handles cases: + # - when the group contains multiple pages with common metadata, this will work + # - when the page contains multiple tables which start AND end on the single page, this will work + # An unhandled edge case, when the page contains multiple tables in this scenario: + # - when the LAST table of a page is the beginning of a table continued onto the next page, we won't get the metadata right + + # If the group has a single page with multiple tables + if len(group_data["group"]) > 1: # if the group contains multiple pages + metadata_list = [ + metadata_list[0] for _ in range(len(dfs_list)) + ] # duplicate the metadata for each page + + if ( + len(group_data["group"]) == 1 + ): # its a single page group with len(dfs_list) number of tables - eg [6]->[6,6,6,6] if the group contains just page "6" and page "6" has 4 tables + group_data["group"] = [group_data["group"][0] for _ in range(len(dfs_list))] + + chunked_tables_page_mapping = defaultdict(list) + + # loop through each page, metadata, and dataframe in the group + for page_num, metadata, df in zip(group_data["group"], metadata_list, dfs_list): + num_rows = df.shape[0] + num_tables = math.ceil(num_rows / row_limit) # number of output tables + + related_tables = [] + + for i in range(num_tables): + start_row = i * row_limit + end_row = min((i + 1) * row_limit, df.shape[0]) + table_df = df.iloc[start_row:end_row, :] + + str_df = str(table_df.drop(columns=["page_num"]).to_dict(orient="list")) + + table_dict = insert_metadata(metadata, str_df) + + assert ( # TODO: convert to `raise` instead of `assert` + table_df["page_num"].nunique() == 1 + ), "All rows in a table should belong to the same page" + + page_num = table_df["page_num"].unique()[0] + + related_tables.append( + table_dict + ) # these tables are splits of a larger table from one page + + chunked_tables_page_mapping[page_num].append(related_tables) + + group_table_split_info[group_start_page] = chunked_tables_page_mapping + + return group_table_split_info + + +def replace_tables_in_page( + page_num: int, modified_text_dict: dict, new_tables_list: list +) -> None: + """ + Replaces existing tables within pages to modified (corrected) tables. + Args: + page_num (int): The page number where the reinsertion should occur. + modified_text_dict (dict): A dictionary containing the text of each page, with page numbers as keys. + new_tables_list (list): A list of lists, where each inner list contains tables to be inserted. + Returns: + None: This function modifies the `modified_text_dict` in place. + """ + + text = modified_text_dict[page_num] + + formatted_tables = [] + + for ( + table_list + ) in new_tables_list: # Format new tables for insertion back into pages + lst = [] + + for table in table_list: + lst.append(f"{START_MARKER}\n{table}\n{END_MARKER}") + + concatenated_tables = "\n".join(lst) + + formatted_tables.append(concatenated_tables) + + detect_tables_pattern = rf"({START_MARKER}.+?{END_MARKER})" + + tables_to_replace = re.findall(detect_tables_pattern, text, re.DOTALL) + + for original_table, replacement_table in zip(tables_to_replace, formatted_tables): + text = re.sub( + re.escape(original_table), replacement_table, text, count=1, flags=re.DOTALL + ) + + modified_text_dict[page_num] = text + + +def reinsert_new_tables_to_text(group_table_split_info: dict, text_dict: dict) -> dict: + """ + Take what's in groups, apply `replace_tables_in_page` to each group, and then reinsert the tables back into the text. + This is really a wrapper function for `replace_tables_in_page` which goes through each page and applies the function. + + Args: + group_table_split_info (dict): A dictionary where the keys are the starting page numbers of groups and the values + are dictionaries mapping page numbers to lists of split tables. + + Returns: + dict: A modified version of the original text dictionary with the tables reinserted. + """ + modified_text_dict = copy.deepcopy(text_dict) + + for group_start_page, chunked_tables_page_mapping in group_table_split_info.items(): + + for page_num in chunked_tables_page_mapping.keys(): + split_tables_list = chunked_tables_page_mapping[page_num] + # perform the reinsertion (inplace) + replace_tables_in_page(page_num, modified_text_dict, split_tables_list) + + return modified_text_dict + +def split_tables_to_separate_subpages(text_dict: dict[str, str]) -> dict[str, str]: + """ + Splits the text content of each page in the input dictionary into separate subpages based on a table end marker. + + Args: + text_dict (dict[str, str]): A dictionary where keys are page identifiers and values are the text content of those pages. + + Returns: + dict[str, str]: A new dictionary where keys are modified page identifiers (including subpage indices) and values are the split text content. Empty strings are removed from the result. + """ + new_text_dict = {} + for page in text_dict.keys(): + # split the text into separate pages based on the table end marker + if END_MARKER in text_dict[page]: + new_text_list = re.split(fr"(?<={END_MARKER})", text_dict[page]) # this regex will split on the end marker, and keep the end marker in the text + for i, text in enumerate(new_text_list): + new_text_dict[f"{page}.{i}"] = text + else: + new_text_dict[page] = text_dict[page] + return {k : v for k, v in new_text_dict.items() if v.strip()} # remove empty strings diff --git a/fieldExtraction/tests/preprocessing_funcs_test.py b/fieldExtraction/tests/preprocessing_funcs_test.py index 9edcfaa..262bfc2 100644 --- a/fieldExtraction/tests/preprocessing_funcs_test.py +++ b/fieldExtraction/tests/preprocessing_funcs_test.py @@ -1,4 +1,5 @@ import pytest +from src.preprocessing_funcs import split_text from src.preprocessing_funcs import ( remove_page_indicators, split_text, @@ -21,31 +22,6 @@ class TestPreprocessingFuncs: def test_clean_newlines(self, input_text, expected_output): assert remove_page_indicators(input_text) == expected_output - # Test cases for split_text - @pytest.mark.parametrize("input_text, expected_output", [ - # Splits on page markers - ( - "Title Start of Page No. = 1\nContent1\nStart of Page No. = 2\nContent2", - {"Title" : "Title ", "1": "Content1\n", "2": "Content2"} - ), - # Ignores "Document" key - ( - "Document\nStart of Page No. = 1\nContent1", - {"1": "Content1"} - ), - # Handles empty input - ("", {}), - # Handles no page markers - ("This is a single page.", {"This": "This is a single page."}), - # Handles multiple page markers - ( - "Introduction\nStart of Page No. = 1\nContent1\nStart of Page No. = 2\nContent2\nStart of Page No. = 3\nContent3", - {'Introduction': 'Introduction\n', '1': 'Content1\n', '2': 'Content2\n', '3': 'Content3'} - ), - ]) - def test_split_text(self, input_text, expected_output): - assert split_text(input_text) == expected_output - # Test cases for clean_law_symbols @pytest.mark.parametrize("input_text, expected_output", [ # Replaces double dollar signs diff --git a/fieldExtraction/tests/string_utils_test.py b/fieldExtraction/tests/string_utils_test.py index 32edaca..629259d 100644 --- a/fieldExtraction/tests/string_utils_test.py +++ b/fieldExtraction/tests/string_utils_test.py @@ -1,6 +1,6 @@ import pytest import src.utils as utils -from src.utils.string_utils import extract_text_from_delimiters, json_parsing_search, secondary_string_to_dict, primary_string_to_dict, contains_reimbursement, is_empty +from src.utils.string_utils import extract_text_from_delimiters, json_parsing_search, secondary_string_to_dict, primary_string_to_dict, contains_reimbursement, is_empty, count_reimbursements_in_exhibit import src.utils.llm_utils as llm_utils import numpy as np import pandas as pd @@ -175,7 +175,7 @@ class TestStringUtils: text = ["This is a list, not a dictionary or string."] page = "1" result = contains_reimbursement(text, page) - assert result is None + assert result is False captured = capsys.readouterr() assert "contains_reimbursement - Invalid data type" in captured.out @@ -194,3 +194,17 @@ class TestStringUtils: def test_is_empty(self, value, expected): result = is_empty(value) assert result == expected + +class TestCountReimbursementsInExhibit: + @pytest.mark.parametrize("exhibit_text, expected_count", [ + ("This exhibit includes a 10% reimbursement and a $100 reimbursement.", 2), + ("No reimbursements mentioned here.", 0), + ("Reimbursement of 50% and another reimbursement of $200.", 2), + ("100 percent reimbursement and fifty dollars reimbursement.", 0), + ("", 0), + ("Reimbursement: 20% and $300.", 2), + ("Reimbursement: 20% and $300. Another 10% reimbursement.", 3), + ]) + def test_count_reimbursements_in_exhibit(self, exhibit_text, expected_count): + result = count_reimbursements_in_exhibit(exhibit_text) + assert result == expected_count diff --git a/fieldExtraction/tests/table_utils_test.py b/fieldExtraction/tests/table_utils_test.py index 4e52ad7..069b1c8 100644 --- a/fieldExtraction/tests/table_utils_test.py +++ b/fieldExtraction/tests/table_utils_test.py @@ -1,9 +1,9 @@ import pytest from src.prompts import preprocessing_prompts import src.utils as utils -from src.utils.table_utils import align_and_format_tables -from src import (config) +from src import config import copy +from src.utils.table_utils import * class TestAlignAndFormatTables: @pytest.fixture @@ -112,4 +112,618 @@ class TestAlignAndFormatTables: ) for call in mock_invoke_claude.call_args_list: prompt, _, _, _ = call[0] # Unpack the arguments - assert input_text_dict["2"] not in prompt \ No newline at end of file + assert input_text_dict["2"] not in prompt + +class TestCountTables: + def test_count_tables_single_table(self): + """Test counting tables when there is a single table.""" + exhibit_text = ( + "-------Table Start-------- Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}-------Table End--------" + ) + result = count_tables(exhibit_text) + assert result == 1 + + def test_count_tables_multiple_tables(self): + """Test counting tables when there are multiple tables.""" + exhibit_text = ( + "-------Table Start-------- Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}-------Table End--------\n" + "-------Table Start-------- Another Table Header {'Column A': ['Value A1', 'Value A2'], 'Column B': ['Value B1', 'Value B2']}-------Table End--------" + ) + result = count_tables(exhibit_text) + assert result == 2 + + def test_count_tables_no_table(self): + """Test counting tables when there are no tables.""" + exhibit_text = ( + "This page does not contain a table.\n" + "Neither does this page." + ) + result = count_tables(exhibit_text) + assert result == 0 + + def test_count_tables_mixed_content(self): + """Test counting tables when there is a mix of pages with and without tables.""" + exhibit_text = ( + "-------Table Start-------- Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}-------Table End--------\n" + "This page does not contain a table.\n" + "-------Table Start-------- Another Table Header {'Column A': ['Value A1', 'Value A2'], 'Column B': ['Value B1', 'Value B2']}-------Table End--------" + ) + result = count_tables(exhibit_text) + assert result == 2 + +class TestExtractFirstTableFromPage: + def test_extract_first_table_from_page_with_table(self): + """Test extracting the first table when the page contains a table.""" + text = ( + "Some text before the table.\n" + "-------Table Start--------\n" + "Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n" + "-------Table End--------\n" + "Some text after the table." + ) + result = extract_first_table_from_page(text) + expected = "Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}" + assert result == expected + + def test_extract_first_table_from_page_no_table(self): + """Test extracting the first table when the page does not contain a table.""" + text = "This page does not contain a table." + result = extract_first_table_from_page(text) + assert result == "" + + def test_extract_first_table_from_page_multiple_tables(self): + """Test extracting the first table when the page contains multiple tables.""" + text = ( + "Some text before the tables.\n" + "-------Table Start--------\n" + "First Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n" + "-------Table End--------\n" + "Some text between the tables.\n" + "-------Table Start--------\n" + "Second Table Header {'Column A': ['Value A1', 'Value A2'], 'Column B': ['Value B1', 'Value B2']}\n" + "-------Table End--------\n" + "Some text after the tables." + ) + result = extract_first_table_from_page(text) + expected = "First Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}" + assert result == expected + + def test_extract_first_table_from_page_no_end_marker(self): + """Test extracting the first table when the end marker is missing.""" + text = ( + "Some text before the table.\n" + "-------Table Start--------\n" + "Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n" + "Some text after the table." + ) + result = extract_first_table_from_page(text) + assert result == "" + + def test_extract_first_table_from_page_no_start_marker(self): + """Test extracting the first table when the start marker is missing.""" + text = ( + "Some text before the table.\n" + "Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n" + "-------Table End--------\n" + "Some text after the table." + ) + result = extract_first_table_from_page(text) + assert result == "" + +class TestExtractSnippetsBetweenMarkers: + def test_extract_snippets_between_markers_single_snippet(self): + """Test extracting snippets when there is a single snippet.""" + text = ( + "Some text before the table.\n" + "-------Table Start--------\n" + "Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n" + "-------Table End--------\n" + "Some text after the table." + ) + result = extract_snippets_between_markers(text) + expected = ["\nTable Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n"] + assert result == expected + + def test_extract_snippets_between_markers_multiple_snippets(self): + """Test extracting snippets when there are multiple snippets.""" + text = ( + "Some text before the tables.\n" + "-------Table Start--------\n" + "First Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n" + "-------Table End--------\n" + "Some text between the tables.\n" + "-------Table Start--------\n" + "Second Table Header {'Column A': ['Value A1', 'Value A2'], 'Column B': ['Value B1', 'Value B2']}\n" + "-------Table End--------\n" + "Some text after the tables." + ) + result = extract_snippets_between_markers(text) + expected = [ + "\nFirst Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n", + "\nSecond Table Header {'Column A': ['Value A1', 'Value A2'], 'Column B': ['Value B1', 'Value B2']}\n" + ] + assert result == expected + + def test_extract_snippets_between_markers_no_snippets(self): + """Test extracting snippets when there are no snippets.""" + text = "This page does not contain a table." + result = extract_snippets_between_markers(text) + assert result == [] + + def test_extract_snippets_between_markers_no_end_marker(self): + """Test extracting snippets when the end marker is missing.""" + text = ( + "Some text before the table.\n" + "-------Table Start--------\n" + "Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n" + "Some text after the table." + ) + result = extract_snippets_between_markers(text) + assert result == [] + + def test_extract_snippets_between_markers_no_start_marker(self): + """Test extracting snippets when the start marker is missing.""" + text = ( + "Some text before the table.\n" + "Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n" + "-------Table End--------\n" + "Some text after the table." + ) + result = extract_snippets_between_markers(text) + assert result == [] + + +class TestGetStrDictionariesFromText: + def test_get_str_dictionaries_from_text_single_match(self): + """Test extracting a single dictionary-like string when only_first is True.""" + text = ( + "Some text before the dictionary.\n" + "{'key1': 'value1', 'key2': 'value2'}\n" + "Some text after the dictionary." + ) + result = get_str_dictionaries_from_text(text, only_first=True) + expected = "{'key1': 'value1', 'key2': 'value2'}" + assert result == expected + + def test_get_str_dictionaries_from_text_multiple_matches(self): + """Test extracting all dictionary-like strings when only_first is False.""" + text = ( + "Some text before the dictionaries.\n" + "{'key1': 'value1', 'key2': 'value2'}\n" + "Some text between the dictionaries.\n" + "{'keyA': 'valueA', 'keyB': 'valueB'}\n" + "Some text after the dictionaries." + ) + result = get_str_dictionaries_from_text(text, only_first=False) + expected = [ + "{'key1': 'value1', 'key2': 'value2'}", + "{'keyA': 'valueA', 'keyB': 'valueB'}" + ] + assert result == expected + + def test_get_str_dictionaries_from_text_no_match(self): + """Test extracting dictionary-like strings when there are no matches.""" + text = "This text does not contain any dictionary-like strings." + result = get_str_dictionaries_from_text(text, only_first=True) + assert result == "" + + def test_get_str_dictionaries_from_text_empty_text(self): + """Test extracting dictionary-like strings from an empty text.""" + text = "" + result = get_str_dictionaries_from_text(text, only_first=True) + assert result == "" + + def test_get_str_dictionaries_from_text_single_match_only_first_false(self): + """Test extracting dictionary-like strings when there is a single match and only_first is False.""" + text = ( + "Some text before the dictionary.\n" + "{'key1': 'value1', 'key2': 'value2'}\n" + "Some text after the dictionary." + ) + result = get_str_dictionaries_from_text(text, only_first=False) + expected = ["{'key1': 'value1', 'key2': 'value2'}"] + assert result == expected + + +class TestPageContainsMultipleTables: + def test_page_contains_multiple_tables_single_table(self): + """Test when the page contains a single table.""" + text = ( + "Some text before the table.\n" + "-------Table Start--------\n" + "Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n" + "-------Table End--------\n" + "Some text after the table." + ) + result = page_contains_multiple_tables(text) + assert result is False + + def test_page_contains_multiple_tables_multiple_tables(self): + """Test when the page contains multiple tables.""" + text = ( + "Some text before the tables.\n" + "-------Table Start--------\n" + "First Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n" + "-------Table End--------\n" + "Some text between the tables.\n" + "-------Table Start--------\n" + "Second Table Header {'Column A': ['Value A1', 'Value A2'], 'Column B': ['Value B1', 'Value B2']}\n" + "-------Table End--------\n" + "Some text after the tables." + ) + result = page_contains_multiple_tables(text) + assert result is True + + def test_page_contains_multiple_tables_no_table(self): + """Test when the page does not contain any tables.""" + text = "This page does not contain a table." + result = page_contains_multiple_tables(text) + assert result is False + + def test_page_contains_multiple_tables_empty_text(self): + """Test when the text is empty.""" + text = "" + result = page_contains_multiple_tables(text) + assert result is False + + def test_page_contains_multiple_tables_table_without_end_marker(self): + """Test when the page contains a table without an end marker.""" + text = ( + "Some text before the table.\n" + "-------Table Start--------\n" + "Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n" + "Some text after the table." + ) + result = page_contains_multiple_tables(text) + assert result is False + + +class TestGetPageMetadata: + def test_get_page_metadata_single_table(self): + """Test extracting metadata when the page contains a single table.""" + page = ( + "Page metadata before the table.\n" + "-------Table Start--------\n" + "Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n" + "-------Table End--------\n" + "Some text after the table." + ) + result = get_page_metadata(page) + expected = ["Page metadata before the table."] + assert result == expected + + def test_get_page_metadata_multiple_tables(self): + """Test extracting metadata when the page contains multiple tables.""" + page = ( + "Page metadata before the tables.\n" + "-------Table Start--------\n" + "First Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n" + "-------Table End--------\n" + "Some text between the tables.\n" + "-------Table Start--------\n" + "Second Table Header {'Column A': ['Value A1', 'Value A2'], 'Column B': ['Value B1', 'Value B2']}\n" + "-------Table End--------\n" + "Some text after the tables." + ) + result = get_page_metadata(page) + expected = [ + ("First Table Header", "{'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}"), + ("Second Table Header", "{'Column A': ['Value A1', 'Value A2'], 'Column B': ['Value B1', 'Value B2']}") + ] + assert result == expected + + def test_get_page_metadata_no_table(self): + """Test extracting metadata when the page does not contain any tables.""" + page = "This page does not contain a table." + result = get_page_metadata(page) + expected = [] + assert result == expected + + def test_get_page_metadata_empty_text(self): + """Test extracting metadata from an empty text.""" + page = "" + result = get_page_metadata(page) + expected = [] + assert result == expected + + def test_get_page_metadata_table_without_end_marker(self): + """Test extracting metadata when the page contains a table without an end marker.""" + page = ( + "Page metadata before the table.\n" + "-------Table Start--------\n" + "Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n" + "Some text after the table." + ) + result = get_page_metadata(page) + expected = [] + assert result == expected + + def test_get_page_metadata_table_without_start_marker(self): + """Test extracting metadata when the page contains a table without a start marker.""" + page = ( + "Page metadata before the table.\n" + "Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n" + "-------Table End--------\n" + "Some text after the table." + ) + result = get_page_metadata(page) + expected = [] + assert result == expected + +class TestConvertStrToDict: + def test_convert_str_to_dict_valid_dict(self): + """Test converting a valid dictionary-like string.""" + text = "{'key1': 'value1', 'key2': 'value2'}" + result = convert_str_to_dict(text) + expected = {'key1': 'value1', 'key2': 'value2'} + assert result == expected + + def test_convert_str_to_dict_no_dict(self): + """Test converting a string that does not contain a dictionary.""" + text = "This text does not contain any dictionary-like strings." + result = convert_str_to_dict(text) + expected = {} + assert result == expected + + def test_convert_str_to_dict_empty_text(self): + """Test converting an empty string.""" + text = "" + result = convert_str_to_dict(text) + expected = {} + assert result == expected + + def test_convert_str_to_dict_partial_dict(self): + """Test converting a string that contains a partial dictionary.""" + text = "Some text before the dictionary {'key1': 'value1', 'key2': 'value2'" + result = convert_str_to_dict(text) + expected = {} + assert result == expected + + def test_convert_str_to_dict_multiple_dicts(self): + """Test converting a string that contains multiple dictionaries.""" + text = ( + "Some text before the dictionaries.\n" + "{'key1': 'value1', 'key2': 'value2'}\n" + "Some text between the dictionaries.\n" + "{'keyA': 'valueA', 'keyB': 'valueB'}\n" + "Some text after the dictionaries." + ) + result = convert_str_to_dict(text) + expected = {'key1': 'value1', 'key2': 'value2'} + assert result == expected + + def test_convert_str_to_dict_invalid_dict(self): + """Test converting a string that contains an invalid dictionary.""" + text = "{'key1': 'value1', 'key2': value2}" # Missing quotes around value2 + result = convert_str_to_dict(text) + expected = {} + assert result == expected + + +class TestInsertMetadata: + def test_insert_metadata_with_valid_data(self): + """Test inserting metadata with valid data.""" + metadata = "Page metadata" + table_content = "Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}" + result = insert_metadata(metadata, table_content) + expected = "Page metadata\nTable Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}" + assert result == expected + + def test_insert_metadata_with_empty_metadata(self): + """Test inserting metadata when metadata is empty.""" + metadata = "" + table_content = "Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}" + result = insert_metadata(metadata, table_content) + expected = "\nTable Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}" + assert result == expected + + def test_insert_metadata_with_empty_table_content(self): + """Test inserting metadata when table content is empty.""" + metadata = "Page metadata" + table_content = "" + result = insert_metadata(metadata, table_content) + expected = "Page metadata\n" + assert result == expected + + def test_insert_metadata_with_both_empty(self): + """Test inserting metadata when both metadata and table content are empty.""" + metadata = "" + table_content = "" + result = insert_metadata(metadata, table_content) + expected = "\n" + assert result == expected + + def test_insert_metadata_with_special_characters(self): + """Test inserting metadata when metadata and table content contain special characters.""" + metadata = "Page metadata with special characters: !@#$%^&*()" + table_content = "Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']} with special characters: <>?/\\|" + result = insert_metadata(metadata, table_content) + expected = "Page metadata with special characters: !@#$%^&*()\nTable Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']} with special characters: <>?/\\|" + assert result == expected + + +class TestInsertColumnHeaders: + def test_insert_column_headers_with_valid_data(self): + """Test inserting column headers with valid data.""" + data_dict = { + "Column 1": ["Value 1", "Value 2"], + "Column 2": ["Value 3", "Value 4"] + } + columns = ["Column 1", "Column 2"] + result = insert_column_headers(data_dict, columns) + expected = { + "Column 1": ["Column 1", "Value 1", "Value 2"], + "Column 2": ["Column 2", "Value 3", "Value 4"] + } + assert result == expected + + def test_insert_column_headers_with_empty_data_dict(self): + """Test inserting column headers when data_dict is empty.""" + data_dict = {} + columns = ["Column 1", "Column 2"] + result = insert_column_headers(data_dict, columns) + expected = {} + assert result == expected + + def test_insert_column_headers_with_mismatched_columns(self): + """Test inserting column headers when columns list does not match data_dict keys.""" + data_dict = { + "Column 1": ["Value 1", "Value 2"], + "Column 2": ["Value 3", "Value 4"] + } + columns = ["Column A", "Column B"] + result = insert_column_headers(data_dict, columns) + expected = { + "Column A": ["Column 1", "Value 1", "Value 2"], + "Column B": ["Column 2", "Value 3", "Value 4"] + } + assert result == expected + + def test_insert_column_headers_with_extra_columns(self): + """Test inserting column headers when columns list has extra headers.""" + data_dict = { + "Column 1": ["Value 1", "Value 2"], + "Column 2": ["Value 3", "Value 4"] + } + columns = ["Column 1", "Column 2", "Column 3"] + result = insert_column_headers(data_dict, columns) + expected = { + "Column 1": ["Column 1", "Value 1", "Value 2"], + "Column 2": ["Column 2", "Value 3", "Value 4"] + } + assert result == expected + + def test_insert_column_headers_with_missing_columns(self): + """Test inserting column headers when columns list has missing headers.""" + data_dict = { + "Column 1": ["Value 1", "Value 2"], + "Column 2": ["Value 3", "Value 4"] + } + columns = ["Column 1"] + result = insert_column_headers(data_dict, columns) + expected = { + "Column 1": ["Column 1", "Value 1", "Value 2"] + } + assert result == expected + + def test_insert_column_headers_with_empty_columns(self): + """Test inserting column headers when columns list is empty.""" + data_dict = { + "Column 1": ["Value 1", "Value 2"], + "Column 2": ["Value 3", "Value 4"] + } + columns = [] + result = insert_column_headers(data_dict, columns) + expected = {} + assert result == expected + + +class TestSplitTablesToSeparateSubpages: + def test_split_tables_to_separate_subpages_single_table(self): + """Test splitting pages with a single table.""" + text_dict = { + "1": "Some text before the table.\n" + "-------Table Start--------\n" + "Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n" + "-------Table End--------\n" + "Some text after the table." + } + result = split_tables_to_separate_subpages(text_dict) + expected = { + "1.0": "Some text before the table.\n" + "-------Table Start--------\n" + "Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n" + "-------Table End--------", + "1.1": "\nSome text after the table." + } + assert result == expected + + def test_split_tables_to_separate_subpages_multiple_tables(self): + """Test splitting pages with multiple tables.""" + text_dict = { + "1": "Some text before the tables.\n" + "-------Table Start--------\n" + "First Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n" + "-------Table End--------\n" + "Some text between the tables.\n" + "-------Table Start--------\n" + "Second Table Header {'Column A': ['Value A1', 'Value A2'], 'Column B': ['Value B1', 'Value B2']}\n" + "-------Table End--------\n" + "Some text after the tables." + } + result = split_tables_to_separate_subpages(text_dict) + expected = { + "1.0": "Some text before the tables.\n" + "-------Table Start--------\n" + "First Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n" + "-------Table End--------", + "1.1": "\nSome text between the tables.\n" + "-------Table Start--------\n" + "Second Table Header {'Column A': ['Value A1', 'Value A2'], 'Column B': ['Value B1', 'Value B2']}\n" + "-------Table End--------", + "1.2": "\nSome text after the tables." + } + assert result == expected + + def test_split_tables_to_separate_subpages_no_table(self): + """Test splitting pages with no tables.""" + text_dict = { + "1": "This page does not contain a table." + } + result = split_tables_to_separate_subpages(text_dict) + expected = { + "1": "This page does not contain a table." + } + assert result == expected + + def test_split_tables_to_separate_subpages_empty_text(self): + """Test splitting pages with empty text.""" + text_dict = { + "1": "" + } + result = split_tables_to_separate_subpages(text_dict) + expected = {} + assert result == expected + + def test_split_tables_to_separate_subpages_mixed_content(self): + """Test splitting pages with mixed content.""" + text_dict = { + "1": "Some text before the table.\n" + "-------Table Start--------\n" + "Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n" + "-------Table End--------\n" + "Some text after the table.", + "2": "This page does not contain a table.", + "3": "Some text before the tables.\n" + "-------Table Start--------\n" + "First Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n" + "-------Table End--------\n" + "Some text between the tables.\n" + "-------Table Start--------\n" + "Second Table Header {'Column A': ['Value A1', 'Value A2'], 'Column B': ['Value B1', 'Value B2']}\n" + "-------Table End--------\n" + "Some text after the tables." + } + result = split_tables_to_separate_subpages(text_dict) + expected = { + "1.0": "Some text before the table.\n" + "-------Table Start--------\n" + "Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n" + "-------Table End--------", + "1.1": "\nSome text after the table.", + "2": "This page does not contain a table.", + "3.0": "Some text before the tables.\n" + "-------Table Start--------\n" + "First Table Header {'Column 1': ['Value 1', 'Value 2'], 'Column 2': ['Value 3', 'Value 4']}\n" + "-------Table End--------", + "3.1": "\nSome text between the tables.\n" + "-------Table Start--------\n" + "Second Table Header {'Column A': ['Value A1', 'Value A2'], 'Column B': ['Value B1', 'Value B2']}\n" + "-------Table End--------", + "3.2": "\nSome text after the tables." + } + assert result == expected + + + +