From 2222ccac5fc66dbdd460a99b7c5731f3e468c2f3 Mon Sep 17 00:00:00 2001 From: thanos Date: Sat, 12 Sep 2026 08:36:08 -0400 Subject: [PATCH 01/11] M0: upgrade arrow-rs to 59.3.0 (tonic 0.14, ADBC 0.24) - closed #263 - closed #264 - closed #265 - closed #266 - closed #267 --- native/ex_arrow_native/Cargo.lock | 810 ++++++++++------------- native/ex_arrow_native/Cargo.toml | 28 +- native/ex_arrow_native/src/flight_sql.rs | 26 +- native/ex_arrow_native/src/parquet.rs | 3 +- 4 files changed, 372 insertions(+), 495 deletions(-) diff --git a/native/ex_arrow_native/Cargo.lock b/native/ex_arrow_native/Cargo.lock index 00c400a..ab448af 100644 --- a/native/ex_arrow_native/Cargo.lock +++ b/native/ex_arrow_native/Cargo.lock @@ -4,9 +4,9 @@ version = 4 [[package]] name = "adbc_core" -version = "0.22.0" +version = "0.24.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e8dbe031527c9856a1e2df5e82aa8e568ffaab3be897f70d874477fb42a783bb" +checksum = "365059b13a01bbf6f324b5bfe328232a819679731431f78105e2883256236858" dependencies = [ "arrow-array", "arrow-schema", @@ -14,15 +14,17 @@ dependencies = [ [[package]] name = "adbc_driver_manager" -version = "0.22.0" +version = "0.24.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5beaa87308b040adcf482fb760f64ced1cbcd9469fe86ddd5be221dabbbc544e" +checksum = "3aac8bb789ff1f493bcb5d423c65b7a1b21ba5516f5e60797b4e1c64b34a2b61" dependencies = [ "adbc_core", "adbc_ffi", "arrow-array", "arrow-schema", "libloading 0.8.9", + "path-slash", + "regex", "toml", "windows-registry", "windows-sys 0.61.2", @@ -30,21 +32,15 @@ dependencies = [ [[package]] name = "adbc_ffi" -version = "0.22.0" +version = "0.24.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3600ae9aec2907516d088189e3b863029280f1953dd0eab903c7f4c862a0ce81" +checksum = "12cdef84c12e9e8858b300440c36ca94ba24fb2e188175251dc39bb44ba3584e" dependencies = [ "adbc_core", "arrow-array", "arrow-schema", ] -[[package]] -name = "adler2" -version = "2.0.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "320119579fcad9c21884f5c4861d16174d0e06250625266f50fe6898340abefa" - [[package]] name = "ahash" version = "0.8.12" @@ -61,33 +57,33 @@ dependencies = [ [[package]] name = "aho-corasick" -version = "1.1.4" +version = "1.1.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ddd31a130427c27518df266943a5308ed92d4b226cc639f5a8f1002816174301" +checksum = "c982642fa9e8606056828ee9a8505737230110bb1099153c79efe865c59d12ba" dependencies = [ "memchr", ] [[package]] name = "android_system_properties" -version = "0.1.5" +version = "0.1.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "819e7219dbd41043ac279b19830f2efc897156490d7fd6ea916720117ee66311" +checksum = "ae221649c9976a6f6c56ae1facf410f3ddb33cc661c4b7b61020a912d4237fbc" dependencies = [ "libc", ] [[package]] name = "anyhow" -version = "1.0.102" +version = "1.0.104" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7f202df86484c868dbad7eaa557ef785d5c66295e41b460ef922eca0723b842c" +checksum = "330a5ed07fa54e4702c9d6c4174f74427fc0ef6e214bbd677ae50a5099946470" [[package]] name = "arrow" -version = "56.2.0" +version = "59.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6e833808ff2d94ed40d9379848a950d995043c7fb3e81a30b383f4c6033821cc" +checksum = "7c14b3d39f306bc28fd639d59f06e17a0f377d0021e1b7e9054e4d6fedc98774" dependencies = [ "arrow-arith", "arrow-array", @@ -104,23 +100,23 @@ dependencies = [ [[package]] name = "arrow-arith" -version = "56.2.0" +version = "59.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ad08897b81588f60ba983e3ca39bda2b179bdd84dced378e7df81a5313802ef8" +checksum = "ce2961626677665b2195eb59242af4c7befe7b8737ca2050295389362380104e" dependencies = [ "arrow-array", "arrow-buffer", "arrow-data", "arrow-schema", "chrono", - "num", + "num-traits", ] [[package]] name = "arrow-array" -version = "56.2.0" +version = "59.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8548ca7c070d8db9ce7aa43f37393e4bfcf3f2d3681df278490772fd1673d08d" +checksum = "1e5f6adeffdf587d7a31db5d2266189624b526730cd3627f9ff9fedae97ad584" dependencies = [ "ahash", "arrow-buffer", @@ -129,57 +125,63 @@ dependencies = [ "chrono", "half", "hashbrown", - "num", + "libc", + "num-complex", + "num-integer", + "num-traits", ] [[package]] name = "arrow-buffer" -version = "56.2.0" +version = "59.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e003216336f70446457e280807a73899dd822feaf02087d31febca1363e2fccc" +checksum = "097d193003ce7995d5d087089069ec2a6e0187faf5a6f8c9f38af2645d987182" dependencies = [ "bytes", "half", - "num", + "num-bigint", + "num-traits", ] [[package]] name = "arrow-cast" -version = "56.2.0" +version = "59.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "919418a0681298d3a77d1a315f625916cb5678ad0d74b9c60108eb15fd083023" +checksum = "635c9c635668ad26adf76cce8fb276c4be7cf06e63bd516de7da514f9680ee53" dependencies = [ "arrow-array", "arrow-buffer", "arrow-data", + "arrow-ord", "arrow-schema", "arrow-select", "atoi", - "base64", + "base64 0.23.1", "chrono", "half", "lexical-core", - "num", + "num-traits", "ryu", ] [[package]] name = "arrow-data" -version = "56.2.0" +version = "59.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a5c64fff1d142f833d78897a772f2e5b55b36cb3e6320376f0961ab0db7bd6d0" +checksum = "9ba2f832eaeca24b8f26143dba750e42ee4ab51cf7d65e701ca9607cfda9f358" dependencies = [ "arrow-buffer", "arrow-schema", "half", - "num", + "num-integer", + "num-traits", ] [[package]] name = "arrow-flight" -version = "56.2.0" +version = "59.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8c8b0ba0784d56bc6266b79f5de7a24b47024e7b3a0045d2ad4df3d9b686099f" +checksum = "be0e6d452fff35cb4a3ef1a719d03fdfe653e590e8012b57d3057a316e295402" dependencies = [ "arrow-arith", "arrow-array", @@ -192,21 +194,21 @@ dependencies = [ "arrow-schema", "arrow-select", "arrow-string", - "base64", + "base64 0.23.1", "bytes", "futures", "once_cell", - "paste", "prost", "prost-types", "tonic", + "tonic-prost", ] [[package]] name = "arrow-ipc" -version = "56.2.0" +version = "59.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1d3594dcddccc7f20fd069bc8e9828ce37220372680ff638c5e00dea427d88f5" +checksum = "dcc41681ea80f521df14c36725b74d4c60702c47f0793af2be469c04527e2599" dependencies = [ "arrow-array", "arrow-buffer", @@ -218,9 +220,9 @@ dependencies = [ [[package]] name = "arrow-ord" -version = "56.2.0" +version = "59.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3c8f82583eb4f8d84d4ee55fd1cb306720cddead7596edce95b50ee418edf66f" +checksum = "2c900759f3bd8354fd4196bc4403eee846894dc2adf66b4225472006a0bf18c5" dependencies = [ "arrow-array", "arrow-buffer", @@ -231,9 +233,9 @@ dependencies = [ [[package]] name = "arrow-row" -version = "56.2.0" +version = "59.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9d07ba24522229d9085031df6b94605e0f4b26e099fb7cdeec37abd941a73753" +checksum = "f4c6425032e28266e3fc4ff680805e57e670d6ea92473043f3e65b7ed6ac79f2" dependencies = [ "arrow-array", "arrow-buffer", @@ -244,32 +246,32 @@ dependencies = [ [[package]] name = "arrow-schema" -version = "56.2.0" +version = "59.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b3aa9e59c611ebc291c28582077ef25c97f1975383f1479b12f3b9ffee2ffabe" +checksum = "10fab8d4563491417ba801fab29d205104d20d4bdf37bda6cd1cf425cff598cd" dependencies = [ "bitflags", ] [[package]] name = "arrow-select" -version = "56.2.0" +version = "59.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8c41dbbd1e97bfcaee4fcb30e29105fb2c75e4d82ae4de70b792a5d3f66b2e7a" +checksum = "fc58569193c2525915f3cc6310edba3792f1200f65d6e9ed330aa33e691493b8" dependencies = [ "ahash", "arrow-array", "arrow-buffer", "arrow-data", "arrow-schema", - "num", + "num-traits", ] [[package]] name = "arrow-string" -version = "56.2.0" +version = "59.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "53f5183c150fbc619eede22b861ea7c0eebed8eaac0333eaa7f6da5205fd504d" +checksum = "2e0813f3c35c1cfea65e14c20a953440f7783c088b7ad2d0db162ccdeefcec14" dependencies = [ "arrow-array", "arrow-buffer", @@ -277,20 +279,20 @@ dependencies = [ "arrow-schema", "arrow-select", "memchr", - "num", + "num-traits", "regex", "regex-syntax", ] [[package]] name = "async-trait" -version = "0.1.89" +version = "0.1.92" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9035ad2d096bed7955a320ee7e2230574d28fd3c3a0f186cbea1ff3c7eed5dbb" +checksum = "82f6aeea286b8eb4dd3431a1be1b59d290ace00f5bfd8e2a159bc2a05e2c1667" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 3.0.5", ] [[package]] @@ -310,15 +312,15 @@ checksum = "1505bd5d3d116872e7271a6d4e16d81d0c8570876c8de68093a09ac269d8aac0" [[package]] name = "autocfg" -version = "1.5.0" +version = "1.5.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c08606f8c3cbf4ce6ec8e28fb0014a2c086708fe954eaa885384a6165172e7e8" +checksum = "f2032f911046de80f0a198e0901378627c33f59ea0ac00e363d481118bd70a53" [[package]] name = "axum" -version = "0.8.8" +version = "0.8.9" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8b52af3cb4058c895d37317bb27508dccc8e5f2d39454016b297bf4a400597b8" +checksum = "31b698c5f9a010f6573133b09e0de5408834d0c82f8d7475a89fc1867a71cd90" dependencies = [ "axum-core", "bytes", @@ -364,34 +366,34 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "72b3254f16251a8381aa12e40e3c4d2f0199f8c6508fbecb9d91f575e0fbb8c6" [[package]] -name = "bitflags" -version = "2.11.0" +name = "base64" +version = "0.23.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "843867be96c8daad0d758b57df9392b6d8d271134fce549de6ce169ff98a92af" +checksum = "ac07cdecf99051d9a5238b80f35af32cdeba5b336e55d957b318b50137e18da5" [[package]] -name = "bumpalo" -version = "3.20.2" +name = "bitflags" +version = "2.13.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5d20789868f4b01b2f2caec9f5c4e0213b41e3e5702a50157d699ae31ced2fcb" +checksum = "3ded4057c258ba199e2d26386d3af3780957ecaee6c4ef4041c6b4b8b97c0b06" [[package]] -name = "byteorder" -version = "1.5.0" +name = "bumpalo" +version = "3.20.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1fd0f2584146f6f2ef48085050886acf353beff7305ebd1ae69500e27c67f64b" +checksum = "72f5acc6cb2ba439de613abc23857ec3d78374d8ed5ac84e9d11336e87da8649" [[package]] name = "bytes" -version = "1.11.1" +version = "1.12.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1e748733b7cbc798e1434b6ac524f0c1ff2ab456fe201501e6497c8417a4fc33" +checksum = "fc652a48c352aef3ea3aed32080501cf3ef6ed5da78602a020c991775b0aff04" [[package]] name = "cc" -version = "1.2.56" +version = "1.4.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "aebf35691d1bfb0ac386a69bac2fde4dd276fb618cf8bf4f5318fe285e821bb2" +checksum = "005ec2760ca554fae18df7a11195552ec576cd665632a881bc011d5bb2fd4d80" dependencies = [ "find-msvc-tools", "jobserver", @@ -407,13 +409,13 @@ checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801" [[package]] name = "chrono" -version = "0.4.44" +version = "0.4.45" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c673075a2e0e5f4a1dde27ce9dee1ea4558c7ffe648f576438a20ca1d2acc4b0" +checksum = "1aa79e62e7697b8e29b513a68abacf485adcd1fe8284a4316c5ae868e6633327" dependencies = [ "iana-time-zone", "num-traits", - "windows-link", + "windows-link 0.2.1", ] [[package]] @@ -460,9 +462,9 @@ checksum = "460fbee9c2c2f33933d720630a6a0bac33ba7053db5344fac858d4b8952d77d5" [[package]] name = "either" -version = "1.15.0" +version = "1.18.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "48c757948c5ede0e46177b7add2e67155f70e33c07fea8284df6576da70b3719" +checksum = "252afb9ae5eaa683babdc6a068b3f5726eb19e05070c731f9b2a23a7c3e8ed34" [[package]] name = "equivalent" @@ -497,9 +499,9 @@ dependencies = [ [[package]] name = "find-msvc-tools" -version = "0.1.9" +version = "0.1.12" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5baebc0774151f905a1a2cc41989300b1e6fbb29aff0ceffa1064fdd3088d582" +checksum = "3e0f1c7c3a72c66fd80abe965175f7523475c0489a87d3ff9d6e8c87d87a9d2d" [[package]] name = "flatbuffers" @@ -513,11 +515,10 @@ dependencies = [ [[package]] name = "flate2" -version = "1.1.9" +version = "1.1.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "843fba2746e448b37e26a819579957415c8cef339bf08564fe8b7ddbd959573c" +checksum = "6e634e2e0ebac1ee034020da1ca582e17ffe4e0f5e985823721e168928136dcb" dependencies = [ - "miniz_oxide", "zlib-rs", ] @@ -529,9 +530,9 @@ checksum = "3f9eec918d3f24069decb9af1554cad7c880e2da24a9afd88aca000531ab82c1" [[package]] name = "futures" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8b147ee9d1f6d097cef9ce628cd2ee62288d963e16fb287bd9286455b241382d" +checksum = "9a31d2a3fbaaeb2af2368bbdd904aa8e812d3c04a1ee10d3171f52d556e5d0a3" dependencies = [ "futures-channel", "futures-core", @@ -544,9 +545,9 @@ dependencies = [ [[package]] name = "futures-channel" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "07bbe89c50d7a535e539b8c17bc0b49bdb77747034daa8087407d655f3f7cc1d" +checksum = "b1f9e3d69d39e4862ffed03ed071a76f9a13ba1d9109d355b0f0aa6b15e393c4" dependencies = [ "futures-core", "futures-sink", @@ -554,15 +555,15 @@ dependencies = [ [[package]] name = "futures-core" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7e3450815272ef58cec6d564423f6e755e25379b217b0bc688e295ba24df6b1d" +checksum = "92d699e522242e69e3003b94ecc1f960f3a5e015aa7c5d7486e65ad01dd94f5e" [[package]] name = "futures-executor" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "baf29c38818342a3b26b5b923639e7b1f4a61fc5e76102d4b1981c6dc7a7579d" +checksum = "031b47cf1a3c6cc8bc2fc76cd437f521619387907d469316e7c0bc278f1f5432" dependencies = [ "futures-core", "futures-task", @@ -571,38 +572,38 @@ dependencies = [ [[package]] name = "futures-io" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cecba35d7ad927e23624b22ad55235f2239cfa44fd10428eecbeba6d6a717718" +checksum = "53c0fa8157de1303bfffdaa1cc2a673bfffb60102f76b0ef4441659124373fed" [[package]] name = "futures-macro" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e835b70203e41293343137df5c0664546da5745f82ec9b84d40be8336958447b" +checksum = "9fb9654ba8355388abeb8dcb4fc62f511300867002afc858860463bdd9fe0c44" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 3.0.5", ] [[package]] name = "futures-sink" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c39754e157331b013978ec91992bde1ac089843443c49cbc7f46150b0fad0893" +checksum = "1944426bf7d03f1d14f708785e4b33efd750b36d48a157b836b3efc15ede8e1d" [[package]] name = "futures-task" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "037711b3d59c33004d3856fbdc83b99d4ff37a24768fa1be9ce3538a1cde4393" +checksum = "cd417de3d1d015fc3bfd2b1ea46dfc7bab72ef86f1cc7cc9c78e728b34a6d1fd" [[package]] name = "futures-util" -version = "0.3.32" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "389ca41296e6190b48053de0321d02a77f32f8a5d2461dd38762c0593805c6d6" +checksum = "0d50a92467f8ba5dd6e3ee5d4bd04d73ab2e4e1c44474a0674821dfce14b79bc" dependencies = [ "futures-channel", "futures-core", @@ -651,9 +652,9 @@ dependencies = [ [[package]] name = "h2" -version = "0.4.13" +version = "0.4.19" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2f44da3a8150a6703ed5d34e164b875fd14c2cdab9af1252a9a1020bde2bdc54" +checksum = "ef8e5e5a340588f4452631496976cf8636d4a7ecf600239fdc27615d2530bc16" dependencies = [ "atomic-waker", "bytes", @@ -682,9 +683,9 @@ dependencies = [ [[package]] name = "hashbrown" -version = "0.16.1" +version = "0.17.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "841d1cc9bed7f9236f321df977030373f4a4163ae1a7dbfe1a51a2c1a51d9100" +checksum = "ed5909b6e89a2db4456e54cd5f673791d7eca6732202bbf2a9cc504fe2f9b84a" [[package]] name = "heck" @@ -694,9 +695,9 @@ checksum = "2304e00983f87ffb38b55b444b5e3b60a884b5d30c0fca7d82fe33449bbe55ea" [[package]] name = "http" -version = "1.4.0" +version = "1.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e3ba2a386d7f85a81f119ad7498ebe444d2e22c2af0b86b069416ace48b3311a" +checksum = "918d3568bebf352712bc2ef3d46a8bcf1a75b373be6539de198e9105cbbf9ce0" dependencies = [ "bytes", "itoa", @@ -704,9 +705,9 @@ dependencies = [ [[package]] name = "http-body" -version = "1.0.1" +version = "1.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1efedce1fb8e6913f23e0c92de8e62cd5b772a67e7b3946df930a62566c93184" +checksum = "ca2a8f2913ee65f60facd6a5905613afaa448497a0230cc41ce022d93290bc2c" dependencies = [ "bytes", "http", @@ -714,9 +715,9 @@ dependencies = [ [[package]] name = "http-body-util" -version = "0.1.3" +version = "0.1.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b021d93e26becf5dc7e1b75b1bed1fd93124b374ceb73f43d4d4eafec896a64a" +checksum = "23169fe34a5fbcdd3f3862e78fb9b6fccd5f02a6dc6f732547005d45631ce71c" dependencies = [ "bytes", "futures-core", @@ -739,9 +740,9 @@ checksum = "df3b46402a9d5adb4c86a0cf463f42e19994e3ee891101b1841f30a545cb49a9" [[package]] name = "hyper" -version = "1.8.1" +version = "1.11.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2ab2d4f250c3d7b1c9fcdff1cece94ea4e2dfbec68614f7b87cb205f24ca9d11" +checksum = "27b501faa50e7a26c3d3560ca625132f4078a17771f4810baf70475ae48cbe43" dependencies = [ "atomic-waker", "bytes", @@ -754,7 +755,6 @@ dependencies = [ "httpdate", "itoa", "pin-project-lite", - "pin-utils", "smallvec", "tokio", "want", @@ -787,7 +787,7 @@ dependencies = [ "hyper", "libc", "pin-project-lite", - "socket2 0.6.2", + "socket2", "tokio", "tower-service", "tracing", @@ -819,25 +819,19 @@ dependencies = [ [[package]] name = "indexmap" -version = "2.13.0" +version = "2.14.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7714e70437a7dc3ac8eb7e6f8df75fd8eb422675fc7678aff7364301092b1017" +checksum = "cc4e190f5d26ca7051642629da2c52fc03bde85a03197c99408dcd291734c855" dependencies = [ "equivalent", "hashbrown", ] -[[package]] -name = "integer-encoding" -version = "3.0.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8bb03732005da905c88227371639bf1ad885cc712789c011c31c5fb3ab3ccf02" - [[package]] name = "inventory" -version = "0.3.22" +version = "0.3.24" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "009ae045c87e7082cb72dab0ccd01ae075dd00141ddc108f43a0ea150a9e7227" +checksum = "a4f0c30c76f2f4ccee3fe55a2435f691ca00c0e4bd87abe4f4a851b1d4dac39b" dependencies = [ "rustversion", ] @@ -853,9 +847,9 @@ dependencies = [ [[package]] name = "itoa" -version = "1.0.17" +version = "1.0.18" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "92ecc6618181def0457392ccd0ee51198e065e016d1d527a7ac1b6dc7c1f09d2" +checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" [[package]] name = "jobserver" @@ -869,11 +863,12 @@ dependencies = [ [[package]] name = "js-sys" -version = "0.3.89" +version = "0.3.105" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f4eacb0641a310445a4c513f2a5e23e19952e269c6a38887254d5f837a305506" +checksum = "ce57d20d1ea864ce2ac172ab472d409214f4fd359f0b2a2775abdf522e2af99e" dependencies = [ - "once_cell", + "cfg-if", + "futures-util", "wasm-bindgen", ] @@ -936,9 +931,9 @@ dependencies = [ [[package]] name = "libc" -version = "0.2.182" +version = "0.2.189" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6800badb6cb2082ffd7b6a67e6125bb39f18782f793520caee8cb8846be06112" +checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" [[package]] name = "libloading" @@ -947,7 +942,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d7c4b02199fee7c5d21a5ae7d8cfa79a6ef5bb2fc834d6e9058e89c825efdc55" dependencies = [ "cfg-if", - "windows-link", + "windows-link 0.2.1", ] [[package]] @@ -957,7 +952,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "754ca22de805bb5744484a5b151a9e1a8e837d5dc232c2d7d8c2e3492edc8b60" dependencies = [ "cfg-if", - "windows-link", + "windows-link 0.2.1", ] [[package]] @@ -968,15 +963,15 @@ checksum = "b6d2cec3eae94f9f509c767b45932f1ada8350c4bdb85af2fcab4a3c14807981" [[package]] name = "log" -version = "0.4.29" +version = "0.4.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5e5032e24019045c762d3c0f28f5b6b8bbf38563a65908389bf7978758920897" +checksum = "f9f8bd3e56ce4dfc153cf470fffbfa98c7620958b312ca5c3a4b8d5181fd13c6" [[package]] name = "lz4_flex" -version = "0.11.6" +version = "0.14.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "373f5eceeeab7925e0c1098212f2fbc4d416adec9d35051a6ab251e824c1854a" +checksum = "ecbdfe44b1bd960b68170b417450a628c43f7cf56bb3c5317e61cb230ee7f226" dependencies = [ "twox-hash", ] @@ -989,9 +984,9 @@ checksum = "47e1ffaa40ddd1f3ed91f717a33c8c0ee23fff369e3aa8772b9605cc1d22f4c3" [[package]] name = "memchr" -version = "2.8.0" +version = "2.8.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f8ca58f447f06ed17d5fc4043ce1b10dd205e060fb3ce5b979b8ed8e59ff3f79" +checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" [[package]] name = "mime" @@ -999,46 +994,22 @@ version = "0.3.17" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "6877bb514081ee2a7ff5ef9de3281f14a4dd4bceac4c09388074a6b5df8a139a" -[[package]] -name = "miniz_oxide" -version = "0.8.9" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1fa76a2c86f704bdb222d66965fb3d63269ce38518b83cb0575fca855ebb6316" -dependencies = [ - "adler2", - "simd-adler32", -] - [[package]] name = "mio" -version = "1.1.1" +version = "1.2.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a69bcab0ad47271a0234d9422b131806bf3968021e5dc9328caf2d4cd58557fc" +checksum = "4b18443e9c262bfe8fa82f51666e2642c53393f7e5c27b3e1aeab922cff5b9d8" dependencies = [ "libc", "wasi", "windows-sys 0.61.2", ] -[[package]] -name = "num" -version = "0.4.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "35bd024e8b2ff75562e5f34e7f4905839deb4b22955ef5e73d2fea1b9813cb23" -dependencies = [ - "num-bigint", - "num-complex", - "num-integer", - "num-iter", - "num-rational", - "num-traits", -] - [[package]] name = "num-bigint" -version = "0.4.6" +version = "0.5.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a5e44f723f1133c9deac646763579fdb3ac745e418f2a7af9cd0c431da1f20b9" +checksum = "93e7820bc0a80a0238e650327316f929ba18d5be054b647490a3a6a339f3e7c0" dependencies = [ "num-integer", "num-traits", @@ -1055,35 +1026,13 @@ dependencies = [ [[package]] name = "num-integer" -version = "0.1.46" +version = "0.1.47" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7969661fd2958a5cb096e56c8e1ad0444ac2bbcd0061bd28660485a44879858f" +checksum = "7ce2d95d4b3734dc35aa2f45e1aa22cd416814592a4f9d9205e11affd5b8e10b" dependencies = [ "num-traits", ] -[[package]] -name = "num-iter" -version = "0.1.45" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1429034a0490724d0075ebb2bc9e875d6503c3cf69e235a8941aa757d83ef5bf" -dependencies = [ - "autocfg", - "num-integer", - "num-traits", -] - -[[package]] -name = "num-rational" -version = "0.4.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f83d14da390562dca69fc84082e73e548e1ad308d24accdedd2720017cb37824" -dependencies = [ - "num-bigint", - "num-integer", - "num-traits", -] - [[package]] name = "num-traits" version = "0.2.19" @@ -1096,61 +1045,50 @@ dependencies = [ [[package]] name = "once_cell" -version = "1.21.3" +version = "1.21.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "42f5e15c9953c5e4ccceeb2e7382a716482c34515315f7b03532b8b4e8393d2d" +checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" [[package]] name = "openssl-probe" -version = "0.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9f50d9b3dabb09ecd771ad0aa242ca6894994c130308ca3d7684634df8037391" - -[[package]] -name = "ordered-float" -version = "2.10.1" +version = "0.2.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "68f19d67e5a2795c94e73e0bb1cc1a7edeb2e28efd39e2e1c9b7a40c1108b11c" -dependencies = [ - "num-traits", -] +checksum = "7c87def4c32ab89d880effc9e097653c8da5d6ef28e6b539d313baaacfbafcbe" [[package]] name = "parquet" -version = "56.2.0" +version = "59.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f0dbd48ad52d7dccf8ea1b90a3ddbfaea4f69878dd7683e51c507d4bc52b5b27" +checksum = "ff322f54b1a0f9288e614ed1f2d329b380af5476420db19f46ffb865e1163d73" dependencies = [ "ahash", "arrow-array", "arrow-buffer", - "arrow-cast", "arrow-data", "arrow-ipc", "arrow-schema", "arrow-select", - "base64", + "base64 0.23.1", "bytes", "chrono", "flate2", "half", "hashbrown", "lz4_flex", - "num", "num-bigint", - "paste", + "num-integer", + "num-traits", "seq-macro", "snap", - "thrift", "twox-hash", "zstd", ] [[package]] -name = "paste" -version = "1.0.15" +name = "path-slash" +version = "0.2.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "57c0d7b74b563b49d38dae00a0c37d4d6de9b432382b2892f0574ddcae73fd0a" +checksum = "1e91099d4268b0e11973f036e885d652fb0b21fedcf69738c627f94db6a44f42" [[package]] name = "percent-encoding" @@ -1160,56 +1098,50 @@ checksum = "9b4f627cb1b25917193a259e49bdad08f671f8d9708acfd5fe0a8c1455d87220" [[package]] name = "pin-project" -version = "1.1.10" +version = "1.1.13" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "677f1add503faace112b9f1373e43e9e054bfdd22ff1a63c1bc485eaec6a6a8a" +checksum = "2466b2336ed02bcdca6b294417127b90ec92038d1d5c4fbeac971a922e0e0924" dependencies = [ "pin-project-internal", ] [[package]] name = "pin-project-internal" -version = "1.1.10" +version = "1.1.13" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6e918e4ff8c4549eb882f14b3a4bc8c8bc93de829416eacf579f1207a8fbf861" +checksum = "c96395f0a926bc13b1c17622aaddda1ecb55d49c8f1bf9777e4d877800a43f8b" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] name = "pin-project-lite" -version = "0.2.16" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3b3cff922bd51709b605d9ead9aa71031d81447142d828eb4a6eba76fe619f9b" - -[[package]] -name = "pin-utils" -version = "0.1.0" +version = "0.2.17" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8b870d8c151b6f2fb93e84a13146138f05d02ed11c7e7c54f8826aaaf7c9f184" +checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd" [[package]] name = "pkg-config" -version = "0.3.33" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "19f132c84eca552bf34cab8ec81f1c1dcc229b811638f9d283dceabe58c5569e" +checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548" [[package]] name = "proc-macro2" -version = "1.0.106" +version = "1.0.107" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8fd00f0bb2e90d81d1044c2b32617f68fcb9fa3bb7640c23e9c748e53fb30934" +checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9" dependencies = [ "unicode-ident", ] [[package]] name = "prost" -version = "0.13.5" +version = "0.14.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2796faa41db3ec313a31f7624d9286acf277b52de526150b7e69f3debf891ee5" +checksum = "528ac67416ff8646872a3c02cad9cc4ee5dc9f9540c9b10771855c95cb2e5ae1" dependencies = [ "bytes", "prost-derive", @@ -1217,31 +1149,31 @@ dependencies = [ [[package]] name = "prost-derive" -version = "0.13.5" +version = "0.14.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8a56d757972c98b346a9b766e3f02746cde6dd1cd1d1d563472929fdd74bec4d" +checksum = "b570b25f7617e43d59005d0990ccb79e950a423952cea19671b7a876da390adf" dependencies = [ "anyhow", "itertools", "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] name = "prost-types" -version = "0.13.5" +version = "0.14.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "52c2c1bf36ddb1a1c396b3601a3cec27c2462e45f07c386894ec3ccf5332bd16" +checksum = "f94967dc7688f3054c7fac87473ffae4cc4c3904800e2d9f5b857246d8963b0a" dependencies = [ "prost", ] [[package]] name = "quote" -version = "1.0.44" +version = "1.0.47" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "21b2ebcf727b7760c461f091f9f0f539b77b8e87f2fd88131e7f1b433b3cece4" +checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001" dependencies = [ "proc-macro2", ] @@ -1260,9 +1192,9 @@ checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf" [[package]] name = "regex" -version = "1.12.3" +version = "1.13.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e10754a14b9137dd7b1e3e5b0493cc9171fdd105e0ab477f51b72e7f3ac0e276" +checksum = "f020237b6c8eed93db2e2cb53c00c60a8e1bc73da7d073199a1180401450218d" dependencies = [ "aho-corasick", "memchr", @@ -1272,9 +1204,9 @@ dependencies = [ [[package]] name = "regex-automata" -version = "0.4.14" +version = "0.4.18" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6e1dd4122fc1595e8162618945476892eefca7b88c52820e74af6262213cae8f" +checksum = "ad8553b9b26413251cbf30e620595c7a41b3887f03da04579c0e6b0d6a06b4b2" dependencies = [ "aho-corasick", "memchr", @@ -1289,9 +1221,9 @@ checksum = "cab834c73d247e67f4fae452806d17d3c7501756d98c8808d7c9c7aa7d18f973" [[package]] name = "regex-syntax" -version = "0.8.9" +version = "0.8.11" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a96887878f22d7bad8a3b6dc5b7440e0ada9a245242924394987b21cf2210a4c" +checksum = "d6f6ff9a378485b298a5286656da665ba74413d36db0979633275d2e708145d4" [[package]] name = "ring" @@ -1318,9 +1250,9 @@ dependencies = [ [[package]] name = "rustler" -version = "0.37.3" +version = "0.37.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c779e2cbfa2987990205d0d8fc142163739e45a4c6592dc637896c77fec01280" +checksum = "875c8fe88089b9bbc0977385e107d35bfae740c6b0734e60a1e9cc82d0017f49" dependencies = [ "inventory", "libloading 0.9.0", @@ -1330,22 +1262,22 @@ dependencies = [ [[package]] name = "rustler_codegen" -version = "0.37.3" +version = "0.37.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e6e120f8936c779b6c2e09992a2dfa9a4e8bcd0794c02bb654fde03e03ce8c31" +checksum = "afb5848e9c4cf3796f190d9b4516523af27f3444a3af1771f20465f6586d40b2" dependencies = [ "heck", "inventory", "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] name = "rustls" -version = "0.23.36" +version = "0.23.44" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c665f33d38cea657d9614f766881e4d510e0eda4239891eea56b4cadcf01801b" +checksum = "6725596c3f2c3a0aef021139e145d4eafe314a6623e4680ca83852b2c67ab2ba" dependencies = [ "log", "once_cell", @@ -1358,9 +1290,9 @@ dependencies = [ [[package]] name = "rustls-native-certs" -version = "0.8.3" +version = "0.8.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "612460d5f7bea540c490b2b6395d8e34a953e52b491accd6c86c8164c5932a63" +checksum = "dab5152771c58876a2146916e53e35057e1a4dfa2b9df0f0305b07f611fdea4d" dependencies = [ "openssl-probe", "rustls-pki-types", @@ -1370,18 +1302,18 @@ dependencies = [ [[package]] name = "rustls-pki-types" -version = "1.13.2" +version = "1.15.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "21e6f2ab2928ca4291b86736a8bd920a277a399bba1589409d72154ff87c1282" +checksum = "2f4925028c7eb5d1fcdaf196971378ed9d2c1c4efc7dc5d011256f76c99c0a96" dependencies = [ "zeroize", ] [[package]] name = "rustls-webpki" -version = "0.103.8" +version = "0.103.15" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2ffdfa2f5286e2247234e03f680868ac2815974dc39e00ea15adc445d0aafe52" +checksum = "f3c3cf1d8b1e7d4927e2d154c3fcb02979afb9939629c62cd9048d4f07b60ac2" dependencies = [ "ring", "rustls-pki-types", @@ -1390,9 +1322,9 @@ dependencies = [ [[package]] name = "rustversion" -version = "1.0.22" +version = "1.0.23" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b39cdef0fa800fc44525c84ccb54a029961a8215f9619753635a9c0d2538d46d" +checksum = "cf54715a573b99ac80df0bc206da022bcd442c974952c7b9720069370852e21f" [[package]] name = "ryu" @@ -1402,18 +1334,18 @@ checksum = "9774ba4a74de5f7b1c1451ed6cd5285a32eddb5cccb8cc655a4e50009e06477f" [[package]] name = "schannel" -version = "0.1.28" +version = "0.1.29" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "891d81b926048e76efe18581bf793546b4c0eaf8448d72be8de2bbee5fd166e1" +checksum = "91c1b7e4904c873ef0710c1f407dde2e6287de2bebc1bbbf7d430bb7cbffd939" dependencies = [ "windows-sys 0.61.2", ] [[package]] name = "security-framework" -version = "3.5.1" +version = "3.7.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b3297343eaf830f66ede390ea39da1d462b6b0c1b000f420d0a83f898bbbe6ef" +checksum = "b7f4bc775c73d9a02cde8bf7b2ec4c9d12743edf609006c7facc23998404cd1d" dependencies = [ "bitflags", "core-foundation", @@ -1424,9 +1356,9 @@ dependencies = [ [[package]] name = "security-framework-sys" -version = "2.15.0" +version = "2.17.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cc1f0cbffaac4852523ce30d8bd3c5cdc873501d96ff467ca09b6767bb8cd5c0" +checksum = "6ce2691df843ecc5d231c0b14ece2acc3efb62c0a398c7e1d875f3983ce020e3" dependencies = [ "core-foundation-sys", "libc", @@ -1434,9 +1366,9 @@ dependencies = [ [[package]] name = "semver" -version = "1.0.27" +version = "1.0.28" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d767eb0aabc880b29956c35734170f26ed551a859dbd361d140cdbeca61ab1e2" +checksum = "8a7852d02fc848982e0c167ef163aaff9cd91dc640ba85e263cb1ce46fae51cd" [[package]] name = "seq-macro" @@ -1446,44 +1378,38 @@ checksum = "1bc711410fbe7399f390ca1c3b60ad0f53f80e95c5eb935e52268a0e2cd49acc" [[package]] name = "serde_core" -version = "1.0.228" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "41d385c7d4ca58e59fc732af25c3983b67ac852c1a25000afe1175de458b67ad" +checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" dependencies = [ "serde_derive", ] [[package]] name = "serde_derive" -version = "1.0.228" +version = "1.0.229" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d540f220d3187173da220f885ab66608367b6574e925011a9353e4badda91d79" +checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 3.0.5", ] [[package]] name = "serde_spanned" -version = "1.0.4" +version = "1.1.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f8bbf91e5a4d6315eee45e704372590b30e260ee83af6639d64557f51b067776" +checksum = "6662b5879511e06e8999a8a235d848113e942c9124f211511b16466ee2995f26" dependencies = [ "serde_core", ] [[package]] name = "shlex" -version = "1.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0fda2ff0d084019ba4d7c6f371c95d8fd75ce3524c3cb8fb653a3023f6323e64" - -[[package]] -name = "simd-adler32" -version = "0.3.9" +version = "2.0.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "703d5c7ef118737c72f1af64ad2f6f8c5e1921f818cdcb97b8fe6fc69bf66214" +checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba" [[package]] name = "slab" @@ -1493,34 +1419,24 @@ checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5" [[package]] name = "smallvec" -version = "1.15.1" +version = "1.16.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "67b1b7a3b5fe4f1376887184045fcf45c69e92af734b7aaddc05fb777b6fbd03" +checksum = "ba467056f1b547ed52077911161fc86985becbc60e8e1857c8a144dab0def891" [[package]] name = "snap" -version = "1.1.1" +version = "1.1.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1b6b67fb9a61334225b5b790716f609cd58395f895b3fe8b328786812a40bc3b" +checksum = "199905e6153d6405f9728fe44daace35f8f837bbf830bb6e85fbd5828709a886" [[package]] name = "socket2" -version = "0.5.10" +version = "0.6.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e22376abed350d73dd1cd119b57ffccad95b4e585a7cda43e286245ce23c0678" +checksum = "c3d1e2c7f27f8d4cb10542a02c49005dbd6e93095799d6f3be745fae9f8fedd4" dependencies = [ "libc", - "windows-sys 0.52.0", -] - -[[package]] -name = "socket2" -version = "0.6.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "86f4aa3ad99f2088c990dfa82d367e19cb29268ed67c574d10d0a4bfe71f07e0" -dependencies = [ - "libc", - "windows-sys 0.60.2", + "windows-sys 0.61.2", ] [[package]] @@ -1531,9 +1447,9 @@ checksum = "13c2bddecc57b384dee18652358fb23172facb8a2c51ccc10d74c157bdea3292" [[package]] name = "syn" -version = "2.0.117" +version = "2.0.119" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e665b8803e7b1d2a727f4023456bbbbe74da67099c585258af0ad9c5013b9b99" +checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297" dependencies = [ "proc-macro2", "quote", @@ -1541,21 +1457,21 @@ dependencies = [ ] [[package]] -name = "sync_wrapper" -version = "1.0.2" +name = "syn" +version = "3.0.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0bf256ce5efdfa370213c1dabab5935a12e49f2c58d15e9eac2870d3b4f27263" +checksum = "12df2e0110f65b775f769bb17ef989067a1d931b2eb822bd4346631eeada89f9" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] [[package]] -name = "thrift" -version = "0.17.0" +name = "sync_wrapper" +version = "1.0.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7e54bc85fc7faa8bc175c4bab5b92ba8d9a3ce893d0e9f42cc455c8ab16a9e09" -dependencies = [ - "byteorder", - "integer-encoding", - "ordered-float", -] +checksum = "0bf256ce5efdfa370213c1dabab5935a12e49f2c58d15e9eac2870d3b4f27263" [[package]] name = "tiny-keccak" @@ -1568,35 +1484,35 @@ dependencies = [ [[package]] name = "tokio" -version = "1.49.0" +version = "1.53.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "72a2903cd7736441aac9df9d7688bd0ce48edccaadf181c3b90be801e81d3d86" +checksum = "202caea871b69668250d242070849eb495be178ed697a3e98aebce5bc81a0bed" dependencies = [ "bytes", "libc", "mio", "pin-project-lite", - "socket2 0.6.2", + "socket2", "tokio-macros", "windows-sys 0.61.2", ] [[package]] name = "tokio-macros" -version = "2.6.0" +version = "2.7.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "af407857209536a95c8e56f8231ef2c2e2aff839b22e07a1ffcbc617e9db9fa5" +checksum = "78773a2a397f451582ce068015985c33193cf6dea8b74d2a639fe457b2f07b0e" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 3.0.5", ] [[package]] name = "tokio-rustls" -version = "0.26.4" +version = "0.26.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1729aa945f29d91ba541258c8df89027d5792d85a8841fb65e8bf0f4ede4ef61" +checksum = "b0c85f2c3ef0b1cd58b36682f4b17aaa995f0e5db534d85692b4903abce21f67" dependencies = [ "rustls", "tokio", @@ -1604,9 +1520,9 @@ dependencies = [ [[package]] name = "tokio-stream" -version = "0.1.18" +version = "0.1.19" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "32da49809aab5c3bc678af03902d4ccddea2a87d028d86392a4b1560c6906c70" +checksum = "a3d06f0b082ba57c26b79407372e57cf2a1e28124f78e9479fe80322cf53420b" dependencies = [ "futures-core", "pin-project-lite", @@ -1615,22 +1531,23 @@ dependencies = [ [[package]] name = "tokio-util" -version = "0.7.18" +version = "0.7.19" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9ae9cec805b01e8fc3fd2fe289f89149a9b66dd16786abd8b19cfa7b48cb0098" +checksum = "494815d09bf52b5548659851081238f0ca39ff638363907596da739561c62c52" dependencies = [ "bytes", "futures-core", "futures-sink", + "libc", "pin-project-lite", "tokio", ] [[package]] name = "toml" -version = "0.9.12+spec-1.1.0" +version = "1.1.6+spec-1.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cf92845e79fc2e2def6a5d828f0801e29a2f8acc037becc5ab08595c7d5e9863" +checksum = "920602543f0911ab71da12c50d59701da54c196d1a2bf5cb4b75667f137a406a" dependencies = [ "serde_spanned", "toml_datetime", @@ -1641,37 +1558,37 @@ dependencies = [ [[package]] name = "toml_datetime" -version = "0.7.5+spec-1.1.0" +version = "1.1.1+spec-1.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "92e1cfed4a3038bc5a127e35a2d360f145e1f4b971b551a2ba5fd7aedf7e1347" +checksum = "3165f65f62e28e0115a00b2ebdd37eb6f3b641855f9d636d3cd4103767159ad7" dependencies = [ "serde_core", ] [[package]] name = "toml_parser" -version = "1.0.9+spec-1.1.0" +version = "1.1.3+spec-1.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "702d4415e08923e7e1ef96cd5727c0dfed80b4d2fa25db9647fe5eb6f7c5a4c4" +checksum = "1d38ac1cf9b95face32296c0a3ede1fdc270627c9d9c02a7274dd6d960dc4d56" dependencies = [ "winnow", ] [[package]] name = "toml_writer" -version = "1.0.6+spec-1.1.0" +version = "1.1.2+spec-1.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ab16f14aed21ee8bfd8ec22513f7287cd4a91aa92e44edfe2c17ddd004e92607" +checksum = "7d56353a2a665ad0f41a421187180aab746c8c325620617ad883a99a1cbe66d2" [[package]] name = "tonic" -version = "0.13.1" +version = "0.14.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7e581ba15a835f4d9ea06c55ab1bd4dce26fc53752c69a04aac00703bfb49ba9" +checksum = "ac2a5518c70fa84342385732db33fb3f44bc4cc748936eb5833d2df34d6445ef" dependencies = [ "async-trait", "axum", - "base64", + "base64 0.22.1", "bytes", "h2", "http", @@ -1682,9 +1599,9 @@ dependencies = [ "hyper-util", "percent-encoding", "pin-project", - "prost", "rustls-native-certs", - "socket2 0.5.10", + "socket2", + "sync_wrapper", "tokio", "tokio-rustls", "tokio-stream", @@ -1694,6 +1611,17 @@ dependencies = [ "tracing", ] +[[package]] +name = "tonic-prost" +version = "0.14.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "50849f68853be452acf590cde0b146665b8d507b3b8af17261df47e02c209ea0" +dependencies = [ + "bytes", + "prost", + "tonic", +] + [[package]] name = "tower" version = "0.5.3" @@ -1744,7 +1672,7 @@ checksum = "7490cfa5ec963746568740651ac6781f701c9c5ea257c58e057f3ba8cf69e8da" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -1764,9 +1692,9 @@ checksum = "e421abadd41a4225275504ea4d6566923418b7f05506fbc9c0fe86ba7396114b" [[package]] name = "twox-hash" -version = "2.1.2" +version = "2.1.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9ea3136b675547379c4bd395ca6b938e5ad3c3d20fad76e7fe85f9e0d011419c" +checksum = "5283634e518fe9e82c7b20520bb4bc209009fd16c82077c802f8111ecbb0117a" [[package]] name = "unicode-ident" @@ -1803,18 +1731,18 @@ checksum = "ccf3ec651a847eb01de73ccad15eb7d99f80485de043efb2f370cd654f4ea44b" [[package]] name = "wasip2" -version = "1.0.2+wasi-0.2.9" +version = "1.0.4+wasi-0.2.12" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9517f9239f02c069db75e65f174b3da828fe5f5b945c4dd26bd25d89c03ebcf5" +checksum = "b67efb37e106e55ce722a510d6b5f9c17f083e5fc79afc2badeb12cc313d9487" dependencies = [ "wit-bindgen", ] [[package]] name = "wasm-bindgen" -version = "0.2.112" +version = "0.2.128" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "05d7d0fce354c88b7982aec4400b3e7fcf723c32737cef571bd165f7613557ee" +checksum = "aecb87a33d3b0c5e3b7aa46336eaf486cffafbd281b195e4c8b80d50df2351bf" dependencies = [ "cfg-if", "once_cell", @@ -1825,9 +1753,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-macro" -version = "0.2.112" +version = "0.2.128" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "55839b71ba921e4f75b674cb16f843f4b1f3b26ddfcb3454de1cf65cc021ec0f" +checksum = "a690d511e3c1a8b3a55e33511e3c2c00c78415cd23650f32b808627f5696b9ed" dependencies = [ "quote", "wasm-bindgen-macro-support", @@ -1835,22 +1763,22 @@ dependencies = [ [[package]] name = "wasm-bindgen-macro-support" -version = "0.2.112" +version = "0.2.128" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "caf2e969c2d60ff52e7e98b7392ff1588bffdd1ccd4769eba27222fd3d621571" +checksum = "411e4887f0071ef2d2164a9d5fdf2d20efbef78fccd3a78b0c10a1dc5295e48a" dependencies = [ "bumpalo", "proc-macro2", "quote", - "syn", + "syn 3.0.5", "wasm-bindgen-shared", ] [[package]] name = "wasm-bindgen-shared" -version = "0.2.112" +version = "0.2.128" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0861f0dcdf46ea819407495634953cdcc8a8c7215ab799a7a7ce366be71c7b30" +checksum = "81941cd78d0c92026c33e5e01312845a4cb1e9af3407f9134b100dd03144103e" dependencies = [ "unicode-ident", ] @@ -1863,9 +1791,9 @@ checksum = "b8e83a14d34d0623b51dce9581199302a221863196a1dde71a7663a4c2be9deb" dependencies = [ "windows-implement", "windows-interface", - "windows-link", - "windows-result", - "windows-strings", + "windows-link 0.2.1", + "windows-result 0.4.1", + "windows-strings 0.5.1", ] [[package]] @@ -1876,7 +1804,7 @@ checksum = "053e2e040ab57b9dc951b72c264860db7eb3b0200ba345b4e4c3b14f67855ddf" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -1887,7 +1815,7 @@ checksum = "3f316c4a2570ba26bbec722032c4099d8c8bc095efccdc15688708623367e358" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] @@ -1896,15 +1824,21 @@ version = "0.2.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f0805222e57f7521d6a62e36fa9163bc891acd422f971defe97d64e70d0a4fe5" +[[package]] +name = "windows-link" +version = "0.100.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9b39de7c7fe78858b0144c50c3fc000bab3be17d5e9b85b1053871693c7f7415" + [[package]] name = "windows-registry" -version = "0.6.1" +version = "0.100.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "02752bf7fbdcce7f2a27a742f798510f3e5ad88dbe84871e5168e2120c3d5720" +checksum = "a14b75dbbc0f2bb1ebbf6dcb6dc0035ae17f10ec280045d10fa27ad8874903b2" dependencies = [ - "windows-link", - "windows-result", - "windows-strings", + "windows-link 0.100.0", + "windows-result 0.100.0", + "windows-strings 0.100.0", ] [[package]] @@ -1913,7 +1847,16 @@ version = "0.4.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7781fa89eaf60850ac3d2da7af8e5242a5ea78d1a11c49bf2910bb5a73853eb5" dependencies = [ - "windows-link", + "windows-link 0.2.1", +] + +[[package]] +name = "windows-result" +version = "0.100.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "44873a1ef61c85d5e2991db91fe656cfcf0731097e00dff1ac860f1cb3b752a0" +dependencies = [ + "windows-link 0.100.0", ] [[package]] @@ -1922,25 +1865,25 @@ version = "0.5.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7837d08f69c77cf6b07689544538e017c1bfcf57e34b4c0ff58e6c2cd3b37091" dependencies = [ - "windows-link", + "windows-link 0.2.1", ] [[package]] -name = "windows-sys" -version = "0.52.0" +name = "windows-strings" +version = "0.100.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "282be5f36a8ce781fad8c8ae18fa3f9beff57ec1b52cb3de0789201425d9a33d" +checksum = "6a70b590c684f1d2854f74999896f7c0558f1bdde62f599088599e20ea1d2f60" dependencies = [ - "windows-targets 0.52.6", + "windows-link 0.100.0", ] [[package]] name = "windows-sys" -version = "0.60.2" +version = "0.52.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f2f500e4d28234f72040990ec9d39e3a6b950f9f22d3dba18416c35882612bcb" +checksum = "282be5f36a8ce781fad8c8ae18fa3f9beff57ec1b52cb3de0789201425d9a33d" dependencies = [ - "windows-targets 0.53.5", + "windows-targets", ] [[package]] @@ -1949,7 +1892,7 @@ version = "0.61.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ae137229bcbd6cdf0f7b80a31df61766145077ddf49416a728b02cb3921ff3fc" dependencies = [ - "windows-link", + "windows-link 0.2.1", ] [[package]] @@ -1958,31 +1901,14 @@ version = "0.52.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9b724f72796e036ab90c1021d4780d4d3d648aca59e491e6b98e725b84e99973" dependencies = [ - "windows_aarch64_gnullvm 0.52.6", - "windows_aarch64_msvc 0.52.6", - "windows_i686_gnu 0.52.6", - "windows_i686_gnullvm 0.52.6", - "windows_i686_msvc 0.52.6", - "windows_x86_64_gnu 0.52.6", - "windows_x86_64_gnullvm 0.52.6", - "windows_x86_64_msvc 0.52.6", -] - -[[package]] -name = "windows-targets" -version = "0.53.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4945f9f551b88e0d65f3db0bc25c33b8acea4d9e41163edf90dcd0b19f9069f3" -dependencies = [ - "windows-link", - "windows_aarch64_gnullvm 0.53.1", - "windows_aarch64_msvc 0.53.1", - "windows_i686_gnu 0.53.1", - "windows_i686_gnullvm 0.53.1", - "windows_i686_msvc 0.53.1", - "windows_x86_64_gnu 0.53.1", - "windows_x86_64_gnullvm 0.53.1", - "windows_x86_64_msvc 0.53.1", + "windows_aarch64_gnullvm", + "windows_aarch64_msvc", + "windows_i686_gnu", + "windows_i686_gnullvm", + "windows_i686_msvc", + "windows_x86_64_gnu", + "windows_x86_64_gnullvm", + "windows_x86_64_msvc", ] [[package]] @@ -1991,139 +1917,91 @@ version = "0.52.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "32a4622180e7a0ec044bb555404c800bc9fd9ec262ec147edd5989ccd0c02cd3" -[[package]] -name = "windows_aarch64_gnullvm" -version = "0.53.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a9d8416fa8b42f5c947f8482c43e7d89e73a173cead56d044f6a56104a6d1b53" - [[package]] name = "windows_aarch64_msvc" version = "0.52.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "09ec2a7bb152e2252b53fa7803150007879548bc709c039df7627cabbd05d469" -[[package]] -name = "windows_aarch64_msvc" -version = "0.53.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b9d782e804c2f632e395708e99a94275910eb9100b2114651e04744e9b125006" - [[package]] name = "windows_i686_gnu" version = "0.52.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "8e9b5ad5ab802e97eb8e295ac6720e509ee4c243f69d781394014ebfe8bbfa0b" -[[package]] -name = "windows_i686_gnu" -version = "0.53.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "960e6da069d81e09becb0ca57a65220ddff016ff2d6af6a223cf372a506593a3" - [[package]] name = "windows_i686_gnullvm" version = "0.52.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "0eee52d38c090b3caa76c563b86c3a4bd71ef1a819287c19d586d7334ae8ed66" -[[package]] -name = "windows_i686_gnullvm" -version = "0.53.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fa7359d10048f68ab8b09fa71c3daccfb0e9b559aed648a8f95469c27057180c" - [[package]] name = "windows_i686_msvc" version = "0.52.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "240948bc05c5e7c6dabba28bf89d89ffce3e303022809e73deaefe4f6ec56c66" -[[package]] -name = "windows_i686_msvc" -version = "0.53.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1e7ac75179f18232fe9c285163565a57ef8d3c89254a30685b57d83a38d326c2" - [[package]] name = "windows_x86_64_gnu" version = "0.52.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "147a5c80aabfbf0c7d901cb5895d1de30ef2907eb21fbbab29ca94c5b08b1a78" -[[package]] -name = "windows_x86_64_gnu" -version = "0.53.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9c3842cdd74a865a8066ab39c8a7a473c0778a3f29370b5fd6b4b9aa7df4a499" - [[package]] name = "windows_x86_64_gnullvm" version = "0.52.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "24d5b23dc417412679681396f2b49f3de8c1473deb516bd34410872eff51ed0d" -[[package]] -name = "windows_x86_64_gnullvm" -version = "0.53.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0ffa179e2d07eee8ad8f57493436566c7cc30ac536a3379fdf008f47f6bb7ae1" - [[package]] name = "windows_x86_64_msvc" version = "0.52.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "589f6da84c646204747d1270a2a5661ea66ed1cced2631d546fdfb155959f9ec" -[[package]] -name = "windows_x86_64_msvc" -version = "0.53.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d6bbff5f0aada427a1e5a6da5f1f98158182f26556f345ac9e04d36d0ebed650" - [[package]] name = "winnow" -version = "0.7.14" +version = "1.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5a5364e9d77fcdeeaa6062ced926ee3381faa2ee02d3eb83a5c27a8825540829" +checksum = "23b97319f7b8343df12cc98938e5c3eb436064524c8d2b4e30a1d3a36eecdf81" [[package]] name = "wit-bindgen" -version = "0.51.0" +version = "0.57.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d7249219f66ced02969388cf2bb044a09756a083d0fab1e566056b04d9fbcaa5" +checksum = "1ebf944e87a7c253233ad6766e082e3cd714b5d03812acc24c318f549614536e" [[package]] name = "zerocopy" -version = "0.8.39" +version = "0.8.57" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "db6d35d663eadb6c932438e763b262fe1a70987f9ae936e60158176d710cae4a" +checksum = "d35102a9f36d089ccae9e4c6802bc118be4487b80aaffc0ab4e0cf5ce92d2873" dependencies = [ "zerocopy-derive", ] [[package]] name = "zerocopy-derive" -version = "0.8.39" +version = "0.8.57" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4122cd3169e94605190e77839c9a40d40ed048d305bfdc146e7df40ab0f3e517" +checksum = "146c01f5ab44258da43cf276c74a2763db2ff3969c9c652c3f2de07041d0b2bc" dependencies = [ "proc-macro2", "quote", - "syn", + "syn 2.0.119", ] [[package]] name = "zeroize" -version = "1.8.2" +version = "1.9.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b97154e67e32c85465826e8bcc1c59429aaaf107c1e4a9e53c8d8ccd5eff88d0" +checksum = "e13c156562582aa81c60cb29407084cdb54c4164760106ab78e6c5b0858cf64e" [[package]] name = "zlib-rs" -version = "0.6.6" +version = "0.6.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b142a20ec14a91d5bc708c1dc21b080c550113d8aa77afa29635673a65dd02c5" +checksum = "34b31d188d9d685a4f9c7b46d6e36631b07058d2cfe190267adce54dc230bf12" [[package]] name = "zstd" @@ -2136,18 +2014,18 @@ dependencies = [ [[package]] name = "zstd-safe" -version = "7.2.4" +version = "7.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8f49c4d5f0abb602a93fb8736af2a4f4dd9512e36f7f570d66e65ff867ed3b9d" +checksum = "64d80649ab6db9d9f6f9c80a40becd948eda4714a0a5ac8c4d157a32231c7882" dependencies = [ "zstd-sys", ] [[package]] name = "zstd-sys" -version = "2.0.16+zstd.1.5.7" +version = "2.1.0+zstd.1.5.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "91e19ebc2adc8f83e43039e79776e3fda8ca919132d68a1fed6a5faca2683748" +checksum = "0ef0a8027ec3ee71300ab3bcbcd0393f434aa72b91ca6d635a39941deae8eea0" dependencies = [ "cc", "pkg-config", diff --git a/native/ex_arrow_native/Cargo.toml b/native/ex_arrow_native/Cargo.toml index 6cb1038..0faa5df 100644 --- a/native/ex_arrow_native/Cargo.toml +++ b/native/ex_arrow_native/Cargo.toml @@ -12,24 +12,24 @@ crate-type = ["cdylib"] [dependencies] rustler = { version = "0.37", default-features = false, features = ["derive"] } -arrow = { version = "56", default-features = false, features = ["ipc", "ffi"] } -arrow-ipc = "56" -arrow-schema = { version = "56", default-features = false } -arrow-array = { version = "56", default-features = false } -arrow-buffer = { version = "56", default-features = false } -arrow-data = { version = "56", default-features = false } -arrow-select = { version = "56", default-features = false } -arrow-ord = { version = "56", default-features = false } -arrow-arith = { version = "56", default-features = false } -parquet = { version = "56", default-features = false, features = ["arrow", "snap", "zstd", "lz4", "flate2-zlib-rs"] } -arrow-flight = { version = "56", features = ["flight-sql"] } +arrow = { version = "59", default-features = false, features = ["ipc", "ffi"] } +arrow-ipc = "59" +arrow-schema = { version = "59", default-features = false } +arrow-array = { version = "59", default-features = false } +arrow-buffer = { version = "59", default-features = false } +arrow-data = { version = "59", default-features = false } +arrow-select = { version = "59", default-features = false } +arrow-ord = { version = "59", default-features = false } +arrow-arith = { version = "59", default-features = false } +parquet = { version = "59", default-features = false, features = ["arrow", "snap", "zstd", "lz4", "flate2-zlib-rs"] } +arrow-flight = { version = "59", features = ["flight-sql"] } futures = "0.3" tokio = { version = "1", features = ["rt-multi-thread", "net", "sync", "time", "macros"] } tokio-stream = "0.1" -tonic = { version = "0.13", features = ["tls-ring", "tls-native-roots"] } +tonic = { version = "0.14", features = ["tls-ring", "tls-native-roots"] } bytes = "1" -adbc_driver_manager = "0.22" -adbc_core = "0.22" +adbc_driver_manager = "0.24" +adbc_core = "0.24" # Default to lowest NIF version we support so CI can build 2.15 (no extra flag) and 2.16 (action adds --features nif_version_2_16). [features] diff --git a/native/ex_arrow_native/src/flight_sql.rs b/native/ex_arrow_native/src/flight_sql.rs index 75fe05b..cd1ecab 100644 --- a/native/ex_arrow_native/src/flight_sql.rs +++ b/native/ex_arrow_native/src/flight_sql.rs @@ -416,7 +416,7 @@ pub fn flight_sql_query<'a>( // Step 1: GetFlightInfo → FlightInfo let flight_info = match rt.block_on(guard.execute(sql, None)) { Ok(info) => info, - Err(e) => return arrow_error_to_term(env, &e), + Err(e) => return flight_error_to_term(env, e), }; // Step 2: Enforce single-endpoint constraint @@ -445,7 +445,7 @@ pub fn flight_sql_query<'a>( // Step 5: DoGet → FlightRecordBatchStream (guard still held) let stream = match rt.block_on(guard.do_get(ticket)) { Ok(s) => s, - Err(e) => return arrow_error_to_term(env, &e), + Err(e) => return flight_error_to_term(env, e), }; // Release the lock; the stream owns its own connection internally. drop(guard); @@ -478,7 +478,7 @@ pub fn flight_sql_execute<'a>( match rt.block_on(guard.execute_update(sql, None)) { Ok(n) if n < 0 => (ok(), unknown()).encode(env), Ok(n) => (ok(), n as u64).encode(env), - Err(e) => arrow_error_to_term(env, &e), + Err(e) => flight_error_to_term(env, e), } } @@ -568,7 +568,7 @@ fn flight_info_to_stream<'a>( }; match rt.block_on(guard.do_get(ticket)) { Ok(s) => s, - Err(e) => return arrow_error_to_term(env, &e), + Err(e) => return flight_error_to_term(env, e), } }; @@ -617,7 +617,7 @@ pub fn flight_sql_get_tables<'a>( }; match rt.block_on(guard.get_tables(cmd)) { Ok(info) => info, - Err(e) => return arrow_error_to_term(env, &e), + Err(e) => return flight_error_to_term(env, e), } }; @@ -650,7 +650,7 @@ pub fn flight_sql_get_db_schemas<'a>( }; match rt.block_on(guard.get_db_schemas(cmd)) { Ok(info) => info, - Err(e) => return arrow_error_to_term(env, &e), + Err(e) => return flight_error_to_term(env, e), } }; @@ -679,7 +679,7 @@ pub fn flight_sql_get_sql_info<'a>( }; match rt.block_on(guard.get_sql_info(vec![])) { Ok(info) => info, - Err(e) => return arrow_error_to_term(env, &e), + Err(e) => return flight_error_to_term(env, e), } }; @@ -715,7 +715,7 @@ pub fn flight_sql_prepare<'a>( }; match rt.block_on(guard.prepare(sql, None)) { Ok(s) => s, - Err(e) => return arrow_error_to_term(env, &e), + Err(e) => return flight_error_to_term(env, e), } }; @@ -756,7 +756,7 @@ pub fn flight_sql_prepared_bind<'a>( match stmt.set_parameters(batch_ref.batch.clone()) { Ok(()) => ok_encode(env, rustler::types::atom::ok()), - Err(e) => arrow_error_to_term(env, &e), + Err(e) => flight_error_to_term(env, e), } } @@ -789,7 +789,7 @@ pub fn flight_sql_prepared_parameter_schema<'a>( }; ok_encode(env, ResourceArc::new(schema_ref)) } - Err(e) => arrow_error_to_term(env, &e), + Err(e) => flight_error_to_term(env, e), } } @@ -817,7 +817,7 @@ pub fn flight_sql_prepared_execute<'a>( }; match rt.block_on(stmt.execute()) { Ok(info) => info, - Err(e) => return arrow_error_to_term(env, &e), + Err(e) => return flight_error_to_term(env, e), } }; @@ -847,7 +847,7 @@ pub fn flight_sql_prepared_execute_update<'a>( match rt.block_on(stmt.execute_update()) { Ok(n) if n < 0 => (ok(), unknown()).encode(env), Ok(n) => (ok(), n as u64).encode(env), - Err(e) => arrow_error_to_term(env, &e), + Err(e) => flight_error_to_term(env, e), } } @@ -898,6 +898,6 @@ pub fn flight_sql_prepared_close<'a>( // ActionClosePreparedStatement to the server. match rt.block_on(stmt.close()) { Ok(()) => ok_encode(env, rustler::types::atom::ok()), - Err(e) => arrow_error_to_term(env, &e), + Err(e) => flight_error_to_term(env, e), } } diff --git a/native/ex_arrow_native/src/parquet.rs b/native/ex_arrow_native/src/parquet.rs index 0a5364c..f1e1bbc 100644 --- a/native/ex_arrow_native/src/parquet.rs +++ b/native/ex_arrow_native/src/parquet.rs @@ -331,7 +331,7 @@ fn build_writer_props(opts: &WriteOpts) -> WriterProperties { b = b.set_compression(c); } if let Some(n) = opts.row_group_size { - b = b.set_max_row_group_size(n); + b = b.set_max_row_group_row_count(Some(n)); } if let Some(d) = opts.dictionary { b = b.set_dictionary_enabled(d); @@ -952,7 +952,6 @@ fn encode_metadata<'a>(env: Env<'a>, metadata: &parquet::file::metadata::Parquet }; let encoding_names: Vec = col .encodings() - .iter() .map(|e| format!("{e:?}")) .collect(); let col_map = rustler::types::map::map_new(env); From a948efcefe4a6bd81eadeafe31bc646ba02ae5e2 Mon Sep 17 00:00:00 2001 From: thanos Date: Sat, 12 Sep 2026 12:06:54 -0400 Subject: [PATCH 02/11] M6: add RecordBatch.from_lists/1 and from_map/1 - closed #270 - closed #271 - closed #272 - closed #273 - closed #274 --- lib/ex_arrow/record_batch.ex | 370 +++++++++++++++++- .../record_batch_from_lists_property_test.exs | 51 +++ .../ex_arrow/record_batch_from_lists_test.exs | 180 +++++++++ 3 files changed, 598 insertions(+), 3 deletions(-) create mode 100644 test/ex_arrow/record_batch_from_lists_property_test.exs create mode 100644 test/ex_arrow/record_batch_from_lists_test.exs diff --git a/lib/ex_arrow/record_batch.ex b/lib/ex_arrow/record_batch.ex index ae834ab..44e851c 100644 --- a/lib/ex_arrow/record_batch.ex +++ b/lib/ex_arrow/record_batch.ex @@ -87,9 +87,9 @@ defmodule ExArrow.RecordBatch do ## Nullability - `from_columns/4` produces non-nullable columns (`Field.nullable = false`). - Pass nulls by binding a separate column or by using a parameter schema - that accepts non-null values only. + `from_columns/4` and `from_lists/1` produce non-nullable columns + (`Field.nullable = false`). `from_lists/1` rejects `nil` cells in 0.9; + null-bitmap support arrives with the core-model release. """ alias ExArrow.Native alias ExArrow.Schema @@ -160,6 +160,81 @@ defmodule ExArrow.RecordBatch do batch |> schema() |> Schema.field_names() end + @doc """ + Create a `RecordBatch` from named columns of Elixir lists. + + Each column is a `{name, dtype, values}` triple. `name` may be a string or + atom. `dtype` may be a `from_columns/4` dtype string (`"s64"`, `"utf8"`, …) + or the matching atom (`:s64`, `:utf8`, …). `values` is a list of scalar + cells; all columns must have the same length. + + Supports every dtype accepted by `from_columns/4`. Temporal and integer + dtypes expect integer cells (days/ticks as in the wire format). Float + dtypes accept integers or floats. Boolean expects `true`/`false`. Utf8 + expects valid UTF-8 binaries; `binary` / `large_binary` accept any binary. + + `nil` cells are rejected in 0.9 (no null-bitmap encoding yet). Nested + lists, maps, and tuples as cells are rejected. + + Packs into the `from_columns/4` wire format and reuses that NIF path. + + ## Examples + + {:ok, batch} = + ExArrow.RecordBatch.from_lists([ + {"id", :s64, [1, 2, 3]}, + {"name", :utf8, ["a", "b", "c"]} + ]) + + {:error, _} = + ExArrow.RecordBatch.from_lists([{"x", :s64, [1, nil]}]) + """ + @spec from_lists([{String.t() | atom(), atom() | String.t(), list()}]) :: + {:ok, t()} | {:error, String.t()} + def from_lists(columns) when is_list(columns) do + with :ok <- validate_from_lists_shape(columns), + {:ok, names, dtypes, binaries, length} <- pack_from_lists(columns) do + from_columns(names, binaries, dtypes, length) + end + end + + def from_lists(_), do: {:error, "from_lists/1 expects a list of {name, dtype, values} triples"} + + @doc """ + Create a `RecordBatch` from a map of column name => value list. + + Keys are sorted lexicographically (after converting atom keys to strings) + so the resulting schema order is stable. Value lists must all have the + same length. + + Dtypes are inferred from the cells of each column: + + | Cells | Dtype | + |-------------------------------|--------| + | integers | `s64` | + | floats (or mix with integers) | `f64` | + | booleans | `bool` | + | binaries (valid UTF-8) | `utf8` | + + Empty columns and mixed incompatible cell types return `{:error, message}`. + For explicit dtypes use `from_lists/1`. + + ## Examples + + {:ok, batch} = + ExArrow.RecordBatch.from_map(%{"id" => [1, 2], "name" => ["a", "b"]}) + """ + @spec from_map(%{optional(String.t() | atom()) => list()}) :: + {:ok, t()} | {:error, String.t()} + def from_map(map) when is_map(map) and map_size(map) > 0 do + with {:ok, columns} <- map_to_list_columns(map) do + from_lists(columns) + end + end + + def from_map(%{}), do: {:error, "from_map/1 requires at least one column"} + def from_map(_), do: {:error, "from_map/1 expects a map of name => list"} + @doc """ Create a `RecordBatch` from column-oriented binary data. @@ -239,4 +314,293 @@ defmodule ExArrow.RecordBatch do {:error, _} = err -> err end end + + # --- from_lists/1 / from_map/1 -------------------------------------------- + + defp validate_from_lists_shape([]), do: {:error, "from_lists/1 requires at least one column"} + + defp validate_from_lists_shape(columns) do + bad_shape? = + Enum.any?(columns, fn + {_n, _d, values} when is_list(values) -> false + _ -> true + end) + + if bad_shape? do + {:error, "from_lists/1 expects {name, dtype, values} triples with list values"} + else + lengths = Enum.map(columns, fn {_n, _d, values} -> length(values) end) + + case Enum.uniq(lengths) do + [_] -> :ok + _ -> {:error, "from_lists/1 column lengths must match, got: #{inspect(lengths)}"} + end + end + end + + defp pack_from_lists(columns) do + {_n, _d, first_values} = hd(columns) + length = length(first_values) + + reduced = + Enum.reduce_while(columns, {:ok, {[], [], []}}, fn {name, dtype, values}, + {:ok, {ns, ds, bs}} -> + with {:ok, name_str} <- normalize_field_name(name), + {:ok, dtype_str} <- normalize_list_dtype(dtype), + {:ok, binary} <- pack_column(dtype_str, values) do + {:cont, {:ok, {[name_str | ns], [dtype_str | ds], [binary | bs]}}} + else + {:error, _} = err -> {:halt, err} + end + end) + + case reduced do + {:ok, {names_rev, dtypes_rev, binaries_rev}} -> + {:ok, Enum.reverse(names_rev), Enum.reverse(dtypes_rev), Enum.reverse(binaries_rev), + length} + + {:error, _} = err -> + err + end + end + + defp map_to_list_columns(map) do + pairs = Enum.map(map, fn {name, values} -> {name, values} end) + + reduced = + Enum.reduce_while(pairs, {:ok, []}, fn {name, values}, {:ok, acc} -> + append_inferred_column(name, values, acc) + end) + + case reduced do + {:ok, columns_rev} -> + columns = + columns_rev + |> Enum.reverse() + |> Enum.sort_by(fn {name, _dtype, _values} -> name end) + + {:ok, columns} + + {:error, _} = err -> + err + end + end + + defp append_inferred_column(_name, values, _acc) when not is_list(values) do + {:halt, {:error, "from_map/1 values must be lists, got: #{inspect(values)}"}} + end + + defp append_inferred_column(_name, [], _acc) do + {:halt, {:error, "from_map/1 cannot infer dtype for an empty column"}} + end + + defp append_inferred_column(name, values, acc) do + with {:ok, dtype} <- infer_list_dtype(values), + {:ok, name_str} <- normalize_field_name(name) do + {:cont, {:ok, [{name_str, dtype, values} | acc]}} + else + {:error, _} = err -> {:halt, err} + end + end + + defp infer_list_dtype(values) do + reduced = + Enum.reduce_while(values, {:ok, :unknown}, fn value, {:ok, acc} -> + merge_inferred_class(acc, value) + end) + + case reduced do + {:ok, class} -> dtype_class_to_string(class) + {:error, _} = err -> err + end + end + + defp merge_inferred_class(_acc, nil) do + {:halt, {:error, "from_lists/1 does not support nil cells (null bitmaps not encoded in 0.9)"}} + end + + defp merge_inferred_class(_acc, v) when is_list(v) or is_map(v) or is_tuple(v) do + {:halt, {:error, "from_lists/1 cell must be a scalar, got: #{inspect(v)}"}} + end + + defp merge_inferred_class(acc, v) do + case {acc, classify_cell(v)} do + {:unknown, class} -> {:cont, {:ok, class}} + {class, class} -> {:cont, {:ok, class}} + {:integer, :float} -> {:cont, {:ok, :float}} + {:float, :integer} -> {:cont, {:ok, :float}} + {a, b} -> {:halt, {:error, "from_map/1 mixed cell types in column (#{a} vs #{b})"}} + end + end + + defp dtype_class_to_string(:integer), do: {:ok, "s64"} + defp dtype_class_to_string(:float), do: {:ok, "f64"} + defp dtype_class_to_string(:boolean), do: {:ok, "bool"} + defp dtype_class_to_string(:utf8), do: {:ok, "utf8"} + + defp dtype_class_to_string(:invalid_utf8), + do: {:error, "from_map/1 binary cells must be valid UTF-8 (use from_lists/1 with :binary)"} + + defp dtype_class_to_string(:other), + do: {:error, "from_map/1 cannot infer dtype from cell values"} + + defp dtype_class_to_string(:unknown), + do: {:error, "from_map/1 cannot infer dtype for an empty column"} + + defp classify_cell(v) when is_integer(v), do: :integer + defp classify_cell(v) when is_float(v), do: :float + defp classify_cell(v) when is_boolean(v), do: :boolean + + defp classify_cell(v) when is_binary(v) do + if String.valid?(v), do: :utf8, else: :invalid_utf8 + end + + defp classify_cell(_), do: :other + + defp normalize_field_name(name) when is_binary(name), do: {:ok, name} + defp normalize_field_name(name) when is_atom(name), do: {:ok, Atom.to_string(name)} + + defp normalize_field_name(other), + do: {:error, "field name must be a string or atom, got: #{inspect(other)}"} + + defp normalize_list_dtype(dtype) when is_atom(dtype), + do: normalize_list_dtype(Atom.to_string(dtype)) + + defp normalize_list_dtype(dtype) when is_binary(dtype) do + known = ~w( + s8 s16 s32 s64 u8 u16 u32 u64 f32 f64 bool + date32 date64 + timestamp_seconds timestamp_millis timestamp_micros timestamp_nanos + duration_seconds duration_millis duration_micros duration_nanos + utf8 large_utf8 binary large_binary + ) + + if dtype in known do + {:ok, dtype} + else + {:error, "unsupported from_lists/1 dtype: #{inspect(dtype)}"} + end + end + + defp normalize_list_dtype(other), + do: {:error, "unsupported from_lists/1 dtype: #{inspect(other)}"} + + defp pack_column(dtype, values) do + reduced = + Enum.reduce_while(values, {:ok, []}, fn value, {:ok, acc} -> + case pack_cell(dtype, value) do + {:ok, chunk} -> {:cont, {:ok, [chunk | acc]}} + {:error, _} = err -> {:halt, err} + end + end) + + case reduced do + {:ok, chunks_rev} -> + binary = chunks_rev |> Enum.reverse() |> IO.iodata_to_binary() + {:ok, binary} + + {:error, _} = err -> + err + end + end + + defp pack_cell(_dtype, nil), + do: {:error, "from_lists/1 does not support nil cells (null bitmaps not encoded in 0.9)"} + + defp pack_cell(_dtype, value) when is_list(value) or is_map(value) or is_tuple(value), + do: {:error, "from_lists/1 cell must be a scalar, got: #{inspect(value)}"} + + defp pack_cell("s8", v) when is_integer(v), do: pack_int(v, -128, 127, 8, "int8") + defp pack_cell("s16", v) when is_integer(v), do: pack_int(v, -32_768, 32_767, 16, "int16") + + defp pack_cell("s32", v) when is_integer(v), + do: pack_int(v, -2_147_483_648, 2_147_483_647, 32, "int32") + + defp pack_cell("s64", v) when is_integer(v), + do: pack_int(v, -9_223_372_036_854_775_808, 9_223_372_036_854_775_807, 64, "int64") + + defp pack_cell("u8", v) when is_integer(v), do: pack_uint(v, 255, 8, "uint8") + defp pack_cell("u16", v) when is_integer(v), do: pack_uint(v, 65_535, 16, "uint16") + defp pack_cell("u32", v) when is_integer(v), do: pack_uint(v, 4_294_967_295, 32, "uint32") + + defp pack_cell("u64", v) when is_integer(v) do + if v >= 0 and v <= 18_446_744_073_709_551_615 do + {:ok, <>} + else + {:error, "uint64 value out of range: #{v}"} + end + end + + defp pack_cell("f32", v) when is_integer(v), do: pack_cell("f32", v * 1.0) + defp pack_cell("f32", v) when is_float(v), do: {:ok, <>} + defp pack_cell("f64", v) when is_integer(v), do: pack_cell("f64", v * 1.0) + defp pack_cell("f64", v) when is_float(v), do: {:ok, <>} + + defp pack_cell("bool", true), do: {:ok, <<1>>} + defp pack_cell("bool", false), do: {:ok, <<0>>} + + defp pack_cell("date32", v) when is_integer(v), + do: pack_int(v, -2_147_483_648, 2_147_483_647, 32, "date32") + + defp pack_cell("date64", v) when is_integer(v), + do: pack_int(v, -9_223_372_036_854_775_808, 9_223_372_036_854_775_807, 64, "date64") + + defp pack_cell(dtype, v) + when dtype in [ + "timestamp_seconds", + "timestamp_millis", + "timestamp_micros", + "timestamp_nanos", + "duration_seconds", + "duration_millis", + "duration_micros", + "duration_nanos" + ] and is_integer(v) do + pack_int(v, -9_223_372_036_854_775_808, 9_223_372_036_854_775_807, 64, dtype) + end + + defp pack_cell(dtype, v) when dtype in ["utf8", "large_utf8"] and is_binary(v) do + if String.valid?(v) do + {:ok, <>} + else + {:error, "#{dtype} cell is not valid UTF-8"} + end + end + + defp pack_cell(dtype, v) when dtype in ["binary", "large_binary"] and is_binary(v) do + {:ok, <>} + end + + defp pack_cell(dtype, value), + do: {:error, "cannot pack #{inspect(value)} as #{dtype}"} + + defp pack_int(v, min, max, 8, label) do + if v >= min and v <= max, do: {:ok, <>}, else: out_of_range(label, v) + end + + defp pack_int(v, min, max, 16, label) do + if v >= min and v <= max, do: {:ok, <>}, else: out_of_range(label, v) + end + + defp pack_int(v, min, max, 32, label) do + if v >= min and v <= max, do: {:ok, <>}, else: out_of_range(label, v) + end + + defp pack_int(v, min, max, 64, label) do + if v >= min and v <= max, do: {:ok, <>}, else: out_of_range(label, v) + end + + defp pack_uint(v, max, 8, label) do + if v >= 0 and v <= max, do: {:ok, <>}, else: out_of_range(label, v) + end + + defp pack_uint(v, max, 16, label) do + if v >= 0 and v <= max, do: {:ok, <>}, else: out_of_range(label, v) + end + + defp pack_uint(v, max, 32, label) do + if v >= 0 and v <= max, do: {:ok, <>}, else: out_of_range(label, v) + end + + defp out_of_range(label, v), do: {:error, "#{label} value out of range: #{v}"} end diff --git a/test/ex_arrow/record_batch_from_lists_property_test.exs b/test/ex_arrow/record_batch_from_lists_property_test.exs new file mode 100644 index 0000000..93ca0e4 --- /dev/null +++ b/test/ex_arrow/record_batch_from_lists_property_test.exs @@ -0,0 +1,51 @@ +defmodule ExArrow.RecordBatchFromListsPropertyTest do + use ExUnit.Case, async: true + use ExUnitProperties + + alias ExArrow.IPC + alias ExArrow.Native + alias ExArrow.RecordBatch + + defp s64_column(batch, name) do + ref = RecordBatch.resource_ref(batch) + {:ok, {binary, "s64", _n}} = Native.record_batch_column_buffer(ref, name) + for <>, do: v + end + + defp pack_utf8(values) do + values + |> Enum.map(fn s -> <> end) + |> IO.iodata_to_binary() + end + + defp ipc_bytes(batch) do + schema = RecordBatch.schema(batch) + assert {:ok, bin} = IPC.Writer.to_binary(schema, [batch]) + bin + end + + property "from_lists s64 round-trips through IPC with exact values" do + check all(values <- list_of(integer(-1_000_000..1_000_000), min_length: 0, max_length: 40)) do + assert {:ok, batch} = RecordBatch.from_lists([{"v", :s64, values}]) + schema = RecordBatch.schema(batch) + assert {:ok, ipc} = IPC.Writer.to_binary(schema, [batch]) + assert {:ok, stream} = IPC.Reader.from_binary(ipc) + assert %RecordBatch{} = restored = ExArrow.Stream.next(stream) + assert s64_column(restored, "v") == values + end + end + + property "from_lists utf8 matches from_columns wire packing exactly" do + check all( + values <- + list_of(string(:alphanumeric, max_length: 20), min_length: 0, max_length: 20) + ) do + assert {:ok, batch} = RecordBatch.from_lists([{"s", :utf8, values}]) + + assert {:ok, expected} = + RecordBatch.from_columns(["s"], [pack_utf8(values)], ["utf8"], length(values)) + + assert ipc_bytes(batch) == ipc_bytes(expected) + end + end +end diff --git a/test/ex_arrow/record_batch_from_lists_test.exs b/test/ex_arrow/record_batch_from_lists_test.exs new file mode 100644 index 0000000..7840493 --- /dev/null +++ b/test/ex_arrow/record_batch_from_lists_test.exs @@ -0,0 +1,180 @@ +defmodule ExArrow.RecordBatchFromListsTest do + use ExUnit.Case, async: true + + alias ExArrow.IPC + alias ExArrow.Native + alias ExArrow.RecordBatch + alias ExArrow.Schema + + defp s64_column(batch, name) do + ref = RecordBatch.resource_ref(batch) + {:ok, {binary, "s64", _n}} = Native.record_batch_column_buffer(ref, name) + for <>, do: v + end + + defp f64_column(batch, name) do + ref = RecordBatch.resource_ref(batch) + {:ok, {binary, "f64", _n}} = Native.record_batch_column_buffer(ref, name) + for <>, do: v + end + + defp bool_column(batch, name) do + ref = RecordBatch.resource_ref(batch) + {:ok, {binary, "bool", _n}} = Native.record_batch_column_buffer(ref, name) + for <>, do: b != 0 + end + + defp pack_utf8(values) do + values + |> Enum.map(fn s -> <> end) + |> IO.iodata_to_binary() + end + + defp ipc_bytes(batch) do + schema = RecordBatch.schema(batch) + assert {:ok, bin} = IPC.Writer.to_binary(schema, [batch]) + bin + end + + @tag :nif + test "from_lists builds mixed columns with exact values" do + names = ["a", "b"] + + assert {:ok, batch} = + RecordBatch.from_lists([ + {"id", :s64, [10, 20]}, + {"score", :f64, [1.5, 2.5]}, + {"ok", :bool, [true, false]}, + {"name", :utf8, names} + ]) + + assert RecordBatch.num_rows(batch) == 2 + assert RecordBatch.column_names(batch) == ["id", "score", "ok", "name"] + assert s64_column(batch, "id") == [10, 20] + assert f64_column(batch, "score") == [1.5, 2.5] + assert bool_column(batch, "ok") == [true, false] + + # Utf8 is not extractable via column_buffer; prove exact bytes vs from_columns. + assert {:ok, expected} = + RecordBatch.from_columns( + ["id", "score", "ok", "name"], + [ + <<10::little-signed-64, 20::little-signed-64>>, + <<1.5::little-float-64, 2.5::little-float-64>>, + <<1, 0>>, + pack_utf8(names) + ], + ["s64", "f64", "bool", "utf8"], + 2 + ) + + assert ipc_bytes(batch) == ipc_bytes(expected) + end + + @tag :nif + test "from_lists accepts string dtypes and atom names" do + assert {:ok, batch} = RecordBatch.from_lists([{:id, "s32", [1, 2]}]) + assert Schema.field_names(RecordBatch.schema(batch)) == ["id"] + + ref = RecordBatch.resource_ref(batch) + {:ok, {binary, "s32", _n}} = Native.record_batch_column_buffer(ref, "id") + assert for(<>, do: v) == [1, 2] + end + + @tag :nif + test "from_lists zero-row column is allowed" do + assert {:ok, batch} = RecordBatch.from_lists([{"id", :s64, []}]) + assert RecordBatch.num_rows(batch) == 0 + assert s64_column(batch, "id") == [] + end + + @tag :nif + test "from_lists round-trips through IPC with exact values" do + values = [1, 2, 3] + names = ["x", "y", "z"] + + assert {:ok, batch} = + RecordBatch.from_lists([ + {"id", :s64, values}, + {"name", :utf8, names} + ]) + + schema = RecordBatch.schema(batch) + assert {:ok, ipc} = IPC.Writer.to_binary(schema, [batch]) + assert {:ok, stream} = IPC.Reader.from_binary(ipc) + assert %RecordBatch{} = restored = ExArrow.Stream.next(stream) + assert s64_column(restored, "id") == values + assert ipc_bytes(restored) == ipc + end + + @tag :nif + test "from_map sorts keys and infers dtypes" do + assert {:ok, batch} = + RecordBatch.from_map(%{ + "name" => ["a", "b"], + "id" => [1, 2], + ok: [true, false] + }) + + assert RecordBatch.column_names(batch) == ["id", "name", "ok"] + assert s64_column(batch, "id") == [1, 2] + assert bool_column(batch, "ok") == [true, false] + + assert {:ok, expected} = + RecordBatch.from_lists([ + {"id", :s64, [1, 2]}, + {"name", :utf8, ["a", "b"]}, + {"ok", :bool, [true, false]} + ]) + + assert ipc_bytes(batch) == ipc_bytes(expected) + end + + @tag :nif + test "from_map promotes integers mixed with floats to f64" do + assert {:ok, batch} = RecordBatch.from_map(%{"x" => [1, 2.5]}) + assert f64_column(batch, "x") == [1.0, 2.5] + end + + test "from_lists rejects empty columns list" do + assert {:error, msg} = RecordBatch.from_lists([]) + assert msg =~ "at least one column" + end + + test "from_lists rejects unequal lengths" do + assert {:error, msg} = + RecordBatch.from_lists([{"a", :s64, [1, 2]}, {"b", :s64, [3]}]) + + assert msg =~ "column lengths must match" + end + + test "from_lists rejects nil cells" do + assert {:error, msg} = RecordBatch.from_lists([{"a", :s64, [1, nil]}]) + assert msg =~ "nil" + end + + test "from_lists rejects nested cells" do + assert {:error, msg} = RecordBatch.from_lists([{"a", :s64, [[1]]}]) + assert msg =~ "scalar" + end + + test "from_lists rejects int32 out of range" do + assert {:error, msg} = RecordBatch.from_lists([{"a", :s32, [2_147_483_648]}]) + assert msg =~ "out of range" + end + + test "from_lists rejects unsupported dtype" do + assert {:error, msg} = RecordBatch.from_lists([{"a", :timestamp, [1]}]) + assert msg =~ "unsupported" + end + + test "from_map rejects empty map" do + assert {:error, msg} = RecordBatch.from_map(%{}) + assert msg =~ "at least one column" + end + + test "from_map rejects empty column list" do + assert {:error, msg} = RecordBatch.from_map(%{"a" => []}) + assert msg =~ "empty" + end +end From 7d1b22032b290a4924419ba9ebe27483e0f775ff Mon Sep 17 00:00:00 2001 From: thanos Date: Sat, 12 Sep 2026 13:14:48 -0400 Subject: [PATCH 03/11] M1;: add Compute.Expression AST with Parquet filter split - closed #275 - closed #276 - closed #277 - closed #278 - closed #279 - closed #280 --- lib/ex_arrow/compute.ex | 4 + lib/ex_arrow/compute/expression.ex | 473 ++++++++++++++++++++++ lib/ex_arrow/parquet/opts.ex | 21 + mix.exs | 2 +- test/ex_arrow/compute/expression_test.exs | 150 +++++++ 5 files changed, 649 insertions(+), 1 deletion(-) create mode 100644 lib/ex_arrow/compute/expression.ex create mode 100644 test/ex_arrow/compute/expression_test.exs diff --git a/lib/ex_arrow/compute.ex b/lib/ex_arrow/compute.ex index dc401bd..cdbeae0 100644 --- a/lib/ex_arrow/compute.ex +++ b/lib/ex_arrow/compute.ex @@ -34,6 +34,10 @@ defmodule ExArrow.Compute do You can also write a Parquet/IPC file that contains a pre-computed boolean column and read it back as the predicate. + + For analyzable predicates (Dataset / Parquet pushdown), see + `ExArrow.Compute.Expression`. Residual expression evaluation on batches + lands in a later milestone. """ alias ExArrow.Native diff --git a/lib/ex_arrow/compute/expression.ex b/lib/ex_arrow/compute/expression.ex new file mode 100644 index 0000000..71b95cc --- /dev/null +++ b/lib/ex_arrow/compute/expression.ex @@ -0,0 +1,473 @@ +defmodule ExArrow.Compute.Expression do + @moduledoc """ + Analyzable compute expression AST for filters and (later) Dataset scanners. + + Builders are the canonical API for 0.9. Macro sugar (`expr do ... end`) is + out of scope. Expressions are data: they can be validated against a schema, + printed for diagnostics, and partially compiled to the Parquet filter tuple + AST used since v0.8.0. They are not Elixir closures. + + ## Example + + alias ExArrow.Compute.Expression, as: E + + filter = + E.and_( + E.gte(E.field("date"), E.scalar(~D[2026-01-01])), + E.ne(E.field("amount"), E.scalar(0)) + ) + + {pushed, residual} = E.to_parquet_filters(filter) + """ + + alias ExArrow.Schema + + @type t :: %__MODULE__{node: expr_node()} + defstruct [:node] + + @type expr_node :: + {:field, String.t()} + | {:scalar, scalar()} + | {:call, op(), [expr_node()]} + + @type op :: :eq | :ne | :gt | :gte | :lt | :lte | :and | :or | :not + + @type scalar :: + integer() + | float() + | boolean() + | String.t() + | Date.t() + | NaiveDateTime.t() + | DateTime.t() + + @compare_ops [:eq, :ne, :gt, :gte, :lt, :lte] + + @doc """ + Reference a column by name. + """ + @spec field(String.t() | atom()) :: t() + def field(name) when is_binary(name), do: %__MODULE__{node: {:field, name}} + + def field(name) when is_atom(name), do: field(Atom.to_string(name)) + + @doc """ + A scalar literal. + + Supported values: integer, float, boolean, UTF-8 string, `Date`, + `NaiveDateTime`, and `DateTime`. + """ + @spec scalar(scalar()) :: t() + def scalar(%Date{} = d), do: %__MODULE__{node: {:scalar, d}} + def scalar(%NaiveDateTime{} = dt), do: %__MODULE__{node: {:scalar, dt}} + def scalar(%DateTime{} = dt), do: %__MODULE__{node: {:scalar, dt}} + + def scalar(v) when is_integer(v) or is_float(v) or is_boolean(v), + do: %__MODULE__{node: {:scalar, v}} + + def scalar(v) when is_binary(v) do + if String.valid?(v) do + %__MODULE__{node: {:scalar, v}} + else + raise ArgumentError, "scalar string must be valid UTF-8" + end + end + + def scalar(other), + do: raise(ArgumentError, "unsupported scalar: #{inspect(other)}") + + @doc """ + Equality comparison. + """ + @spec eq(t(), t()) :: t() + def eq(%__MODULE__{} = l, %__MODULE__{} = r), do: call(:eq, [l, r]) + + @doc """ + Inequality comparison. + """ + @spec ne(t(), t()) :: t() + def ne(%__MODULE__{} = l, %__MODULE__{} = r), do: call(:ne, [l, r]) + + @doc """ + Greater-than comparison. + """ + @spec gt(t(), t()) :: t() + def gt(%__MODULE__{} = l, %__MODULE__{} = r), do: call(:gt, [l, r]) + + @doc """ + Greater-than-or-equal comparison. + """ + @spec gte(t(), t()) :: t() + def gte(%__MODULE__{} = l, %__MODULE__{} = r), do: call(:gte, [l, r]) + + @doc """ + Less-than comparison. + """ + @spec lt(t(), t()) :: t() + def lt(%__MODULE__{} = l, %__MODULE__{} = r), do: call(:lt, [l, r]) + + @doc """ + Less-than-or-equal comparison. + """ + @spec lte(t(), t()) :: t() + def lte(%__MODULE__{} = l, %__MODULE__{} = r), do: call(:lte, [l, r]) + + @doc """ + Boolean AND of two expressions. + + Named `and_/2` because `and/2` is a Kernel special form. + """ + @spec and_(t(), t()) :: t() + def and_(%__MODULE__{} = l, %__MODULE__{} = r), do: call(:and, [l, r]) + + @doc """ + Boolean OR of two expressions. + + Named `or_/2` because `or/2` is a Kernel special form. + """ + @spec or_(t(), t()) :: t() + def or_(%__MODULE__{} = l, %__MODULE__{} = r), do: call(:or, [l, r]) + + @doc """ + Boolean NOT. + + Named `not_/1` because `not/1` is a Kernel special form. + """ + @spec not_(t()) :: t() + def not_(%__MODULE__{} = e), do: call(:not, [e]) + + @doc """ + Returns `true` if `term` is an `ExArrow.Compute.Expression`. + """ + @spec expression?(term()) :: boolean() + def expression?(%__MODULE__{}), do: true + def expression?(_), do: false + + @doc """ + Type-check `expr` against `schema`. + + Checks that field names exist and that comparisons are type-compatible + with the referenced column (and the other side, when both are fields). + """ + @spec validate(t(), Schema.t()) :: {:ok, t()} | {:error, String.t()} + def validate(%__MODULE__{} = expr, schema) do + fields = + schema + |> Schema.fields() + |> Map.new(fn f -> {f.name, f.type} end) + + case validate_node(expr.node, fields) do + :ok -> {:ok, expr} + {:error, _} = err -> err + end + end + + @doc """ + Split `expr` into a Parquet-pushable filter AST and an optional residual + expression. + + Returns `{pushed, residual}` where: + + - `pushed` is `nil` or a v0.8.0 filter tuple + (`{:eq|:ne|:gt|:gte|:lt|:lte, col, value}` / `{:and|:or, [...]}`) + - `residual` is `nil` or an `Expression` that still needs post-decode + evaluation (e.g. `not_/1`, temporal scalars the Parquet reader cannot + bind yet, field-vs-field comparisons) + + AND may push one side and residual the other. OR is pushed only when both + sides are fully pushable; otherwise the whole OR is residual. + """ + @spec to_parquet_filters(t()) :: {term() | nil, t() | nil} + def to_parquet_filters(%__MODULE__{node: node}) do + {pushed, residual_node} = split_node(node) + residual = if residual_node, do: %__MODULE__{node: residual_node}, else: nil + {pushed, residual} + end + + @doc """ + Render `expr` as a diagnostic string. + """ + @spec to_string(t()) :: String.t() + def to_string(%__MODULE__{node: node}), do: render(node) + + defimpl String.Chars do + alias ExArrow.Compute.Expression + + @spec to_string(Expression.t()) :: String.t() + def to_string(expr), do: Expression.to_string(expr) + end + + defimpl Inspect do + alias ExArrow.Compute.Expression + import Inspect.Algebra + + @spec inspect(Expression.t(), Inspect.Opts.t()) :: Inspect.Algebra.t() + def inspect(%Expression{} = expr, opts) do + concat([ + "#ExArrow.Compute.Expression<", + to_doc(Expression.to_string(expr), opts), + ">" + ]) + end + end + + # --- builders ------------------------------------------------------------- + + defp call(op, exprs) do + %__MODULE__{node: {:call, op, Enum.map(exprs, & &1.node)}} + end + + # --- validate ------------------------------------------------------------- + + defp validate_node({:field, name}, fields) do + if Map.has_key?(fields, name) do + :ok + else + {:error, "unknown field #{inspect(name)}"} + end + end + + defp validate_node({:scalar, value}, _fields) do + if supported_scalar?(value) do + :ok + else + {:error, "unsupported scalar: #{inspect(value)}"} + end + end + + defp validate_node({:call, op, [left, right]}, fields) when op in @compare_ops do + with {:ok, lt} <- node_type(left, fields), + {:ok, rt} <- node_type(right, fields), + :ok <- compatible_compare(lt, rt, op) do + :ok + end + end + + defp validate_node({:call, op, [left, right]}, fields) when op in [:and, :or] do + with :ok <- validate_node(left, fields), + :ok <- validate_node(right, fields) do + :ok + end + end + + defp validate_node({:call, :not, [inner]}, fields), do: validate_node(inner, fields) + + defp validate_node(other, _fields), + do: {:error, "invalid expression node: #{inspect(other)}"} + + defp node_type({:field, name}, fields) do + case Map.fetch(fields, name) do + {:ok, type} -> {:ok, {:column, type}} + :error -> {:error, "unknown field #{inspect(name)}"} + end + end + + defp node_type({:scalar, value}, _fields), do: {:ok, {:scalar, value}} + + defp node_type({:call, _, _}, _fields), + do: {:error, "comparison operands must be field or scalar"} + + defp compatible_compare({:column, col_type}, {:scalar, value}, _op) do + if scalar_matches_type?(value, col_type) do + :ok + else + {:error, "type mismatch: column type #{inspect(col_type)} vs scalar #{inspect(value)}"} + end + end + + defp compatible_compare({:scalar, value}, {:column, col_type}, op), + do: compatible_compare({:column, col_type}, {:scalar, value}, op) + + defp compatible_compare({:column, t1}, {:column, t2}, _op) do + if types_comparable?(t1, t2) do + :ok + else + {:error, "cannot compare columns of types #{inspect(t1)} and #{inspect(t2)}"} + end + end + + defp compatible_compare({:scalar, _}, {:scalar, _}, _op), + do: {:error, "comparison requires at least one field reference"} + + defp supported_scalar?(v) + when is_integer(v) or is_float(v) or is_boolean(v) or is_binary(v), + do: true + + defp supported_scalar?(%Date{}), do: true + defp supported_scalar?(%NaiveDateTime{}), do: true + defp supported_scalar?(%DateTime{}), do: true + defp supported_scalar?(_), do: false + + defp scalar_matches_type?(v, :int64) when is_integer(v), do: true + defp scalar_matches_type?(v, :int32) when is_integer(v), do: in_i32?(v) + defp scalar_matches_type?(v, :int16) when is_integer(v), do: v >= -32_768 and v <= 32_767 + defp scalar_matches_type?(v, :int8) when is_integer(v), do: v >= -128 and v <= 127 + defp scalar_matches_type?(v, :uint64) when is_integer(v), do: v >= 0 + defp scalar_matches_type?(v, :uint32) when is_integer(v), do: v >= 0 and v <= 4_294_967_295 + defp scalar_matches_type?(v, :uint16) when is_integer(v), do: v >= 0 and v <= 65_535 + defp scalar_matches_type?(v, :uint8) when is_integer(v), do: v >= 0 and v <= 255 + defp scalar_matches_type?(v, :float64) when is_number(v), do: true + defp scalar_matches_type?(v, :float32) when is_number(v), do: true + defp scalar_matches_type?(v, :boolean) when is_boolean(v), do: true + defp scalar_matches_type?(v, :utf8) when is_binary(v), do: String.valid?(v) + defp scalar_matches_type?(v, :large_utf8) when is_binary(v), do: String.valid?(v) + defp scalar_matches_type?(%Date{}, :date32), do: true + defp scalar_matches_type?(%Date{}, :date64), do: true + defp scalar_matches_type?(v, :date32) when is_integer(v), do: in_i32?(v) + defp scalar_matches_type?(v, :date64) when is_integer(v), do: true + + defp scalar_matches_type?(%NaiveDateTime{}, t) + when t in [ + :timestamp, + :timestamp_seconds, + :timestamp_millis, + :timestamp_micros, + :timestamp_nanos + ], + do: true + + defp scalar_matches_type?(%DateTime{}, t) + when t in [ + :timestamp, + :timestamp_seconds, + :timestamp_millis, + :timestamp_micros, + :timestamp_nanos + ], + do: true + + defp scalar_matches_type?(v, t) + when is_integer(v) and + t in [ + :timestamp, + :timestamp_seconds, + :timestamp_millis, + :timestamp_micros, + :timestamp_nanos, + :duration_seconds, + :duration_millis, + :duration_micros, + :duration_nanos + ], + do: true + + defp scalar_matches_type?(_, _), do: false + + defp types_comparable?(t, t), do: true + + defp types_comparable?(a, b) + when a in [:int8, :int16, :int32, :int64] and b in [:int8, :int16, :int32, :int64], + do: true + + defp types_comparable?(a, b) when a in [:float32, :float64] and b in [:float32, :float64], + do: true + + defp types_comparable?(:utf8, :large_utf8), do: true + defp types_comparable?(:large_utf8, :utf8), do: true + defp types_comparable?(_, _), do: false + + defp in_i32?(v), do: v >= -2_147_483_648 and v <= 2_147_483_647 + + # --- split to parquet ----------------------------------------------------- + + defp split_node({:call, op, [left, right]}) when op in @compare_ops do + case pushable_compare(op, left, right) do + {:ok, tuple} -> {tuple, nil} + :residual -> {nil, {:call, op, [left, right]}} + end + end + + defp split_node({:call, :and, [left, right]}) do + {lp, lr} = split_node(left) + {rp, rr} = split_node(right) + + pushed = + case {lp, rp} do + {nil, nil} -> nil + {l, nil} -> l + {nil, r} -> r + {l, r} -> {:and, [l, r]} + end + + residual = + case {lr, rr} do + {nil, nil} -> nil + {l, nil} -> l + {nil, r} -> r + {l, r} -> {:call, :and, [l, r]} + end + + {pushed, residual} + end + + defp split_node({:call, :or, [left, right]}) do + {lp, lr} = split_node(left) + {rp, rr} = split_node(right) + + if lr == nil and rr == nil and lp != nil and rp != nil do + {{:or, [lp, rp]}, nil} + else + {nil, {:call, :or, [left, right]}} + end + end + + defp split_node({:call, :not, [inner]}) do + # Parquet filter AST has no NOT; keep as residual. + {nil, {:call, :not, [inner]}} + end + + defp split_node(other), do: {nil, other} + + defp pushable_compare(op, {:field, col}, {:scalar, value}) do + case parquet_scalar(value) do + {:ok, v} -> {:ok, {op, col, v}} + :error -> :residual + end + end + + defp pushable_compare(op, {:scalar, value}, {:field, col}) do + # Flip comparison for scalar-on-left: 5 > field("x") => field("x") < 5 + case {parquet_scalar(value), flip_op(op)} do + {{:ok, v}, flipped} -> {:ok, {flipped, col, v}} + _ -> :residual + end + end + + defp pushable_compare(_op, _l, _r), do: :residual + + defp flip_op(:eq), do: :eq + defp flip_op(:ne), do: :ne + defp flip_op(:gt), do: :lt + defp flip_op(:gte), do: :lte + defp flip_op(:lt), do: :gt + defp flip_op(:lte), do: :gte + + defp parquet_scalar(v) when is_integer(v) or is_float(v) or is_boolean(v), do: {:ok, v} + + defp parquet_scalar(v) when is_binary(v) do + if String.valid?(v), do: {:ok, v}, else: :error + end + + # Temporal scalars are not accepted by Parquet.Opts / the current NIF filter + # binder; leave them for residual evaluation (M2). + defp parquet_scalar(%Date{}), do: :error + defp parquet_scalar(%NaiveDateTime{}), do: :error + defp parquet_scalar(%DateTime{}), do: :error + defp parquet_scalar(_), do: :error + + # --- render --------------------------------------------------------------- + + defp render({:field, name}), do: "field(#{inspect(name)})" + defp render({:scalar, v}), do: "scalar(#{inspect(v)})" + + defp render({:call, :not, [inner]}), do: "not_(#{render(inner)})" + + defp render({:call, :and, [l, r]}), do: "and_(#{render(l)}, #{render(r)})" + defp render({:call, :or, [l, r]}), do: "or_(#{render(l)}, #{render(r)})" + + defp render({:call, op, [l, r]}) when op in @compare_ops do + "#{op}(#{render(l)}, #{render(r)})" + end + + defp render(other), do: inspect(other) +end diff --git a/lib/ex_arrow/parquet/opts.ex b/lib/ex_arrow/parquet/opts.ex index bda53cb..d331e84 100644 --- a/lib/ex_arrow/parquet/opts.ex +++ b/lib/ex_arrow/parquet/opts.ex @@ -1,6 +1,8 @@ defmodule ExArrow.Parquet.Opts do @moduledoc false + alias ExArrow.Compute.Expression + @read_keys [:columns, :row_groups, :filters] @write_keys [:compression, :row_group_size, :dictionary] # `:uncompressed` is accepted as a synonym for `:none`. @@ -94,6 +96,25 @@ defmodule ExArrow.Parquet.Opts do defp validate_filters(nil), do: {:ok, nil} + defp validate_filters(%Expression{} = expr) do + case Expression.to_parquet_filters(expr) do + {nil, _residual} -> + {:error, + ":filters expression has no Parquet-pushable part (use Scanner for residual-only filters)"} + + {pushed, nil} -> + case check_filter(pushed) do + :ok -> {:ok, pushed} + {:error, _} = err -> err + end + + {_pushed, %Expression{} = residual} -> + {:error, + ":filters expression has a non-pushable residual (#{Expression.to_string(residual)}); " <> + "pass only pushable predicates to Parquet.Reader, or use Dataset.Scanner"} + end + end + defp validate_filters(expr) do case check_filter(expr) do :ok -> {:ok, expr} diff --git a/mix.exs b/mix.exs index e5aceab..1a0b254 100644 --- a/mix.exs +++ b/mix.exs @@ -135,7 +135,7 @@ defmodule ExArrow.MixProject do "Data interchange": [ExArrow.DataFrame, ExArrow.Schema.Mapper], IPC: [ExArrow.IPC.Reader, ExArrow.IPC.Writer, ExArrow.IPC.File], Parquet: [ExArrow.Parquet.Reader, ExArrow.Parquet.Writer, ExArrow.Parquet.Metadata], - "Compute kernels": [ExArrow.Compute], + "Compute kernels": [ExArrow.Compute, ExArrow.Compute.Expression], "Batch operations": [ExArrow.Batch], Pipeline: [ ExArrow.Pipeline, diff --git a/test/ex_arrow/compute/expression_test.exs b/test/ex_arrow/compute/expression_test.exs new file mode 100644 index 0000000..64f3481 --- /dev/null +++ b/test/ex_arrow/compute/expression_test.exs @@ -0,0 +1,150 @@ +defmodule ExArrow.Compute.ExpressionTest do + use ExUnit.Case, async: true + + alias ExArrow.Compute.Expression, as: E + alias ExArrow.Parquet.Opts + alias ExArrow.RecordBatch + + defp schema_for(columns) do + assert {:ok, batch} = RecordBatch.from_lists(columns) + RecordBatch.schema(batch) + end + + describe "builders" do + test "field/1 accepts string or atom names" do + assert E.field("amount") == E.field(:amount) + assert to_string(E.field("amount")) == "field(\"amount\")" + assert inspect(E.field("amount")) =~ "field(" + end + + test "scalar/1 accepts primitive and temporal values" do + assert to_string(E.scalar(1)) == "scalar(1)" + assert to_string(E.scalar(1.5)) == "scalar(1.5)" + assert to_string(E.scalar(true)) == "scalar(true)" + assert to_string(E.scalar("x")) == "scalar(\"x\")" + assert to_string(E.scalar(~D[2026-01-01])) =~ "2026-01-01" + end + + test "scalar/1 rejects invalid UTF-8" do + assert_raise ArgumentError, fn -> E.scalar(<<0xFF>>) end + end + + test "comparisons and boolean composition render" do + expr = + E.and_( + E.gte(E.field("date"), E.scalar(~D[2026-01-01])), + E.ne(E.field("amount"), E.scalar(0)) + ) + + rendered = to_string(expr) + assert rendered =~ "and_(" + assert rendered =~ "gte(" + assert rendered =~ "ne(" + assert inspect(expr) =~ "#ExArrow.Compute.Expression<" + end + end + + describe "validate/2" do + test "accepts typed comparisons against schema" do + schema = + schema_for([ + {"amount", :s64, [0]}, + {"score", :f64, [0.0]}, + {"name", :utf8, ["a"]}, + {"ok", :bool, [true]} + ]) + + assert {:ok, _} = E.validate(E.gt(E.field("amount"), E.scalar(0)), schema) + assert {:ok, _} = E.validate(E.eq(E.field("score"), E.scalar(0.5)), schema) + assert {:ok, _} = E.validate(E.eq(E.field("name"), E.scalar("a")), schema) + assert {:ok, _} = E.validate(E.eq(E.field("ok"), E.scalar(true)), schema) + end + + test "rejects unknown fields and type mismatches" do + schema = schema_for([{"amount", :s64, [1]}]) + + assert {:error, msg} = E.validate(E.eq(E.field("missing"), E.scalar(1)), schema) + assert msg =~ "unknown field" + + assert {:error, msg} = E.validate(E.eq(E.field("amount"), E.scalar("x")), schema) + assert msg =~ "type mismatch" + end + + test "rejects int32 out of range scalars" do + schema = schema_for([{"x", :s32, [1]}]) + assert {:error, msg} = E.validate(E.eq(E.field("x"), E.scalar(2_147_483_648)), schema) + assert msg =~ "type mismatch" + end + end + + describe "to_parquet_filters/1" do + test "pushes field-vs-scalar comparisons" do + expr = E.gt(E.field("score"), E.scalar(0.9)) + assert {{:gt, "score", 0.9}, nil} = E.to_parquet_filters(expr) + end + + test "flips scalar-on-left comparisons" do + expr = E.gt(E.scalar(5), E.field("id")) + assert {{:lt, "id", 5}, nil} = E.to_parquet_filters(expr) + end + + test "AND pushes both sides when possible" do + expr = E.and_(E.gte(E.field("id"), E.scalar(10)), E.lt(E.field("id"), E.scalar(100))) + + assert {{:and, [{:gte, "id", 10}, {:lt, "id", 100}]}, nil} = E.to_parquet_filters(expr) + end + + test "AND can push one side and residual the other" do + expr = + E.and_( + E.gt(E.field("amount"), E.scalar(0)), + E.gte(E.field("date"), E.scalar(~D[2026-01-01])) + ) + + assert {{:gt, "amount", 0}, %E{} = residual} = E.to_parquet_filters(expr) + assert to_string(residual) =~ "gte(" + assert to_string(residual) =~ "2026-01-01" + end + + test "OR with a residual side keeps the whole OR as residual" do + expr = + E.or_( + E.eq(E.field("name"), E.scalar("a")), + E.eq(E.field("date"), E.scalar(~D[2026-01-01])) + ) + + assert {nil, %E{} = residual} = E.to_parquet_filters(expr) + assert to_string(residual) =~ "or_(" + end + + test "not_/1 is always residual" do + expr = E.not_(E.eq(E.field("ok"), E.scalar(true))) + assert {nil, %E{}} = E.to_parquet_filters(expr) + end + + test "field-vs-field comparison is residual" do + expr = E.gt(E.field("a"), E.field("b")) + assert {nil, %E{}} = E.to_parquet_filters(expr) + end + end + + describe "Parquet.Opts normalisation" do + test "accepts a fully pushable Expression and stores the tuple AST" do + expr = E.and_(E.gt(E.field("x"), E.scalar(1)), E.lt(E.field("x"), E.scalar(10))) + + assert {:ok, [filters: {:and, [{:gt, "x", 1}, {:lt, "x", 10}]}]} = + Opts.validate_read(filters: expr) + end + + test "still accepts legacy tuple AST" do + assert {:ok, [filters: {:gt, "score", 0.9}]} = + Opts.validate_read(filters: {:gt, "score", 0.9}) + end + + test "rejects Expression with residual" do + expr = E.gte(E.field("date"), E.scalar(~D[2026-01-01])) + assert {:error, msg} = Opts.validate_read(filters: expr) + assert msg =~ "residual" + end + end +end From 40dacd1c1c32b36e6a17c8e7e2d2104e8a13d62f Mon Sep 17 00:00:00 2001 From: thanos Date: Sat, 12 Sep 2026 14:07:48 -0400 Subject: [PATCH 04/11] Add residual Expression evaluation via Compute.filter/2. Evaluate Compute.Expression ASTs in a NIF to a boolean mask and filter batches in native memory, covering comparisons that Parquet cannot push. - closed #281 - closed #282 - closed #283 - closed #284 - closed #285 --- lib/ex_arrow/batch.ex | 21 +- lib/ex_arrow/compute.ex | 52 +- lib/ex_arrow/compute/expression.ex | 34 ++ lib/ex_arrow/native.ex | 1 + native/ex_arrow_native/src/compute.rs | 629 ++++++++++++++++++++- test/ex_arrow/compute_filter_expr_test.exs | 148 +++++ test/ex_arrow/native_test.exs | 5 + 7 files changed, 867 insertions(+), 23 deletions(-) create mode 100644 test/ex_arrow/compute_filter_expr_test.exs diff --git a/lib/ex_arrow/batch.ex b/lib/ex_arrow/batch.ex index e5f8ccb..dab3f68 100644 --- a/lib/ex_arrow/batch.ex +++ b/lib/ex_arrow/batch.ex @@ -43,6 +43,7 @@ defmodule ExArrow.Batch do """ alias ExArrow.Compute + alias ExArrow.Compute.Expression alias ExArrow.Native alias ExArrow.RecordBatch alias ExArrow.Schema @@ -198,21 +199,27 @@ defmodule ExArrow.Batch do end @doc """ - Filter rows of `batch` using the first (boolean) column of `predicate_batch`. + Filter rows of `batch` using a boolean mask batch or an `Expression`. - Delegates directly to `ExArrow.Compute.filter/2`. Rows where the predicate - is `true` are kept; rows where it is `false` or `null` are dropped. The - predicate's first column must be a boolean Arrow array with the same row - count as `batch`. + Delegates directly to `ExArrow.Compute.filter/2`. Rows where the predicate + is `true` are kept; rows where it is `false` or `null` are dropped. + + When `predicate` is a `RecordBatch`, its first column must be a boolean Arrow + array with the same row count as `batch`. When it is an + `ExArrow.Compute.Expression`, the expression is evaluated in native memory. Returns `{:ok, filtered_batch}` or `{:error, message}`. - ## Example + ## Examples {:ok, mask} = ExArrow.Compute.project(batch, ["is_active"]) {:ok, filtered} = ExArrow.Batch.filter(batch, mask) + + alias ExArrow.Compute.Expression, as: E + {:ok, filtered} = + ExArrow.Batch.filter(batch, E.gt(E.field("score"), E.scalar(0.9))) """ - @spec filter(RecordBatch.t(), RecordBatch.t()) :: + @spec filter(RecordBatch.t(), RecordBatch.t() | Expression.t()) :: {:ok, RecordBatch.t()} | {:error, String.t()} def filter(batch, predicate) do Compute.filter(batch, predicate) diff --git a/lib/ex_arrow/compute.ex b/lib/ex_arrow/compute.ex index cdbeae0..23c5e62 100644 --- a/lib/ex_arrow/compute.ex +++ b/lib/ex_arrow/compute.ex @@ -20,9 +20,12 @@ defmodule ExArrow.Compute do ## Building a boolean predicate for `filter/2` - `filter/2` expects the **first column** of a second record batch to be a - boolean Arrow array. The most common source is a query result that already - contains a boolean column: + `filter/2` accepts either: + + 1. A **mask batch** whose first column is a boolean Arrow array, or + 2. An `ExArrow.Compute.Expression` evaluated in native memory to a mask + + Mask-batch example: # e.g. "SELECT id, score, is_active FROM users" {:ok, stream} = ExArrow.ADBC.Statement.execute(stmt) @@ -32,36 +35,59 @@ defmodule ExArrow.Compute do {:ok, mask} = ExArrow.Compute.project(batch, ["is_active"]) {:ok, filtered} = ExArrow.Compute.filter(batch, mask) + Expression example: + + alias ExArrow.Compute.Expression, as: E + + {:ok, filtered} = + ExArrow.Compute.filter(batch, E.gt(E.field("score"), E.scalar(0.9))) + You can also write a Parquet/IPC file that contains a pre-computed boolean column and read it back as the predicate. For analyzable predicates (Dataset / Parquet pushdown), see - `ExArrow.Compute.Expression`. Residual expression evaluation on batches - lands in a later milestone. + `ExArrow.Compute.Expression`. Residual predicates that cannot be pushed + are evaluated with this same `filter/2` path after decode. """ + alias ExArrow.Compute.Expression alias ExArrow.Native alias ExArrow.RecordBatch @doc """ - Filter rows from `batch` using the first (boolean) column of `predicate_batch`. + Filter rows from `batch` using a boolean mask batch or an `Expression`. + + When `predicate` is a `RecordBatch`, its first column must be a boolean Arrow + array with the same row count as `batch`. Rows where the predicate is `true` + are kept; rows where it is `false` or `null` are dropped. - `predicate_batch` must have at least one column and its first column must be - a boolean Arrow array with the same row count as `batch`. Rows where the - predicate is `true` are kept; rows where it is `false` or `null` are dropped. + When `predicate` is an `ExArrow.Compute.Expression`, the expression is + evaluated in the NIF to a boolean mask and then applied the same way. Returns `{:ok, filtered_batch}` or `{:error, message}`. - ## Example + ## Examples # Keep only rows where "is_active" is true. - # batch has columns [id, score, is_active]; extract the bool column first. {:ok, mask} = ExArrow.Compute.project(batch, ["is_active"]) {:ok, filtered} = ExArrow.Compute.filter(batch, mask) - # filtered has the same columns as batch but only the rows where is_active = true + + alias ExArrow.Compute.Expression, as: E + {:ok, filtered} = + ExArrow.Compute.filter(batch, E.gt(E.field("score"), E.scalar(0.9))) """ + @spec filter(RecordBatch.t(), RecordBatch.t() | Expression.t()) :: + {:ok, RecordBatch.t()} | {:error, String.t()} + def filter(batch, %Expression{} = expr) do + b = RecordBatch.resource_ref(batch) + encoded = Expression.encode_for_nif(expr) + + case Native.compute_filter_expr(b, encoded) do + {:ok, ref} -> {:ok, RecordBatch.from_ref(ref)} + {:error, msg} -> {:error, msg} + end + end - @spec filter(RecordBatch.t(), RecordBatch.t()) :: {:ok, RecordBatch.t()} | {:error, String.t()} def filter(batch, predicate_batch) do b = RecordBatch.resource_ref(batch) p = RecordBatch.resource_ref(predicate_batch) diff --git a/lib/ex_arrow/compute/expression.ex b/lib/ex_arrow/compute/expression.ex index 71b95cc..3b6e013 100644 --- a/lib/ex_arrow/compute/expression.ex +++ b/lib/ex_arrow/compute/expression.ex @@ -184,6 +184,10 @@ defmodule ExArrow.Compute.Expression do {pushed, residual} end + @doc false + @spec encode_for_nif(t()) :: term() + def encode_for_nif(%__MODULE__{node: node}), do: encode_node(node) + @doc """ Render `expr` as a diagnostic string. """ @@ -470,4 +474,34 @@ defmodule ExArrow.Compute.Expression do end defp render(other), do: inspect(other) + + # --- NIF encoding --------------------------------------------------------- + + defp encode_node({:field, name}), do: {:field, name} + defp encode_node({:scalar, v}), do: {:scalar, encode_scalar(v)} + defp encode_node({:call, op, args}), do: {:call, op, Enum.map(args, &encode_node/1)} + + defp encode_scalar(v) when is_integer(v) or is_float(v) or is_boolean(v), do: v + + defp encode_scalar(v) when is_binary(v) do + if String.valid?(v) do + v + else + raise ArgumentError, "invalid UTF-8 scalar" + end + end + + defp encode_scalar(%Date{} = d), do: {:date32, Date.diff(d, ~D[1970-01-01])} + + defp encode_scalar(%NaiveDateTime{} = ndt) do + {:timestamp_micros, NaiveDateTime.diff(ndt, ~N[1970-01-01 00:00:00], :microsecond)} + end + + defp encode_scalar(%DateTime{} = dt) do + {:timestamp_micros, DateTime.to_unix(dt, :microsecond)} + end + + defp encode_scalar(other) do + raise ArgumentError, "unsupported scalar for NIF encode: #{inspect(other)}" + end end diff --git a/lib/ex_arrow/native.ex b/lib/ex_arrow/native.ex index 1d2fcea..64b8522 100644 --- a/lib/ex_arrow/native.ex +++ b/lib/ex_arrow/native.ex @@ -145,6 +145,7 @@ defmodule ExArrow.Native do # Compute kernels def compute_filter(_batch_ref, _predicate_ref), do: :erlang.nif_error(:nif_not_loaded) + def compute_filter_expr(_batch_ref, _encoded_expr), do: :erlang.nif_error(:nif_not_loaded) def compute_project(_batch_ref, _column_names), do: :erlang.nif_error(:nif_not_loaded) def compute_sort(_batch_ref, _column_name, _ascending), do: :erlang.nif_error(:nif_not_loaded) diff --git a/native/ex_arrow_native/src/compute.rs b/native/ex_arrow_native/src/compute.rs index 38d6572..50ba86a 100644 --- a/native/ex_arrow_native/src/compute.rs +++ b/native/ex_arrow_native/src/compute.rs @@ -4,17 +4,45 @@ use std::sync::Arc; -use arrow_array::{Array, ArrayRef, BooleanArray, RecordBatch}; +use arrow_array::{ + Array, ArrayRef, BooleanArray, Date32Array, Date64Array, Float32Array, Float64Array, + Int16Array, Int32Array, Int64Array, Int8Array, LargeStringArray, RecordBatch, Scalar, + StringArray, TimestampMicrosecondArray, TimestampMillisecondArray, + TimestampNanosecondArray, TimestampSecondArray, UInt16Array, UInt32Array, UInt64Array, + UInt8Array, +}; +use arrow_ord::cmp; use arrow_ord::sort::sort_to_indices; -use arrow_schema::SortOptions; +use arrow_schema::{DataType, SortOptions, TimeUnit}; use arrow_select::filter::filter_record_batch; use arrow_select::take::take; use rustler::ResourceArc; -use rustler::{Env, Term}; +use rustler::{Atom, Env, Term}; use crate::resources::ExArrowRecordBatch; use crate::util::{err_encode, ok_encode}; +rustler::atoms! { + field, + scalar, + call, + eq, + ne, + gt, + gte, + lt, + lte, + atom_and = "and", + atom_or = "or", + atom_not = "not", + date32, + date64, + timestamp_micros, + timestamp_millis, + timestamp_seconds, + timestamp_nanos, +} + /// Filter rows from `batch` using the first column of `predicate_batch` (must be boolean). /// /// Returns `{:ok, filtered_batch_ref}` or `{:error, msg}`. @@ -40,6 +68,601 @@ pub fn compute_filter<'a>( } } +/// Evaluate an encoded `Compute.Expression` AST to a boolean mask and filter `batch`. +/// +/// Encoding (Elixir → term tree): +/// - `{:field, name}` +/// - `{:scalar, value}` where value is bool/int/float/utf8 or +/// `{:date32|:date64|:timestamp_*, i}` +/// - `{:call, op, [args...]}` with op in eq/ne/gt/gte/lt/lte/and/or/not +#[rustler::nif] +pub fn compute_filter_expr<'a>( + env: Env<'a>, + batch: ResourceArc, + encoded: Term<'a>, +) -> Term<'a> { + let expr = match decode_expr(encoded) { + Ok(e) => e, + Err(msg) => return err_encode(env, &msg), + }; + let mask = match eval_to_bool(&batch.batch, &expr) { + Ok(m) => m, + Err(msg) => return err_encode(env, &msg), + }; + match filter_record_batch(&batch.batch, &mask) { + Ok(filtered) => ok_encode( + env, + ResourceArc::new(ExArrowRecordBatch { batch: filtered }), + ), + Err(e) => err_encode(env, &e.to_string()), + } +} + +// ── Expression AST (M2 residual filter) ────────────────────────────────────── + +#[derive(Debug, Clone)] +enum Expr { + Field(String), + Scalar(ScalarValue), + Call(Op, Vec), +} + +#[derive(Debug, Clone, Copy)] +enum Op { + Eq, + Ne, + Gt, + Gte, + Lt, + Lte, + And, + Or, + Not, +} + +#[derive(Debug, Clone)] +enum ScalarValue { + Bool(bool), + Int(i64), + Float(f64), + Utf8(String), + Date32(i32), + Date64(i64), + TimestampMicros(i64), + TimestampMillis(i64), + TimestampSeconds(i64), + TimestampNanos(i64), +} + +#[derive(Debug)] +enum Value { + Array(ArrayRef), + Lit(ScalarValue), +} + +fn decode_expr(term: Term<'_>) -> Result { + let tuple = rustler::types::tuple::get_tuple(term) + .map_err(|_| "expression must be a tuple {:field|:scalar|:call, ...}")?; + if tuple.is_empty() { + return Err("empty expression tuple".into()); + } + let tag: Atom = tuple[0] + .decode() + .map_err(|_| "expression tag must be an atom")?; + if tag == field() { + if tuple.len() != 2 { + return Err("{:field, name} expects exactly 2 elements".into()); + } + let name: String = tuple[1] + .decode() + .map_err(|_| "field name must be a UTF-8 string")?; + return Ok(Expr::Field(name)); + } + if tag == scalar() { + if tuple.len() != 2 { + return Err("{:scalar, value} expects exactly 2 elements".into()); + } + return Ok(Expr::Scalar(decode_scalar(tuple[1])?)); + } + if tag == call() { + if tuple.len() != 3 { + return Err("{:call, op, args} expects exactly 3 elements".into()); + } + let op = decode_op(tuple[1])?; + let list: rustler::types::list::ListIterator = tuple[2] + .decode() + .map_err(|_| "call args must be a list")?; + let args: Result, _> = list.map(decode_expr).collect(); + let args = args?; + match op { + Op::Not => { + if args.len() != 1 { + return Err(":not expects exactly one argument".into()); + } + } + Op::And | Op::Or | Op::Eq | Op::Ne | Op::Gt | Op::Gte | Op::Lt | Op::Lte => { + if args.len() != 2 { + return Err(format!("{:?} expects exactly two arguments", op)); + } + } + } + return Ok(Expr::Call(op, args)); + } + Err("expression tag must be :field, :scalar, or :call".into()) +} + +fn decode_op(term: Term<'_>) -> Result { + let a: Atom = term.decode().map_err(|_| "call op must be an atom")?; + if a == eq() { + Ok(Op::Eq) + } else if a == ne() { + Ok(Op::Ne) + } else if a == gt() { + Ok(Op::Gt) + } else if a == gte() { + Ok(Op::Gte) + } else if a == lt() { + Ok(Op::Lt) + } else if a == lte() { + Ok(Op::Lte) + } else if a == atom_and() { + Ok(Op::And) + } else if a == atom_or() { + Ok(Op::Or) + } else if a == atom_not() { + Ok(Op::Not) + } else { + Err("unsupported call op (eq ne gt gte lt lte and or not)".into()) + } +} + +fn decode_scalar(term: Term<'_>) -> Result { + if let Ok(b) = term.decode::() { + return Ok(ScalarValue::Bool(b)); + } + if let Ok(i) = term.decode::() { + return Ok(ScalarValue::Int(i)); + } + if let Ok(f) = term.decode::() { + return Ok(ScalarValue::Float(f)); + } + if let Ok(s) = term.decode::() { + return Ok(ScalarValue::Utf8(s)); + } + let tuple = rustler::types::tuple::get_tuple(term).map_err(|_| { + "scalar must be bool, integer, float, UTF-8 string, or {:date32|:date64|:timestamp_*, i}" + .to_string() + })?; + if tuple.len() != 2 { + return Err("temporal scalar expects {:unit, integer}".into()); + } + let unit: Atom = tuple[0] + .decode() + .map_err(|_| "temporal scalar unit must be an atom")?; + let v: i64 = tuple[1] + .decode() + .map_err(|_| "temporal scalar value must be an integer")?; + if unit == date32() { + let d = i32::try_from(v).map_err(|_| format!("date32 value {v} out of range"))?; + Ok(ScalarValue::Date32(d)) + } else if unit == date64() { + Ok(ScalarValue::Date64(v)) + } else if unit == timestamp_micros() { + Ok(ScalarValue::TimestampMicros(v)) + } else if unit == timestamp_millis() { + Ok(ScalarValue::TimestampMillis(v)) + } else if unit == timestamp_seconds() { + Ok(ScalarValue::TimestampSeconds(v)) + } else if unit == timestamp_nanos() { + Ok(ScalarValue::TimestampNanos(v)) + } else { + Err("unsupported temporal scalar unit".into()) + } +} + +fn eval_to_bool(batch: &RecordBatch, expr: &Expr) -> Result { + let value = eval_value(batch, expr)?; + value_as_bool(value, batch.num_rows()) +} + +fn eval_value(batch: &RecordBatch, expr: &Expr) -> Result { + match expr { + Expr::Field(name) => { + let col = batch.column_by_name(name).ok_or_else(|| { + format!("column '{}' not found", name) + })?; + Ok(Value::Array(Arc::clone(col))) + } + Expr::Scalar(s) => Ok(Value::Lit(s.clone())), + Expr::Call(op, args) => match op { + Op::And => { + let left = value_as_bool(eval_value(batch, &args[0])?, batch.num_rows())?; + let right = value_as_bool(eval_value(batch, &args[1])?, batch.num_rows())?; + arrow_arith::boolean::and(&left, &right).map_err(|e| e.to_string()) + .map(|a| Value::Array(Arc::new(a) as ArrayRef)) + } + Op::Or => { + let left = value_as_bool(eval_value(batch, &args[0])?, batch.num_rows())?; + let right = value_as_bool(eval_value(batch, &args[1])?, batch.num_rows())?; + arrow_arith::boolean::or(&left, &right).map_err(|e| e.to_string()) + .map(|a| Value::Array(Arc::new(a) as ArrayRef)) + } + Op::Not => { + let inner = value_as_bool(eval_value(batch, &args[0])?, batch.num_rows())?; + arrow_arith::boolean::not(&inner) + .map_err(|e| e.to_string()) + .map(|a| Value::Array(Arc::new(a) as ArrayRef)) + } + Op::Eq | Op::Ne | Op::Gt | Op::Gte | Op::Lt | Op::Lte => { + let left = eval_value(batch, &args[0])?; + let right = eval_value(batch, &args[1])?; + eval_cmp(left, right, *op, batch.num_rows()) + .map(|a| Value::Array(Arc::new(a) as ArrayRef)) + } + }, + } +} + +fn value_as_bool(value: Value, len: usize) -> Result { + match value { + Value::Array(arr) => { + let Some(b) = arr.as_any().downcast_ref::() else { + return Err(format!( + "expected boolean expression result, got column type {:?}", + arr.data_type() + )); + }; + if b.len() != len { + return Err(format!( + "boolean mask length {} does not match batch rows {}", + b.len(), + len + )); + } + Ok(b.clone()) + } + Value::Lit(ScalarValue::Bool(b)) => Ok(BooleanArray::from(vec![b; len])), + Value::Lit(other) => Err(format!( + "expected boolean expression result, got scalar {:?}", + other + )), + } +} + +fn eval_cmp(left: Value, right: Value, op: Op, len: usize) -> Result { + match (left, right) { + (Value::Array(l), Value::Array(r)) => { + if l.len() != r.len() { + return Err(format!( + "cannot compare arrays of lengths {} and {}", + l.len(), + r.len() + )); + } + // `&dyn Array` implements Datum; pass `&&dyn Array` so it coerces to `&dyn Datum`. + apply_cmp_dyn(l.as_ref(), r.as_ref(), op) + } + (Value::Array(l), Value::Lit(s)) => { + let scalar_arr = make_scalar_array(&s, l.data_type())?; + let scalar = Scalar::new(scalar_arr); + let lhs: &dyn Array = l.as_ref(); + match op { + Op::Eq => cmp::eq(&lhs, &scalar), + Op::Ne => cmp::neq(&lhs, &scalar), + Op::Gt => cmp::gt(&lhs, &scalar), + Op::Gte => cmp::gt_eq(&lhs, &scalar), + Op::Lt => cmp::lt(&lhs, &scalar), + Op::Lte => cmp::lt_eq(&lhs, &scalar), + Op::And | Op::Or | Op::Not => unreachable!("boolean ops handled separately"), + } + .map_err(|e| e.to_string()) + } + (Value::Lit(s), Value::Array(r)) => { + let scalar_arr = make_scalar_array(&s, r.data_type())?; + let scalar = Scalar::new(scalar_arr); + let rhs: &dyn Array = r.as_ref(); + match op { + Op::Eq => cmp::eq(&scalar, &rhs), + Op::Ne => cmp::neq(&scalar, &rhs), + Op::Gt => cmp::gt(&scalar, &rhs), + Op::Gte => cmp::gt_eq(&scalar, &rhs), + Op::Lt => cmp::lt(&scalar, &rhs), + Op::Lte => cmp::lt_eq(&scalar, &rhs), + Op::And | Op::Or | Op::Not => unreachable!("boolean ops handled separately"), + } + .map_err(|e| e.to_string()) + } + (Value::Lit(l), Value::Lit(r)) => { + let dt = infer_lit_compare_type(&l, &r)?; + let la = make_scalar_array(&l, &dt)?; + let ra = make_scalar_array(&r, &dt)?; + let ls = Scalar::new(la); + let rs = Scalar::new(ra); + let one = match op { + Op::Eq => cmp::eq(&ls, &rs), + Op::Ne => cmp::neq(&ls, &rs), + Op::Gt => cmp::gt(&ls, &rs), + Op::Gte => cmp::gt_eq(&ls, &rs), + Op::Lt => cmp::lt(&ls, &rs), + Op::Lte => cmp::lt_eq(&ls, &rs), + Op::And | Op::Or | Op::Not => unreachable!("boolean ops handled separately"), + } + .map_err(|e| e.to_string())?; + let flag = one.value(0); + Ok(BooleanArray::from(vec![flag; len])) + } + } +} + +fn apply_cmp_dyn(left: &dyn Array, right: &dyn Array, op: Op) -> Result { + match op { + Op::Eq => cmp::eq(&left, &right), + Op::Ne => cmp::neq(&left, &right), + Op::Gt => cmp::gt(&left, &right), + Op::Gte => cmp::gt_eq(&left, &right), + Op::Lt => cmp::lt(&left, &right), + Op::Lte => cmp::lt_eq(&left, &right), + Op::And | Op::Or | Op::Not => unreachable!("boolean ops handled separately"), + } + .map_err(|e| e.to_string()) +} + +fn infer_lit_compare_type(l: &ScalarValue, r: &ScalarValue) -> Result { + match (l, r) { + (ScalarValue::Bool(_), ScalarValue::Bool(_)) => Ok(DataType::Boolean), + (ScalarValue::Int(_), ScalarValue::Int(_)) => Ok(DataType::Int64), + (ScalarValue::Float(_), ScalarValue::Float(_)) + | (ScalarValue::Int(_), ScalarValue::Float(_)) + | (ScalarValue::Float(_), ScalarValue::Int(_)) => Ok(DataType::Float64), + (ScalarValue::Utf8(_), ScalarValue::Utf8(_)) => Ok(DataType::Utf8), + (ScalarValue::Date32(_), ScalarValue::Date32(_)) => Ok(DataType::Date32), + (ScalarValue::Date64(_), ScalarValue::Date64(_)) + | (ScalarValue::Date32(_), ScalarValue::Date64(_)) + | (ScalarValue::Date64(_), ScalarValue::Date32(_)) => Ok(DataType::Date64), + (ScalarValue::TimestampMicros(_), _) | (_, ScalarValue::TimestampMicros(_)) => { + Ok(DataType::Timestamp(TimeUnit::Microsecond, None)) + } + (ScalarValue::TimestampMillis(_), _) | (_, ScalarValue::TimestampMillis(_)) => { + Ok(DataType::Timestamp(TimeUnit::Millisecond, None)) + } + (ScalarValue::TimestampSeconds(_), _) | (_, ScalarValue::TimestampSeconds(_)) => { + Ok(DataType::Timestamp(TimeUnit::Second, None)) + } + (ScalarValue::TimestampNanos(_), _) | (_, ScalarValue::TimestampNanos(_)) => { + Ok(DataType::Timestamp(TimeUnit::Nanosecond, None)) + } + _ => Err(format!( + "cannot compare scalars {:?} and {:?}", + l, r + )), + } +} + +fn make_scalar_array(value: &ScalarValue, data_type: &DataType) -> Result { + match (value, data_type) { + (ScalarValue::Int(v), DataType::Int64) => { + Ok(Arc::new(Int64Array::from(vec![*v])) as ArrayRef) + } + (ScalarValue::Int(v), DataType::Int32) => { + let i = i32::try_from(*v).map_err(|_| { + format!("filter value {v} out of range for Int32 column") + })?; + Ok(Arc::new(Int32Array::from(vec![i])) as ArrayRef) + } + (ScalarValue::Int(v), DataType::Int16) => { + let i = i16::try_from(*v).map_err(|_| { + format!("filter value {v} out of range for Int16 column") + })?; + Ok(Arc::new(Int16Array::from(vec![i])) as ArrayRef) + } + (ScalarValue::Int(v), DataType::Int8) => { + let i = i8::try_from(*v).map_err(|_| { + format!("filter value {v} out of range for Int8 column") + })?; + Ok(Arc::new(Int8Array::from(vec![i])) as ArrayRef) + } + (ScalarValue::Int(v), DataType::UInt64) => { + if *v < 0 { + return Err(format!("filter value {v} out of range for UInt64 column")); + } + Ok(Arc::new(UInt64Array::from(vec![*v as u64])) as ArrayRef) + } + (ScalarValue::Int(v), DataType::UInt32) => { + let i = u32::try_from(*v).map_err(|_| { + format!("filter value {v} out of range for UInt32 column") + })?; + Ok(Arc::new(UInt32Array::from(vec![i])) as ArrayRef) + } + (ScalarValue::Int(v), DataType::UInt16) => { + let i = u16::try_from(*v).map_err(|_| { + format!("filter value {v} out of range for UInt16 column") + })?; + Ok(Arc::new(UInt16Array::from(vec![i])) as ArrayRef) + } + (ScalarValue::Int(v), DataType::UInt8) => { + let i = u8::try_from(*v).map_err(|_| { + format!("filter value {v} out of range for UInt8 column") + })?; + Ok(Arc::new(UInt8Array::from(vec![i])) as ArrayRef) + } + (ScalarValue::Float(v), DataType::Float64) => { + Ok(Arc::new(Float64Array::from(vec![*v])) as ArrayRef) + } + (ScalarValue::Int(v), DataType::Float64) => { + Ok(Arc::new(Float64Array::from(vec![*v as f64])) as ArrayRef) + } + (ScalarValue::Float(v), DataType::Float32) => { + let f = *v as f32; + if (f as f64) != *v { + return Err(format!( + "filter value {v} is not exactly representable as Float32" + )); + } + Ok(Arc::new(Float32Array::from(vec![f])) as ArrayRef) + } + (ScalarValue::Int(v), DataType::Float32) => { + let f = *v as f32; + if (f as i64) != *v { + return Err(format!( + "filter value {v} is not exactly representable as Float32" + )); + } + Ok(Arc::new(Float32Array::from(vec![f])) as ArrayRef) + } + (ScalarValue::Utf8(s), DataType::Utf8) => { + Ok(Arc::new(StringArray::from(vec![s.as_str()])) as ArrayRef) + } + (ScalarValue::Utf8(s), DataType::LargeUtf8) => { + Ok(Arc::new(LargeStringArray::from(vec![s.as_str()])) as ArrayRef) + } + (ScalarValue::Bool(b), DataType::Boolean) => { + Ok(Arc::new(BooleanArray::from(vec![*b])) as ArrayRef) + } + (ScalarValue::Date32(d), DataType::Date32) => { + Ok(Arc::new(Date32Array::from(vec![*d])) as ArrayRef) + } + (ScalarValue::Date32(d), DataType::Date64) => { + let millis = i64::from(*d) + .checked_mul(86_400_000) + .ok_or_else(|| format!("date32 {} overflows Date64 millis", d))?; + Ok(Arc::new(Date64Array::from(vec![millis])) as ArrayRef) + } + (ScalarValue::Date64(d), DataType::Date64) => { + Ok(Arc::new(Date64Array::from(vec![*d])) as ArrayRef) + } + (ScalarValue::Date64(d), DataType::Date32) => { + if *d % 86_400_000 != 0 { + return Err(format!( + "date64 value {d} is not an exact number of days for Date32" + )); + } + let days = *d / 86_400_000; + let i = i32::try_from(days).map_err(|_| { + format!("date64 value {d} out of range for Date32") + })?; + Ok(Arc::new(Date32Array::from(vec![i])) as ArrayRef) + } + (ScalarValue::Int(v), DataType::Date32) => { + let i = i32::try_from(*v).map_err(|_| { + format!("filter value {v} out of range for Date32 column") + })?; + Ok(Arc::new(Date32Array::from(vec![i])) as ArrayRef) + } + (ScalarValue::Int(v), DataType::Date64) => { + Ok(Arc::new(Date64Array::from(vec![*v])) as ArrayRef) + } + ( + ScalarValue::TimestampMicros(v) + | ScalarValue::TimestampMillis(v) + | ScalarValue::TimestampSeconds(v) + | ScalarValue::TimestampNanos(v) + | ScalarValue::Int(v), + DataType::Timestamp(unit, tz), + ) => make_timestamp_scalar(value, *v, *unit, tz.clone()), + _ => Err(format!( + "cannot compare filter value {:?} against column type {:?}", + value, data_type + )), + } +} + +fn make_timestamp_scalar( + original: &ScalarValue, + raw: i64, + unit: TimeUnit, + tz: Option>, +) -> Result { + let micros = match original { + ScalarValue::TimestampMicros(v) => *v, + ScalarValue::TimestampMillis(v) => v + .checked_mul(1_000) + .ok_or_else(|| format!("timestamp millis {v} overflows micros"))?, + ScalarValue::TimestampSeconds(v) => v + .checked_mul(1_000_000) + .ok_or_else(|| format!("timestamp seconds {v} overflows micros"))?, + ScalarValue::TimestampNanos(v) => { + if *v % 1_000 != 0 { + return Err(format!( + "timestamp nanos {v} is not an exact number of microseconds" + )); + } + *v / 1_000 + } + ScalarValue::Int(v) => { + // Bare integers are already in the column's unit (validate path). + return timestamp_array_from_ticks(*v, unit, tz); + } + _ => raw, + }; + let ticks = micros_to_unit(micros, unit)?; + timestamp_array_from_ticks(ticks, unit, tz) +} + +fn micros_to_unit(micros: i64, unit: TimeUnit) -> Result { + match unit { + TimeUnit::Microsecond => Ok(micros), + TimeUnit::Millisecond => { + if micros % 1_000 != 0 { + return Err(format!( + "timestamp micros {micros} is not an exact number of milliseconds" + )); + } + Ok(micros / 1_000) + } + TimeUnit::Second => { + if micros % 1_000_000 != 0 { + return Err(format!( + "timestamp micros {micros} is not an exact number of seconds" + )); + } + Ok(micros / 1_000_000) + } + TimeUnit::Nanosecond => micros + .checked_mul(1_000) + .ok_or_else(|| format!("timestamp micros {micros} overflows nanos")), + } +} + +fn timestamp_array_from_ticks( + ticks: i64, + unit: TimeUnit, + tz: Option>, +) -> Result { + // arrow-array constructors ignore tz on the array values; DataType carries tz. + // Scalar::new uses the array's data_type, so build typed arrays then cast schema via + // with_timezone when needed. + match unit { + TimeUnit::Second => { + let arr = TimestampSecondArray::from(vec![ticks]); + Ok(Arc::new(match tz { + Some(tz) => arr.with_timezone(tz), + None => arr, + }) as ArrayRef) + } + TimeUnit::Millisecond => { + let arr = TimestampMillisecondArray::from(vec![ticks]); + Ok(Arc::new(match tz { + Some(tz) => arr.with_timezone(tz), + None => arr, + }) as ArrayRef) + } + TimeUnit::Microsecond => { + let arr = TimestampMicrosecondArray::from(vec![ticks]); + Ok(Arc::new(match tz { + Some(tz) => arr.with_timezone(tz), + None => arr, + }) as ArrayRef) + } + TimeUnit::Nanosecond => { + let arr = TimestampNanosecondArray::from(vec![ticks]); + Ok(Arc::new(match tz { + Some(tz) => arr.with_timezone(tz), + None => arr, + }) as ArrayRef) + } + } +} + /// Project (select) a subset of columns from `batch` by name. /// /// Returns `{:ok, projected_batch_ref}` or `{:error, msg}`. diff --git a/test/ex_arrow/compute_filter_expr_test.exs b/test/ex_arrow/compute_filter_expr_test.exs new file mode 100644 index 0000000..4bca3eb --- /dev/null +++ b/test/ex_arrow/compute_filter_expr_test.exs @@ -0,0 +1,148 @@ +defmodule ExArrow.ComputeFilterExprTest do + use ExUnit.Case, async: true + + alias ExArrow.Batch + alias ExArrow.Compute + alias ExArrow.Compute.Expression, as: E + alias ExArrow.Native + alias ExArrow.RecordBatch + + defp s64_column(batch, name) do + ref = RecordBatch.resource_ref(batch) + {:ok, {binary, "s64", _n}} = Native.record_batch_column_buffer(ref, name) + for <>, do: v + end + + defp f64_column(batch, name) do + ref = RecordBatch.resource_ref(batch) + {:ok, {binary, "f64", _n}} = Native.record_batch_column_buffer(ref, name) + for <>, do: v + end + + defp sample_batch do + assert {:ok, batch} = + RecordBatch.from_lists([ + {"id", :s64, [1, 2, 3, 4]}, + {"score", :f64, [0.5, 0.95, 0.91, 0.2]}, + {"name", :utf8, ["a", "b", "c", "d"]}, + {"ok", :bool, [true, false, true, false]} + ]) + + batch + end + + @tag :nif + test "filters rows with score > 0.9" do + batch = sample_batch() + expr = E.gt(E.field("score"), E.scalar(0.9)) + + assert {:ok, filtered} = Compute.filter(batch, expr) + assert RecordBatch.num_rows(filtered) == 2 + assert s64_column(filtered, "id") == [2, 3] + assert f64_column(filtered, "score") == [0.95, 0.91] + end + + @tag :nif + test "Batch.filter/2 accepts Expression" do + batch = sample_batch() + expr = E.lte(E.field("id"), E.scalar(2)) + + assert {:ok, filtered} = Batch.filter(batch, expr) + assert s64_column(filtered, "id") == [1, 2] + end + + @tag :nif + test "and_/or_/not_ compose" do + batch = sample_batch() + + expr = + E.and_( + E.gt(E.field("score"), E.scalar(0.9)), + E.eq(E.field("ok"), E.scalar(true)) + ) + + assert {:ok, filtered} = Compute.filter(batch, expr) + assert s64_column(filtered, "id") == [3] + + expr2 = E.or_(E.eq(E.field("id"), E.scalar(1)), E.eq(E.field("id"), E.scalar(4))) + assert {:ok, filtered2} = Compute.filter(batch, expr2) + assert s64_column(filtered2, "id") == [1, 4] + + expr3 = E.not_(E.eq(E.field("ok"), E.scalar(true))) + assert {:ok, filtered3} = Compute.filter(batch, expr3) + assert s64_column(filtered3, "id") == [2, 4] + end + + @tag :nif + test "utf8 equality and field-vs-field compare" do + batch = sample_batch() + + assert {:ok, filtered} = Compute.filter(batch, E.eq(E.field("name"), E.scalar("b"))) + assert s64_column(filtered, "id") == [2] + + assert {:ok, batch2} = + RecordBatch.from_lists([ + {"a", :s64, [1, 5, 3]}, + {"b", :s64, [1, 2, 3]} + ]) + + assert {:ok, equal} = Compute.filter(batch2, E.eq(E.field("a"), E.field("b"))) + assert s64_column(equal, "a") == [1, 3] + end + + @tag :nif + test "date32 and timestamp_micros residual filters" do + days = [ + Date.diff(~D[2025-12-31], ~D[1970-01-01]), + Date.diff(~D[2026-01-01], ~D[1970-01-01]), + Date.diff(~D[2026-06-01], ~D[1970-01-01]) + ] + + micros = [ + NaiveDateTime.diff(~N[2026-01-01 00:00:00], ~N[1970-01-01 00:00:00], :microsecond), + NaiveDateTime.diff(~N[2026-01-02 12:00:00], ~N[1970-01-01 00:00:00], :microsecond), + NaiveDateTime.diff(~N[2025-01-01 00:00:00], ~N[1970-01-01 00:00:00], :microsecond) + ] + + assert {:ok, batch} = + RecordBatch.from_lists([ + {"id", :s64, [1, 2, 3]}, + {"day", :date32, days}, + {"ts", :timestamp_micros, micros} + ]) + + assert {:ok, by_date} = + Compute.filter(batch, E.gte(E.field("day"), E.scalar(~D[2026-01-01]))) + + assert s64_column(by_date, "id") == [2, 3] + + assert {:ok, by_ts} = + Compute.filter( + batch, + E.gt(E.field("ts"), E.scalar(~N[2026-01-01 00:00:00])) + ) + + assert s64_column(by_ts, "id") == [2] + end + + @tag :nif + test "errors on unknown column and type mismatch" do + batch = sample_batch() + + assert {:error, msg} = Compute.filter(batch, E.eq(E.field("missing"), E.scalar(1))) + assert msg =~ "missing" + + assert {:error, msg} = Compute.filter(batch, E.eq(E.field("score"), E.scalar("x"))) + assert msg =~ ~r/cannot compare|type/i + end + + @tag :nif + test "errors on Int32 out-of-range scalar" do + assert {:ok, batch} = RecordBatch.from_lists([{"x", :s32, [1, 2]}]) + + assert {:error, msg} = + Compute.filter(batch, E.eq(E.field("x"), E.scalar(2_147_483_648))) + + assert msg =~ "out of range" + end +end diff --git a/test/ex_arrow/native_test.exs b/test/ex_arrow/native_test.exs index 6d27aff..d2fb042 100644 --- a/test/ex_arrow/native_test.exs +++ b/test/ex_arrow/native_test.exs @@ -109,6 +109,11 @@ defmodule ExArrow.NativeTest do assert_raise ErlangError, fn -> ExArrow.Native.compute_filter(:fake, :fake) end end + @tag :no_nif + test "compute_filter_expr/2 raises nif_not_loaded" do + assert_raise ErlangError, fn -> ExArrow.Native.compute_filter_expr(:fake, :fake) end + end + @tag :no_nif test "adbc_database_open/1 raises nif_not_loaded" do assert_raise ErlangError, fn -> ExArrow.Native.adbc_database_open("fake.so") end From 5f6b7b5976f2cc8ae5557ec84e7d1ea2c9782c5c Mon Sep 17 00:00:00 2001 From: thanos Date: Sat, 12 Sep 2026 14:43:48 -0400 Subject: [PATCH 05/11] M3: add FileSystem behaviour with Local and Memory backends. Provide list/glob/exists discovery over the OS or an in-memory tree so Dataset can open the same path logic in tests without tmp dirs. - closed #286 - closed #287 - closed #288 - closed #289 --- lib/ex_arrow/file_system.ex | 172 ++++++++++++++++++++ lib/ex_arrow/file_system/local.ex | 142 +++++++++++++++++ lib/ex_arrow/file_system/memory.ex | 246 +++++++++++++++++++++++++++++ test/ex_arrow/file_system_test.exs | 210 ++++++++++++++++++++++++ 4 files changed, 770 insertions(+) create mode 100644 lib/ex_arrow/file_system.ex create mode 100644 lib/ex_arrow/file_system/local.ex create mode 100644 lib/ex_arrow/file_system/memory.ex create mode 100644 test/ex_arrow/file_system_test.exs diff --git a/lib/ex_arrow/file_system.ex b/lib/ex_arrow/file_system.ex new file mode 100644 index 0000000..5cb8045 --- /dev/null +++ b/lib/ex_arrow/file_system.ex @@ -0,0 +1,172 @@ +defmodule ExArrow.FileSystem do + @moduledoc """ + Capability-oriented filesystem abstraction for Dataset discovery. + + Reads still go through path-based NIFs (`Parquet`, `IPC`). This module only + answers discovery questions: what paths exist, which match a glob, and + whether a path is present. + + ## Implementations + + - `ExArrow.FileSystem.Local` — OS filesystem (default for Dataset) + - `ExArrow.FileSystem.Memory` — in-memory tree for tests (no tmp dirs) + + S3 / object-store adapters are out of scope for 0.9.0; the behaviour leaves + room for them later. + + ## Hidden entries + + When `ignore_hidden: true` (the default), any path component whose basename + starts with `.` or `_` is skipped. That matches Dataset's + `:ignore_hidden` option (dotfiles and `_`-prefixed Hive / staging dirs). + + ## Example + + fs = ExArrow.FileSystem.Local.new() + {:ok, entries} = ExArrow.FileSystem.list(fs, "/data/events") + {:ok, paths} = ExArrow.FileSystem.glob(fs, "/data/events/**/*.parquet") + true = ExArrow.FileSystem.exists?(fs, "/data/events") + """ + + @typedoc "Filesystem handle (struct whose module implements this behaviour)." + @type t :: struct() + + @typedoc "One discovered path." + @type entry :: %{ + path: String.t(), + type: :file | :directory, + size: non_neg_integer() + } + + @type list_opt :: {:recursive, boolean()} | {:ignore_hidden, boolean()} + @type glob_opt :: {:ignore_hidden, boolean()} + + @callback list(t(), String.t(), keyword()) :: {:ok, [entry()]} | {:error, String.t()} + @callback glob(t(), String.t(), keyword()) :: {:ok, [String.t()]} | {:error, String.t()} + @callback exists?(t(), String.t()) :: boolean() + + @doc """ + List entries under `path`. + + ## Options + + * `:recursive` — when `true` (default), walk the whole tree; when `false`, + only immediate children + * `:ignore_hidden` — when `true` (default), skip `.` / `_`-prefixed names + """ + @spec list(t(), String.t(), [list_opt()]) :: {:ok, [entry()]} | {:error, String.t()} + def list(fs, path, opts \\ []) + + def list(%mod{} = fs, path, opts) when is_binary(path) and is_list(opts) do + with :ok <- validate_opts(opts, [:recursive, :ignore_hidden]) do + mod.list(fs, path, opts) + end + end + + def list(_fs, path, _opts) when not is_binary(path), + do: {:error, "path must be a UTF-8 string"} + + def list(_fs, _path, opts) when not is_list(opts), + do: {:error, "opts must be a keyword list"} + + @doc """ + Return file paths matching `pattern` (sorted). + + Patterns use `/` separators. `*` matches within one path segment; `**` + matches across segments (including zero segments). + + ## Options + + * `:ignore_hidden` — when `true` (default), skip matches with a `.` / + `_`-prefixed path component + """ + @spec glob(t(), String.t(), [glob_opt()]) :: {:ok, [String.t()]} | {:error, String.t()} + def glob(fs, pattern, opts \\ []) + + def glob(%mod{} = fs, pattern, opts) when is_binary(pattern) and is_list(opts) do + with :ok <- validate_opts(opts, [:ignore_hidden]) do + mod.glob(fs, pattern, opts) + end + end + + def glob(_fs, pattern, _opts) when not is_binary(pattern), + do: {:error, "pattern must be a UTF-8 string"} + + def glob(_fs, _pattern, opts) when not is_list(opts), + do: {:error, "opts must be a keyword list"} + + @doc """ + Return whether `path` exists as a file or directory. + """ + @spec exists?(t(), String.t()) :: boolean() + def exists?(%mod{} = fs, path) when is_binary(path), do: mod.exists?(fs, path) + def exists?(_fs, _path), do: false + + @doc false + @spec hidden_basename?(String.t()) :: boolean() + def hidden_basename?(name) when is_binary(name) do + name != "." and name != ".." and + (String.starts_with?(name, ".") or String.starts_with?(name, "_")) + end + + @doc false + @spec path_has_hidden_component?(String.t()) :: boolean() + def path_has_hidden_component?(path) when is_binary(path) do + path + |> Path.split() + |> Enum.any?(&hidden_basename?/1) + end + + @doc false + @spec match_glob?(String.t(), String.t()) :: boolean() + def match_glob?(path, pattern) when is_binary(path) and is_binary(pattern) do + match_parts?(Path.split(path), Path.split(pattern)) + end + + defp match_parts?([], []), do: true + defp match_parts?(_path, ["**"]), do: true + + defp match_parts?(path, ["**" | rest_pat]) do + Enum.any?(0..length(path), fn n -> + match_parts?(Enum.drop(path, n), rest_pat) + end) + end + + defp match_parts?([name | path_rest], [pat | pat_rest]) do + match_segment?(name, pat) and match_parts?(path_rest, pat_rest) + end + + defp match_parts?([], _pat), do: false + defp match_parts?(_path, []), do: false + + defp match_segment?(_name, "*"), do: true + + defp match_segment?(name, pat) do + if String.contains?(pat, "*") do + regex = + pat + |> Regex.escape() + |> String.replace("\\*", ".*") + |> then(&("^" <> &1 <> "$")) + |> Regex.compile!() + + Regex.match?(regex, name) + else + name == pat + end + end + + defp validate_opts(opts, allowed) do + if Keyword.keyword?(opts) do + bad = Enum.reject(Keyword.keys(opts), &(&1 in allowed)) + + if bad == [] do + :ok + else + {:error, "unknown option(s): #{inspect(bad)}"} + end + else + {:error, "opts must be a keyword list"} + end + end +end diff --git a/lib/ex_arrow/file_system/local.ex b/lib/ex_arrow/file_system/local.ex new file mode 100644 index 0000000..73e0f96 --- /dev/null +++ b/lib/ex_arrow/file_system/local.ex @@ -0,0 +1,142 @@ +defmodule ExArrow.FileSystem.Local do + @moduledoc """ + Local OS filesystem implementation of `ExArrow.FileSystem`. + + Paths are expanded with `Path.expand/1` before use. File contents are not + read here; Dataset / Parquet NIFs open paths returned by discovery. + """ + + @behaviour ExArrow.FileSystem + + alias ExArrow.FileSystem + + defstruct [] + + @type t :: %__MODULE__{} + + @doc """ + Build a local filesystem handle. + """ + @spec new() :: t() + def new, do: %__MODULE__{} + + @impl true + def list(%__MODULE__{}, path, opts) when is_binary(path) and is_list(opts) do + recursive = Keyword.get(opts, :recursive, true) + ignore_hidden = Keyword.get(opts, :ignore_hidden, true) + root = Path.expand(path) + + cond do + not File.exists?(root) -> + {:error, "path does not exist: #{root}"} + + File.regular?(root) -> + list_file(root, ignore_hidden) + + File.dir?(root) -> + case collect_dir(root, recursive, ignore_hidden) do + {:ok, entries} -> {:ok, Enum.sort_by(entries, & &1.path)} + {:error, _} = err -> err + end + + true -> + {:error, "path is not a file or directory: #{root}"} + end + end + + @impl true + def glob(%__MODULE__{}, pattern, opts) when is_binary(pattern) and is_list(opts) do + ignore_hidden = Keyword.get(opts, :ignore_hidden, true) + + try do + paths = + pattern + |> Path.wildcard(match_dot: not ignore_hidden) + |> Enum.filter(&File.regular?/1) + |> Enum.map(&Path.expand/1) + |> Enum.reject(fn p -> + ignore_hidden and FileSystem.path_has_hidden_component?(p) + end) + |> Enum.sort() + + {:ok, paths} + rescue + e in [ErlangError, ArgumentError] -> + {:error, "invalid glob pattern: #{Exception.message(e)}"} + end + end + + @impl true + def exists?(%__MODULE__{}, path) when is_binary(path) do + File.exists?(Path.expand(path)) + end + + defp list_file(root, ignore_hidden) do + if ignore_hidden and FileSystem.path_has_hidden_component?(root) do + {:ok, []} + else + case file_entry(root) do + {:ok, entry} -> {:ok, [entry]} + {:error, _} = err -> err + end + end + end + + defp collect_dir(dir, recursive, ignore_hidden) do + case File.ls(dir) do + {:ok, names} -> reduce_names(Enum.sort(names), dir, recursive, ignore_hidden, []) + {:error, reason} -> {:error, "cannot list #{dir}: #{inspect(reason)}"} + end + end + + defp reduce_names([], _dir, _recursive, _ignore_hidden, acc), do: {:ok, acc} + + defp reduce_names([name | rest], dir, recursive, ignore_hidden, acc) do + if ignore_hidden and FileSystem.hidden_basename?(name) do + reduce_names(rest, dir, recursive, ignore_hidden, acc) + else + case append_child(Path.join(dir, name), recursive, ignore_hidden, acc) do + {:ok, acc2} -> reduce_names(rest, dir, recursive, ignore_hidden, acc2) + {:error, _} = err -> err + end + end + end + + defp append_child(full, recursive, ignore_hidden, acc) do + cond do + File.dir?(full) -> + entry = %{path: full, type: :directory, size: 0} + + if recursive do + case collect_dir(full, true, ignore_hidden) do + {:ok, child} -> {:ok, acc ++ [entry | child]} + {:error, _} = err -> err + end + else + {:ok, [entry | acc]} + end + + File.regular?(full) -> + case file_entry(full) do + {:ok, entry} -> {:ok, [entry | acc]} + {:error, _} = err -> err + end + + true -> + {:ok, acc} + end + end + + defp file_entry(path) do + case File.stat(path) do + {:ok, %File.Stat{size: size}} when is_integer(size) and size >= 0 -> + {:ok, %{path: path, type: :file, size: size}} + + {:ok, _} -> + {:error, "cannot stat file size for #{path}"} + + {:error, reason} -> + {:error, "cannot stat #{path}: #{inspect(reason)}"} + end + end +end diff --git a/lib/ex_arrow/file_system/memory.ex b/lib/ex_arrow/file_system/memory.ex new file mode 100644 index 0000000..15575af --- /dev/null +++ b/lib/ex_arrow/file_system/memory.ex @@ -0,0 +1,246 @@ +defmodule ExArrow.FileSystem.Memory do + @moduledoc """ + In-memory filesystem for Dataset discovery tests. + + Holds a flat map of normalized absolute paths to entries. Adding a file + also registers parent directories so `list/3` can walk a Hive-style tree + without touching the OS. + + ## Examples + + fs = ExArrow.FileSystem.Memory.new() + {:ok, fs} = ExArrow.FileSystem.Memory.put_file(fs, "/data/a.parquet", size: 128) + + {:ok, fs} = + ExArrow.FileSystem.Memory.new(%{ + "/data/year=2026/part-0.parquet" => 128 + }) + """ + + @behaviour ExArrow.FileSystem + + alias ExArrow.FileSystem + + defstruct entries: %{} + + @type entry_map :: %{optional(String.t()) => FileSystem.entry()} + @type t :: %__MODULE__{entries: entry_map()} + + @doc """ + Build an empty memory filesystem. + """ + @spec new() :: t() + def new, do: %__MODULE__{} + + @doc """ + Build a memory filesystem from a path → size map or `{path, size}` list. + + Returns `{:ok, fs}` or `{:error, message}`. + """ + @spec new(map() | [{String.t(), non_neg_integer()}]) :: + {:ok, t()} | {:error, String.t()} + def new(seed) when is_map(seed), do: seed |> Map.to_list() |> new_from_list() + def new(seed) when is_list(seed), do: new_from_list(seed) + + @doc """ + Register a file at `path` with `size` (default `0`). + + Creates missing parent directories. Returns `{:ok, fs}` or `{:error, msg}`. + """ + @spec put_file(t(), String.t(), keyword()) :: {:ok, t()} | {:error, String.t()} + def put_file(fs, path, opts \\ []) + + def put_file(%__MODULE__{} = fs, path, opts) when is_binary(path) and is_list(opts) do + with :ok <- validate_put_opts(opts), + {:ok, norm} <- normalize_path(path) do + size = Keyword.get(opts, :size, 0) + + if not is_integer(size) or size < 0 do + {:error, "size must be a non-negative integer"} + else + entries = + norm + |> parent_dirs() + |> Enum.reduce(fs.entries, fn dir, acc -> + Map.put_new(acc, dir, %{path: dir, type: :directory, size: 0}) + end) + |> Map.put(norm, %{path: norm, type: :file, size: size}) + + {:ok, %{fs | entries: entries}} + end + end + end + + def put_file(%__MODULE__{}, path, _opts) when not is_binary(path), + do: {:error, "path must be a UTF-8 string"} + + def put_file(%__MODULE__{}, _path, opts) when not is_list(opts), + do: {:error, "opts must be a keyword list"} + + @impl true + def list(%__MODULE__{} = fs, path, opts) when is_binary(path) and is_list(opts) do + recursive = Keyword.get(opts, :recursive, true) + ignore_hidden = Keyword.get(opts, :ignore_hidden, true) + + with {:ok, root} <- normalize_path(path) do + list_at(fs, root, recursive, ignore_hidden) + end + end + + defp list_at(fs, root, recursive, ignore_hidden) do + case Map.fetch(fs.entries, root) do + {:ok, %{type: :file} = entry} -> + list_file_entry(entry, root, ignore_hidden) + + {:ok, %{type: :directory}} -> + {:ok, select_children(fs, root, recursive, ignore_hidden, include_root?: false)} + + :error -> + if implicit_dir?(fs, root) do + {:ok, select_children(fs, root, recursive, ignore_hidden, include_root?: false)} + else + {:error, "path does not exist: #{root}"} + end + end + end + + defp list_file_entry(entry, root, ignore_hidden) do + if ignore_hidden and FileSystem.path_has_hidden_component?(root) do + {:ok, []} + else + {:ok, [entry]} + end + end + + defp select_children(fs, root, recursive, ignore_hidden, include_root?: include_root?) do + fs.entries + |> Map.values() + |> Enum.filter(fn %{path: p} -> child_path?(p, root, recursive, include_root?) end) + |> Enum.reject(fn %{path: p} -> + ignore_hidden and hidden_under_root?(p, root) + end) + |> Enum.sort_by(& &1.path) + end + + defp child_path?(path, root, recursive, include_root?) do + cond do + path == root -> include_root? + not under?(path, root) -> false + recursive -> true + true -> Path.dirname(path) == root + end + end + + @impl true + def glob(%__MODULE__{} = fs, pattern, opts) when is_binary(pattern) and is_list(opts) do + ignore_hidden = Keyword.get(opts, :ignore_hidden, true) + + with {:ok, norm_pat} <- normalize_path(pattern) do + paths = + fs.entries + |> Map.values() + |> Enum.filter(&(&1.type == :file)) + |> Enum.map(& &1.path) + |> Enum.filter(&FileSystem.match_glob?(&1, norm_pat)) + |> Enum.reject(fn p -> + ignore_hidden and FileSystem.path_has_hidden_component?(p) + end) + |> Enum.sort() + + {:ok, paths} + end + end + + @impl true + def exists?(%__MODULE__{} = fs, path) when is_binary(path) do + case normalize_path(path) do + {:ok, root} -> + Map.has_key?(fs.entries, root) or implicit_dir?(fs, root) + + {:error, _} -> + false + end + end + + defp new_from_list(list) do + Enum.reduce_while(list, {:ok, %__MODULE__{}}, fn + {path, size}, {:ok, fs} when is_binary(path) and is_integer(size) and size >= 0 -> + case put_file(fs, path, size: size) do + {:ok, fs2} -> {:cont, {:ok, fs2}} + {:error, _} = err -> {:halt, err} + end + + other, _ -> + {:halt, {:error, "seed entries must be {path, size} pairs, got: #{inspect(other)}"}} + end) + end + + defp validate_put_opts(opts) do + if Keyword.keyword?(opts) do + bad = Enum.reject(Keyword.keys(opts), &(&1 in [:size])) + + if bad == [], do: :ok, else: {:error, "unknown option(s): #{inspect(bad)}"} + else + {:error, "opts must be a keyword list"} + end + end + + defp normalize_path(path) when is_binary(path) do + if String.valid?(path) do + norm = + path + |> String.replace("\\", "/") + |> String.replace(~r/\/+/, "/") + |> ensure_absolute() + |> trim_trailing_slash() + + if norm == "" do + {:error, "path must be a UTF-8 string"} + else + {:ok, norm} + end + else + {:error, "path must be a UTF-8 string"} + end + end + + defp ensure_absolute("/" <> _ = path), do: path + defp ensure_absolute(path), do: "/" <> path + + defp trim_trailing_slash("/"), do: "/" + defp trim_trailing_slash(path), do: String.trim_trailing(path, "/") + + defp parent_dirs("/"), do: [] + + defp parent_dirs(path) do + path + |> Path.split() + |> Enum.drop(-1) + |> Enum.scan(fn part, acc -> Path.join(acc, part) end) + |> Enum.map(fn + "/" <> _ = p -> p + p -> "/" <> p + end) + end + + defp under?(path, root) do + root == "/" or String.starts_with?(path, root <> "/") + end + + defp implicit_dir?(%__MODULE__{entries: entries}, root) do + Enum.any?(entries, fn {p, _} -> under?(p, root) end) + end + + defp hidden_under_root?(path, root) do + relative = + cond do + root == "/" -> path + String.starts_with?(path, root <> "/") -> String.replace_prefix(path, root <> "/", "") + true -> path + end + + relative + |> Path.split() + |> Enum.any?(&FileSystem.hidden_basename?/1) + end +end diff --git a/test/ex_arrow/file_system_test.exs b/test/ex_arrow/file_system_test.exs new file mode 100644 index 0000000..221175c --- /dev/null +++ b/test/ex_arrow/file_system_test.exs @@ -0,0 +1,210 @@ +defmodule ExArrow.FileSystemTest do + use ExUnit.Case, async: true + + alias ExArrow.FileSystem + alias ExArrow.FileSystem.Local + alias ExArrow.FileSystem.Memory + + describe "dispatch validation" do + test "list/3 rejects non-string path and unknown opts" do + fs = Local.new() + assert {:error, msg} = FileSystem.list(fs, :not_a_path) + assert msg =~ "UTF-8" + + assert {:error, msg} = FileSystem.list(fs, "/tmp", bogus: true) + assert msg =~ "unknown option" + end + + test "glob/3 rejects non-string pattern" do + fs = Local.new() + assert {:error, msg} = FileSystem.glob(fs, 123) + assert msg =~ "pattern" + end + + test "exists?/2 is false for non-string path" do + refute FileSystem.exists?(Local.new(), :nope) + end + end + + describe "Local" do + @tag :tmp_dir + test "list/3 recursive discovers files and directories with exact sizes", %{tmp_dir: dir} do + hive = Path.join(dir, "year=2026") + File.mkdir_p!(hive) + file = Path.join(hive, "part-0.parquet") + File.write!(file, "abcdefgh") + + hidden_dir = Path.join(dir, ".staging") + File.mkdir_p!(hidden_dir) + File.write!(Path.join(hidden_dir, "secret.parquet"), "x") + + underscored = Path.join(dir, "_temporary") + File.mkdir_p!(underscored) + File.write!(Path.join(underscored, "tmp.parquet"), "y") + + fs = Local.new() + assert FileSystem.exists?(fs, dir) + + assert {:ok, entries} = FileSystem.list(fs, dir) + paths = Enum.map(entries, & &1.path) + + assert Path.expand(hive) in paths + assert Path.expand(file) in paths + refute Enum.any?(paths, &String.contains?(&1, ".staging")) + refute Enum.any?(paths, &String.contains?(&1, "_temporary")) + + file_entry = Enum.find(entries, &(&1.path == Path.expand(file))) + assert file_entry.type == :file + assert file_entry.size == 8 + + dir_entry = Enum.find(entries, &(&1.path == Path.expand(hive))) + assert dir_entry.type == :directory + assert dir_entry.size == 0 + end + + @tag :tmp_dir + test "list/3 non-recursive returns only immediate children", %{tmp_dir: dir} do + nested = Path.join(dir, "a/b") + File.mkdir_p!(nested) + File.write!(Path.join(nested, "f.parquet"), "z") + File.write!(Path.join(dir, "top.parquet"), "tt") + + fs = Local.new() + assert {:ok, entries} = FileSystem.list(fs, dir, recursive: false) + paths = entries |> Enum.map(& &1.path) |> Enum.sort() + + assert paths == + Enum.sort([ + Path.expand(Path.join(dir, "a")), + Path.expand(Path.join(dir, "top.parquet")) + ]) + end + + @tag :tmp_dir + test "list/3 ignore_hidden: false includes dot and underscore entries", %{tmp_dir: dir} do + File.mkdir_p!(Path.join(dir, ".hidden")) + File.write!(Path.join(dir, ".hidden/x.parquet"), "1") + File.mkdir_p!(Path.join(dir, "_tmp")) + File.write!(Path.join(dir, "_tmp/y.parquet"), "22") + + fs = Local.new() + assert {:ok, entries} = FileSystem.list(fs, dir, ignore_hidden: false) + paths = Enum.map(entries, & &1.path) + + assert Enum.any?(paths, &String.contains?(&1, ".hidden")) + assert Enum.any?(paths, &String.contains?(&1, "_tmp")) + end + + @tag :tmp_dir + test "glob/3 matches parquet files and respects ignore_hidden", %{tmp_dir: dir} do + File.mkdir_p!(Path.join(dir, "year=2026")) + keep = Path.join(dir, "year=2026/part-0.parquet") + File.write!(keep, "abc") + File.mkdir_p!(Path.join(dir, ".skip")) + File.write!(Path.join(dir, ".skip/no.parquet"), "no") + + fs = Local.new() + pattern = Path.join(dir, "**/*.parquet") + + assert {:ok, [only]} = FileSystem.glob(fs, pattern) + assert only == Path.expand(keep) + + assert {:ok, paths} = FileSystem.glob(fs, pattern, ignore_hidden: false) + assert length(paths) == 2 + assert Path.expand(keep) in paths + end + + test "list/3 errors when path is missing" do + fs = Local.new() + + missing = + Path.join(System.tmp_dir!(), "ex-arrow-missing-#{System.unique_integer([:positive])}") + + assert {:error, msg} = FileSystem.list(fs, missing) + assert msg =~ "does not exist" + end + end + + describe "Memory" do + test "new/1 seeds files and parent directories" do + assert {:ok, fs} = + Memory.new(%{ + "/data/year=2026/part-0.parquet" => 128, + "/data/year=2025/part-0.parquet" => 64 + }) + + assert FileSystem.exists?(fs, "/data") + assert FileSystem.exists?(fs, "/data/year=2026") + assert FileSystem.exists?(fs, "/data/year=2026/part-0.parquet") + refute FileSystem.exists?(fs, "/data/year=2024") + + assert {:ok, entries} = FileSystem.list(fs, "/data") + files = Enum.filter(entries, &(&1.type == :file)) + file_paths = files |> Enum.map(& &1.path) |> Enum.sort() + + assert file_paths == [ + "/data/year=2025/part-0.parquet", + "/data/year=2026/part-0.parquet" + ] + + assert Enum.find(files, &(&1.path == "/data/year=2026/part-0.parquet")).size == 128 + end + + test "list/3 non-recursive and ignore_hidden" do + assert {:ok, fs} = + Memory.new([ + {"/data/year=2026/part.parquet", 1}, + {"/data/.staging/secret.parquet", 2}, + {"/data/_tmp/x.parquet", 3} + ]) + + assert {:ok, entries} = FileSystem.list(fs, "/data", recursive: false) + paths = entries |> Enum.map(& &1.path) |> Enum.sort() + assert paths == ["/data/year=2026"] + + assert {:ok, all_hidden} = FileSystem.list(fs, "/data", ignore_hidden: false) + assert Enum.any?(all_hidden, &String.contains?(&1.path, ".staging")) + assert Enum.any?(all_hidden, &String.contains?(&1.path, "_tmp")) + end + + test "glob/3 supports * and **" do + assert {:ok, fs} = + Memory.new([ + {"/data/year=2026/part-0.parquet", 1}, + {"/data/year=2025/part-0.parquet", 1}, + {"/data/year=2026/notes.txt", 1} + ]) + + assert {:ok, paths} = FileSystem.glob(fs, "/data/**/*.parquet") + + assert paths == [ + "/data/year=2025/part-0.parquet", + "/data/year=2026/part-0.parquet" + ] + + assert {:ok, [only]} = FileSystem.glob(fs, "/data/year=2026/*.parquet") + assert only == "/data/year=2026/part-0.parquet" + end + + test "put_file/3 and missing path errors" do + fs = Memory.new() + assert {:ok, fs} = Memory.put_file(fs, "relative/a.parquet", size: 9) + assert FileSystem.exists?(fs, "/relative/a.parquet") + + assert {:error, msg} = FileSystem.list(fs, "/nope") + assert msg =~ "does not exist" + + assert {:error, _} = Memory.new([{:bad, 1}]) + end + end + + describe "match_glob?" do + test "segment and recursive wildcards" do + assert FileSystem.match_glob?("/a/b/c.parquet", "/a/**/*.parquet") + assert FileSystem.match_glob?("/a/c.parquet", "/a/**/*.parquet") + refute FileSystem.match_glob?("/a/b/c.txt", "/a/**/*.parquet") + assert FileSystem.match_glob?("/a/foo.parquet", "/a/*.parquet") + refute FileSystem.match_glob?("/a/b/foo.parquet", "/a/*.parquet") + end + end +end From 399b6ff638032b4376d4b791fc2bde176ec56485 Mon Sep 17 00:00:00 2001 From: thanos Date: Sat, 12 Sep 2026 23:01:44 -0400 Subject: [PATCH 06/11] M4: add Dataset discovery with Hive partitions and Fragment metadata. Open directories, globs, or path lists via FileSystem, parse typed Hive keys from paths, and resolve schema from the first fragment footer. --- lib/ex_arrow/dataset.ex | 433 ++++++++++++++++++++++ lib/ex_arrow/dataset/fragment.ex | 34 ++ lib/ex_arrow/dataset/hive.ex | 204 ++++++++++ test/ex_arrow/compute/expression_test.exs | 70 ++++ test/ex_arrow/dataset/hive_test.exs | 124 +++++++ test/ex_arrow/dataset_test.exs | 251 +++++++++++++ test/ex_arrow/file_system_test.exs | 36 ++ test/ex_arrow/gen_stage_test.exs | 8 +- test/ex_arrow/telemetry_test.exs | 25 +- 9 files changed, 1177 insertions(+), 8 deletions(-) create mode 100644 lib/ex_arrow/dataset.ex create mode 100644 lib/ex_arrow/dataset/fragment.ex create mode 100644 lib/ex_arrow/dataset/hive.ex create mode 100644 test/ex_arrow/dataset/hive_test.exs create mode 100644 test/ex_arrow/dataset_test.exs diff --git a/lib/ex_arrow/dataset.ex b/lib/ex_arrow/dataset.ex new file mode 100644 index 0000000..26428ae --- /dev/null +++ b/lib/ex_arrow/dataset.ex @@ -0,0 +1,433 @@ +defmodule ExArrow.Dataset do + @moduledoc """ + Dataset discovery over Parquet (and IPC) files. + + A Dataset is a discovered set of fragments — usually files under a directory, + optionally with Hive partition values parsed from the path. Scanning is + handled by `ExArrow.Scanner` (v0.9.0 M5); this module only discovers and + describes fragments. + + ## Example + + {:ok, dataset} = + ExArrow.Dataset.open("/data/events", + format: :parquet, + partitioning: {:hive, schema: [{"year", :int32}, {"month", :int32}]} + ) + + ExArrow.Dataset.fragments(dataset) + ExArrow.Dataset.schema(dataset) + + ## Sources + + `open/2` accepts: + + - a directory path (recursive discovery) + - a single file path + - a glob pattern (`*` / `**`) + - an explicit list of file paths + + ## Options + + * `:format` — `:parquet` (default) or `:ipc` + * `:partitioning` — `:none` (default) or `{:hive, schema: [{name, type}, ...]}` + * `:filesystem` — `ExArrow.FileSystem` handle (default `FileSystem.Local.new()`) + * `:ignore_hidden` — skip `.` / `_`-prefixed path components (default `true`) + * `:schema` — optional `ExArrow.Schema` to skip footer schema resolution + * `:root` — dataset root for Hive relative paths (inferred when omitted) + """ + + alias ExArrow.Dataset.Fragment + alias ExArrow.Dataset.Hive + alias ExArrow.FileSystem + alias ExArrow.FileSystem.Local + alias ExArrow.IPC + alias ExArrow.Parquet + alias ExArrow.Schema + alias ExArrow.Stream + + @enforce_keys [:format, :partitioning, :filesystem, :ignore_hidden, :root, :fragments, :schema] + defstruct [:format, :partitioning, :filesystem, :ignore_hidden, :root, :fragments, :schema] + + @typedoc "Arrow-ish type atom used when coercing Hive `key=value` path segments." + @type partition_type :: + :int8 + | :int16 + | :int32 + | :int64 + | :uint8 + | :uint16 + | :uint32 + | :uint64 + | :float32 + | :float64 + | :utf8 + | :boolean + | :date32 + + @typedoc "Ordered list of `{name, type}` pairs for `{:hive, schema: ...}`." + @type partition_schema :: [{String.t(), partition_type()}] + + @type partitioning :: :none | {:hive, partition_schema()} + @type format :: :parquet | :ipc + + @type t :: %__MODULE__{ + format: format(), + partitioning: partitioning(), + filesystem: FileSystem.t(), + ignore_hidden: boolean(), + root: String.t(), + fragments: [Fragment.t()], + schema: Schema.t() + } + + @allowed_opts [:format, :partitioning, :filesystem, :ignore_hidden, :schema, :root] + + @doc """ + Discover fragments for `source` and resolve the dataset schema. + """ + @spec open(String.t() | [String.t()], keyword()) :: {:ok, t()} | {:error, String.t()} + def open(source, opts \\ []) + + def open(source, opts) when (is_binary(source) or is_list(source)) and is_list(opts) do + with :ok <- validate_opts_keys(opts), + {:ok, format} <- fetch_format(opts), + {:ok, partitioning} <- fetch_partitioning(opts), + {:ok, filesystem} <- fetch_filesystem(opts), + {:ok, ignore_hidden} <- fetch_ignore_hidden(opts), + {:ok, sized_paths, root} <- + discover_paths(source, filesystem, format, ignore_hidden, opts), + root = finalize_root(root, filesystem, sized_paths), + {:ok, fragments} <- build_fragments(sized_paths, root, format, partitioning), + {:ok, schema} <- resolve_schema(fragments, format, opts) do + {:ok, + %__MODULE__{ + format: format, + partitioning: partitioning, + filesystem: filesystem, + ignore_hidden: ignore_hidden, + root: root, + fragments: fragments, + schema: schema + }} + end + end + + def open(_source, opts) when not is_list(opts), do: {:error, "opts must be a keyword list"} + def open(_source, _opts), do: {:error, "source must be a path string or a list of paths"} + + @doc """ + Return discovered fragments in path-sorted order. + """ + @spec fragments(t()) :: [Fragment.t()] + def fragments(%__MODULE__{fragments: fragments}), do: fragments + + @doc """ + Return the dataset schema resolved at open time (footer / IPC metadata only). + """ + @spec schema(t()) :: Schema.t() + def schema(%__MODULE__{schema: schema}), do: schema + + # --- options -------------------------------------------------------------- + + defp validate_opts_keys(opts) do + if Keyword.keyword?(opts) do + bad = Enum.reject(Keyword.keys(opts), &(&1 in @allowed_opts)) + + if bad == [] do + :ok + else + {:error, "unknown option(s): #{inspect(bad)}"} + end + else + {:error, "opts must be a keyword list"} + end + end + + defp fetch_format(opts) do + case Keyword.get(opts, :format, :parquet) do + format when format in [:parquet, :ipc] -> {:ok, format} + other -> {:error, "format must be :parquet or :ipc, got #{inspect(other)}"} + end + end + + defp fetch_partitioning(opts) do + case Keyword.get(opts, :partitioning, :none) do + :none -> + {:ok, :none} + + {:hive, schema: schema} -> + with {:ok, schema} <- Hive.validate_schema(schema) do + if schema == [] do + {:error, "hive partition schema must not be empty"} + else + {:ok, {:hive, schema}} + end + end + + {:hive, kw} when is_list(kw) -> + case Keyword.fetch(kw, :schema) do + {:ok, schema} -> fetch_partitioning(partitioning: {:hive, schema: schema}) + :error -> {:error, "hive partitioning requires schema: [{name, type}, ...]"} + end + + other -> + {:error, "partitioning must be :none or {:hive, schema: ...}, got #{inspect(other)}"} + end + end + + defp fetch_filesystem(opts) do + case Keyword.get(opts, :filesystem) do + nil -> {:ok, Local.new()} + %_{} = fs -> {:ok, fs} + other -> {:error, "filesystem must be a FileSystem struct, got #{inspect(other)}"} + end + end + + defp fetch_ignore_hidden(opts) do + case Keyword.get(opts, :ignore_hidden, true) do + bool when is_boolean(bool) -> {:ok, bool} + other -> {:error, "ignore_hidden must be a boolean, got #{inspect(other)}"} + end + end + + # --- discovery ------------------------------------------------------------ + + defp discover_paths(paths, filesystem, format, ignore_hidden, opts) when is_list(paths) do + with :ok <- validate_path_list(paths), + {:ok, root} <- resolve_root(opts, paths), + {:ok, sized} <- attach_sizes(paths, filesystem) do + filtered = + sized + |> Enum.filter(fn {path, _} -> format_match?(path, format) end) + |> Enum.reject(fn {path, _} -> + ignore_hidden and FileSystem.path_has_hidden_component?(path) + end) + |> Enum.sort_by(&elem(&1, 0)) + + if filtered == [] do + {:error, "no #{format} files found in path list"} + else + {:ok, filtered, root} + end + end + end + + defp discover_paths(path, filesystem, format, ignore_hidden, opts) when is_binary(path) do + if glob_pattern?(path) do + discover_glob(path, filesystem, format, ignore_hidden, opts) + else + discover_single_source(path, filesystem, format, ignore_hidden, opts) + end + end + + defp discover_glob(pattern, filesystem, format, ignore_hidden, opts) do + with {:ok, root} <- resolve_root(opts, glob_root(pattern)), + {:ok, paths} <- FileSystem.glob(filesystem, pattern, ignore_hidden: ignore_hidden), + {:ok, sized} <- attach_sizes(Enum.filter(paths, &format_match?(&1, format)), filesystem) do + sized = Enum.sort_by(sized, &elem(&1, 0)) + + if sized == [] do + {:error, "no #{format} files matching #{pattern}"} + else + {:ok, sized, root} + end + end + end + + defp discover_single_source(path, filesystem, format, ignore_hidden, opts) do + if FileSystem.exists?(filesystem, path) do + with {:ok, root} <- resolve_root(opts, inferred_root(path, format)), + {:ok, entries} <- + FileSystem.list(filesystem, path, recursive: true, ignore_hidden: ignore_hidden) do + sized = + entries + |> Enum.filter(&(&1.type == :file)) + |> Enum.filter(&format_match?(&1.path, format)) + |> Enum.map(&{&1.path, &1.size}) + |> Enum.sort_by(&elem(&1, 0)) + + if sized == [] do + {:error, "no #{format} files under #{path}"} + else + {:ok, sized, root} + end + end + else + {:error, "path does not exist: #{path}"} + end + end + + defp inferred_root(path, format) do + if format_match?(path, format), do: Path.dirname(path), else: path + end + + defp validate_path_list([]), do: {:error, "path list must not be empty"} + + defp validate_path_list(paths) do + if Enum.all?(paths, &is_binary/1) do + :ok + else + {:error, "path list entries must be strings"} + end + end + + defp attach_sizes(paths, filesystem) do + Enum.reduce_while(paths, {:ok, []}, fn path, {:ok, acc} -> + case lookup_size(filesystem, path) do + {:ok, size} -> {:cont, {:ok, acc ++ [{path, size}]}} + {:error, _} = err -> {:halt, err} + end + end) + end + + defp lookup_size(filesystem, path) do + case FileSystem.list(filesystem, path, recursive: false) do + {:ok, [%{type: :file, size: size}]} -> {:ok, size} + {:ok, []} -> {:error, "path does not exist: #{path}"} + {:ok, _} -> {:error, "expected a file at #{path}"} + {:error, _} = err -> err + end + end + + defp resolve_root(opts, inferred) when is_binary(inferred) do + case Keyword.fetch(opts, :root) do + {:ok, root} when is_binary(root) -> {:ok, normalize_root(root)} + {:ok, other} -> {:error, "root must be a string, got #{inspect(other)}"} + :error -> {:ok, normalize_root(inferred)} + end + end + + defp resolve_root(opts, paths) when is_list(paths) do + case Keyword.fetch(opts, :root) do + {:ok, root} when is_binary(root) -> + {:ok, normalize_root(root)} + + {:ok, other} -> + {:error, "root must be a string, got #{inspect(other)}"} + + :error -> + dirnames = Enum.map(paths, &Path.dirname(normalize_root(&1))) + + case Enum.uniq(dirnames) do + [only] -> {:ok, only} + _ -> {:error, "cannot infer dataset root from path list; pass root:"} + end + end + end + + defp normalize_root(root) do + root = + root + |> String.replace("\\", "/") + |> String.replace(~r/\/+/, "/") + + if root in ["", "/"] do + "/" + else + String.trim_trailing(root, "/") + end + end + + defp glob_pattern?(path), do: String.contains?(path, ["*", "?"]) + + defp glob_root(pattern) do + pattern + |> Path.split() + |> Enum.take_while(&(not String.contains?(&1, ["*", "?"]))) + |> case do + [] -> "/" + parts -> Path.join(parts) + end + end + + defp format_match?(path, :parquet), do: String.ends_with?(path, ".parquet") + defp format_match?(path, :ipc), do: String.ends_with?(path, [".arrow", ".ipc"]) + + # Local discovery expands paths; align the Hive root to that absolute form. + defp finalize_root(root, %Local{}, [{path, _} | _]) do + expanded = normalize_root(Path.expand(root)) + + if path == expanded or String.starts_with?(path, expanded <> "/") do + expanded + else + normalize_root(root) + end + end + + defp finalize_root(root, _filesystem, _sized), do: normalize_root(root) + + # --- fragments ------------------------------------------------------------ + + defp build_fragments(sized_paths, _root, format, :none) do + fragments = + Enum.map(sized_paths, fn {path, size} -> + %Fragment{path: path, format: format, partition_values: %{}, size: size} + end) + + {:ok, fragments} + end + + defp build_fragments(sized_paths, root, format, {:hive, schema}) do + Enum.reduce_while(sized_paths, {:ok, []}, fn {path, size}, {:ok, acc} -> + case Hive.parse_path(path, root, schema) do + {:ok, values} -> + frag = %Fragment{ + path: path, + format: format, + partition_values: values, + size: size + } + + {:cont, {:ok, acc ++ [frag]}} + + {:error, _} = err -> + {:halt, err} + end + end) + end + + # --- schema --------------------------------------------------------------- + + defp resolve_schema(fragments, format, opts) do + case Keyword.fetch(opts, :schema) do + {:ok, %Schema{} = schema} -> + {:ok, schema} + + {:ok, other} -> + {:error, "schema option must be an ExArrow.Schema, got #{inspect(other)}"} + + :error -> + case fragments do + [%Fragment{path: path} | _] -> read_schema(path, format) + [] -> {:error, "cannot resolve schema: no fragments"} + end + end + end + + defp read_schema(path, :parquet) do + # Opening the reader parses the footer only; we never call Stream.next/1. + case Parquet.Reader.from_file(path) do + {:ok, stream} -> + result = Stream.schema(stream) + _ = Stream.close(stream) + result + + {:error, msg} -> + {:error, "cannot read schema from #{path}: #{msg}"} + end + end + + defp read_schema(path, :ipc) do + case IPC.File.from_file(path) do + {:ok, file} -> + case IPC.File.schema(file) do + {:ok, schema} -> {:ok, schema} + {:error, msg} -> {:error, "cannot read schema from #{path}: #{msg}"} + end + + {:error, msg} -> + {:error, "cannot read schema from #{path}: #{msg}"} + end + end +end diff --git a/lib/ex_arrow/dataset/fragment.ex b/lib/ex_arrow/dataset/fragment.ex new file mode 100644 index 0000000..b11549b --- /dev/null +++ b/lib/ex_arrow/dataset/fragment.ex @@ -0,0 +1,34 @@ +defmodule ExArrow.Dataset.Fragment do + @moduledoc """ + One readable unit in a Dataset: a file path plus optional Hive partition + values discovered from the path. + """ + + alias ExArrow.Parquet.Metadata + + @enforce_keys [:path, :format, :partition_values, :size] + defstruct [:path, :format, :partition_values, :size] + + @type format :: :parquet | :ipc + + @type t :: %__MODULE__{ + path: String.t(), + format: format(), + partition_values: %{optional(String.t()) => term()}, + size: non_neg_integer() + } + + @doc """ + Read Parquet footer metadata for this fragment (no data pages). + + Only supported for `:parquet` fragments with an OS-readable path. + """ + @spec metadata(t()) :: {:ok, Metadata.t()} | {:error, String.t()} + def metadata(%__MODULE__{format: :parquet, path: path}) when is_binary(path) do + Metadata.from_file(path) + end + + def metadata(%__MODULE__{format: format}) do + {:error, "Fragment.metadata/1 requires format :parquet, got #{inspect(format)}"} + end +end diff --git a/lib/ex_arrow/dataset/hive.ex b/lib/ex_arrow/dataset/hive.ex new file mode 100644 index 0000000..4d0516c --- /dev/null +++ b/lib/ex_arrow/dataset/hive.ex @@ -0,0 +1,204 @@ +defmodule ExArrow.Dataset.Hive do + @moduledoc false + + # Parse Hive-style `key=value` path segments into typed partition values. + + alias ExArrow.Dataset + + @type partition_schema :: Dataset.partition_schema() + + @spec parse_path(String.t(), String.t(), partition_schema()) :: + {:ok, %{optional(String.t()) => term()}} | {:error, String.t()} + def parse_path(file_path, root, schema) when is_binary(file_path) and is_binary(root) do + with {:ok, schema} <- validate_schema(schema), + {:ok, relative} <- relative_path(file_path, root), + {:ok, pairs} <- extract_pairs(relative), + {:ok, values} <- coerce_pairs(pairs, schema) do + {:ok, values} + end + end + + @spec validate_schema(term()) :: {:ok, partition_schema()} | {:error, String.t()} + def validate_schema(schema) when is_list(schema) do + Enum.reduce_while(schema, {:ok, []}, fn entry, {:ok, acc} -> + case normalize_entry(entry) do + {:ok, {name, type}} -> + if supported_type?(type) do + {:cont, {:ok, acc ++ [{name, type}]}} + else + {:halt, {:error, "unsupported partition type #{inspect(type)} for #{inspect(name)}"}} + end + + {:error, _} = err -> + {:halt, err} + end + end) + end + + def validate_schema(other), + do: {:error, "partition schema must be a list of {name, type} pairs, got: #{inspect(other)}"} + + defp normalize_entry({name, type}) when is_binary(name), do: {:ok, {name, type}} + defp normalize_entry({name, type}) when is_atom(name), do: {:ok, {Atom.to_string(name), type}} + + defp normalize_entry(other), + do: {:error, "partition schema entries must be {name, type} pairs, got: #{inspect(other)}"} + + defp supported_type?(t) + when t in [ + :int8, + :int16, + :int32, + :int64, + :uint8, + :uint16, + :uint32, + :uint64, + :float32, + :float64, + :utf8, + :boolean, + :date32 + ], + do: true + + defp supported_type?(_), do: false + + defp relative_path(file_path, root) do + file = normalize(file_path) + root = normalize(root) + + cond do + file == root -> + {:ok, Path.basename(file)} + + String.starts_with?(file, root <> "/") -> + {:ok, String.replace_prefix(file, root <> "/", "")} + + true -> + {:error, "path #{inspect(file_path)} is not under dataset root #{inspect(root)}"} + end + end + + defp normalize(path) do + path + |> String.replace("\\", "/") + |> String.replace(~r/\/+/, "/") + |> String.trim_trailing("/") + end + + defp extract_pairs(relative) do + # Drop the file basename; only directory segments can be key=value. + dirs = + relative + |> Path.dirname() + |> Path.split() + |> Enum.reject(&(&1 in [".", "/"])) + + Enum.reduce_while(dirs, {:ok, []}, fn segment, {:ok, acc} -> + case String.split(segment, "=", parts: 2) do + [key, value] when key != "" -> + {:cont, {:ok, acc ++ [{key, URI.decode(value)}]}} + + [_alone] -> + # Non-partition directory segment (e.g. intermediate folder). + {:cont, {:ok, acc}} + + _ -> + {:halt, {:error, "unparseable hive partition segment: #{inspect(segment)}"}} + end + end) + end + + defp coerce_pairs(pairs, schema) do + by_key = Map.new(pairs) + + case Enum.reduce_while(schema, {:ok, %{}}, fn {name, type}, {:ok, acc} -> + case Map.fetch(by_key, name) do + :error -> + {:halt, {:error, "missing hive partition key #{inspect(name)}"}} + + {:ok, raw} -> + case coerce(raw, type, name) do + {:ok, value} -> {:cont, {:ok, Map.put(acc, name, value)}} + {:error, _} = err -> {:halt, err} + end + end + end) do + {:ok, values} -> + unknown = + by_key + |> Map.keys() + |> Enum.reject(fn k -> Enum.any?(schema, fn {n, _} -> n == k end) end) + + if unknown == [] do + {:ok, values} + else + {:error, "unexpected hive partition key(s): #{inspect(unknown)}"} + end + + {:error, _} = err -> + err + end + end + + defp coerce(raw, :utf8, _name) when is_binary(raw) do + if String.valid?(raw), do: {:ok, raw}, else: {:error, "invalid UTF-8 partition value"} + end + + defp coerce(raw, :boolean, name) do + case String.downcase(raw) do + "true" -> {:ok, true} + "false" -> {:ok, false} + "1" -> {:ok, true} + "0" -> {:ok, false} + _ -> {:error, "cannot parse boolean partition #{name}=#{inspect(raw)}"} + end + end + + defp coerce(raw, :date32, name) do + case Date.from_iso8601(raw) do + {:ok, date} -> + {:ok, date} + + {:error, _} -> + case Integer.parse(raw) do + {i, ""} -> coerce_int(i, :int32, name) + _ -> {:error, "cannot parse date32 partition #{name}=#{inspect(raw)}"} + end + end + end + + defp coerce(raw, type, name) + when type in [:float32, :float64] do + case Float.parse(raw) do + {f, ""} -> {:ok, f} + _ -> {:error, "cannot parse float partition #{name}=#{inspect(raw)}"} + end + end + + defp coerce(raw, type, name) + when type in [:int8, :int16, :int32, :int64, :uint8, :uint16, :uint32, :uint64] do + case Integer.parse(raw) do + {i, ""} -> coerce_int(i, type, name) + _ -> {:error, "cannot parse integer partition #{name}=#{inspect(raw)}"} + end + end + + defp coerce_int(i, :int8, name), do: in_range(i, -128, 127, name, :int8) + defp coerce_int(i, :int16, name), do: in_range(i, -32_768, 32_767, name, :int16) + defp coerce_int(i, :int32, name), do: in_range(i, -2_147_483_648, 2_147_483_647, name, :int32) + defp coerce_int(i, :int64, _name), do: {:ok, i} + defp coerce_int(i, :uint8, name), do: in_range(i, 0, 255, name, :uint8) + defp coerce_int(i, :uint16, name), do: in_range(i, 0, 65_535, name, :uint16) + defp coerce_int(i, :uint32, name), do: in_range(i, 0, 4_294_967_295, name, :uint32) + + defp coerce_int(i, :uint64, name) do + if i >= 0, do: {:ok, i}, else: {:error, "partition #{name}=#{i} out of range for uint64"} + end + + defp in_range(i, min, max, _name, _type) when i >= min and i <= max, do: {:ok, i} + + defp in_range(i, _min, _max, name, type), + do: {:error, "partition #{name}=#{i} out of range for #{type}"} +end diff --git a/test/ex_arrow/compute/expression_test.exs b/test/ex_arrow/compute/expression_test.exs index 64f3481..cb94fd3 100644 --- a/test/ex_arrow/compute/expression_test.exs +++ b/test/ex_arrow/compute/expression_test.exs @@ -147,4 +147,74 @@ defmodule ExArrow.Compute.ExpressionTest do assert msg =~ "residual" end end + + describe "coverage — types, predicates, encode" do + test "expression?/1, DateTime scalar, and unsupported scalar" do + assert E.expression?(E.field("x")) + refute E.expression?(:nope) + + dt = DateTime.from_naive!(~N[2026-01-01 00:00:00], "Etc/UTC") + assert to_string(E.scalar(dt)) =~ "2026-01-01" + + assert_raise ArgumentError, fn -> E.scalar(%{not: :supported}) end + end + + test "validate covers more column types and compositions" do + schema = + schema_for([ + {"i8", :s8, [1]}, + {"i16", :s16, [1]}, + {"u8", :u8, [1]}, + {"u16", :u16, [1]}, + {"u32", :u32, [1]}, + {"f32", :f32, [1.0]}, + {"a", :s64, [1]}, + {"b", :s64, [2]}, + {"day", :date32, [0]}, + {"name", :utf8, ["x"]}, + {"flag", :bool, [true]} + ]) + + assert {:ok, _} = E.validate(E.eq(E.field("i8"), E.scalar(1)), schema) + assert {:ok, _} = E.validate(E.eq(E.field("i16"), E.scalar(1)), schema) + assert {:ok, _} = E.validate(E.eq(E.field("u8"), E.scalar(1)), schema) + assert {:ok, _} = E.validate(E.eq(E.field("u16"), E.scalar(1)), schema) + assert {:ok, _} = E.validate(E.eq(E.field("u32"), E.scalar(1)), schema) + assert {:ok, _} = E.validate(E.eq(E.field("f32"), E.scalar(1.5)), schema) + assert {:ok, _} = E.validate(E.eq(E.field("day"), E.scalar(~D[1970-01-01])), schema) + assert {:ok, _} = E.validate(E.eq(E.field("a"), E.field("b")), schema) + + assert {:ok, _} = + E.validate( + E.and_(E.eq(E.field("flag"), E.scalar(true)), E.gt(E.field("a"), E.scalar(0))), + schema + ) + + assert {:ok, _} = E.validate(E.not_(E.eq(E.field("flag"), E.scalar(false))), schema) + + assert {:error, msg} = E.validate(E.eq(E.scalar(1), E.scalar(2)), schema) + assert msg =~ "field" + + assert {:error, msg} = E.validate(E.eq(E.field("a"), E.field("name")), schema) + assert msg =~ "cannot compare" + + dt = DateTime.from_naive!(~N[2026-06-01 12:00:00], "Etc/UTC") + + assert {:ok, _} = + E.validate( + E.gte(E.field("day"), E.scalar(dt)), + schema_for([{"day", :timestamp_micros, [0]}]) + ) + end + + test "encode_for_nif covers temporal scalars" do + dt = DateTime.from_naive!(~N[2026-01-01 00:00:00], "Etc/UTC") + + assert {:call, :gt, [_, {:scalar, {:timestamp_micros, _}}]} = + E.encode_for_nif(E.gt(E.field("ts"), E.scalar(dt))) + + assert {:call, :gte, [_, {:scalar, {:date32, _}}]} = + E.encode_for_nif(E.gte(E.field("d"), E.scalar(~D[2026-01-01]))) + end + end end diff --git a/test/ex_arrow/dataset/hive_test.exs b/test/ex_arrow/dataset/hive_test.exs new file mode 100644 index 0000000..985e948 --- /dev/null +++ b/test/ex_arrow/dataset/hive_test.exs @@ -0,0 +1,124 @@ +defmodule ExArrow.Dataset.HiveTest do + use ExUnit.Case, async: true + + alias ExArrow.Dataset.Hive + + test "parses typed hive segments with URL decoding" do + assert {:ok, values} = + Hive.parse_path( + "/data/year=2026/month=01/name=hello%20world/part-0.parquet", + "/data", + [{"year", :int32}, {"month", :int32}, {"name", :utf8}] + ) + + assert values == %{"year" => 2026, "month" => 1, "name" => "hello world"} + end + + test "parses date32 ISO values and booleans" do + assert {:ok, values} = + Hive.parse_path( + "/data/day=2026-01-15/ok=true/f.parquet", + "/data", + [{"day", :date32}, {"ok", :boolean}] + ) + + assert values["day"] == ~D[2026-01-15] + assert values["ok"] == true + end + + test "errors on missing key, unexpected key, and out-of-range int32" do + assert {:error, msg} = + Hive.parse_path("/data/month=1/f.parquet", "/data", [{"year", :int32}]) + + assert msg =~ "missing" + + assert {:error, msg} = + Hive.parse_path( + "/data/year=1/extra=2/f.parquet", + "/data", + [{"year", :int32}] + ) + + assert msg =~ "unexpected" + + assert {:error, msg} = + Hive.parse_path( + "/data/year=2147483648/f.parquet", + "/data", + [{"year", :int32}] + ) + + assert msg =~ "out of range" + end + + test "allows non-partition directory segments" do + assert {:ok, values} = + Hive.parse_path( + "/data/staging/year=2026/part.parquet", + "/data", + [{"year", :int32}] + ) + + assert values == %{"year" => 2026} + end + + test "covers alternate types, atom keys, and error branches" do + assert {:ok, %{"x" => 1}} = + Hive.parse_path("/data/x=1/f.parquet", "/data", [{:x, :int8}]) + + assert {:ok, %{"x" => 1}} = + Hive.parse_path("/data/x=1/f.parquet", "/data", [{"x", :int16}]) + + assert {:ok, %{"x" => 1}} = + Hive.parse_path("/data/x=1/f.parquet", "/data", [{"x", :int64}]) + + assert {:ok, %{"x" => 1}} = + Hive.parse_path("/data/x=1/f.parquet", "/data", [{"x", :uint8}]) + + assert {:ok, %{"x" => 1}} = + Hive.parse_path("/data/x=1/f.parquet", "/data", [{"x", :uint16}]) + + assert {:ok, %{"x" => 1}} = + Hive.parse_path("/data/x=1/f.parquet", "/data", [{"x", :uint32}]) + + assert {:ok, %{"x" => 1}} = + Hive.parse_path("/data/x=1/f.parquet", "/data", [{"x", :uint64}]) + + assert {:ok, %{"x" => 1.5}} = + Hive.parse_path("/data/x=1.5/f.parquet", "/data", [{"x", :float64}]) + + assert {:ok, %{"ok" => false}} = + Hive.parse_path("/data/ok=false/f.parquet", "/data", [{"ok", :boolean}]) + + assert {:ok, %{"ok" => true}} = + Hive.parse_path("/data/ok=1/f.parquet", "/data", [{"ok", :boolean}]) + + assert {:ok, %{"ok" => false}} = + Hive.parse_path("/data/ok=0/f.parquet", "/data", [{"ok", :boolean}]) + + assert {:ok, %{"day" => 10}} = + Hive.parse_path("/data/day=10/f.parquet", "/data", [{"day", :date32}]) + + assert {:error, _} = Hive.validate_schema(:not_a_list) + assert {:error, _} = Hive.validate_schema([{"x", :decimal128}]) + assert {:error, _} = Hive.validate_schema([:bad]) + + assert {:error, msg} = + Hive.parse_path("/data/ok=maybe/f.parquet", "/data", [{"ok", :boolean}]) + + assert msg =~ "boolean" + + assert {:error, msg} = + Hive.parse_path("/data/x=abc/f.parquet", "/data", [{"x", :float64}]) + + assert msg =~ "float" + + assert {:error, msg} = + Hive.parse_path("/other/x=1/f.parquet", "/data", [{"x", :int32}]) + + assert msg =~ "not under" + + assert {:ok, %{}} = + Hive.parse_path("/data/f.parquet", "/data/f.parquet", []) + end +end diff --git a/test/ex_arrow/dataset_test.exs b/test/ex_arrow/dataset_test.exs new file mode 100644 index 0000000..31b563a --- /dev/null +++ b/test/ex_arrow/dataset_test.exs @@ -0,0 +1,251 @@ +defmodule ExArrow.DatasetTest do + use ExUnit.Case, async: true + + alias ExArrow.Dataset + alias ExArrow.Dataset.Fragment + alias ExArrow.FileSystem.Memory + alias ExArrow.IPC + alias ExArrow.Parquet + alias ExArrow.RecordBatch + alias ExArrow.Schema + + defp sample_schema_and_batch do + assert {:ok, batch} = + RecordBatch.from_lists([ + {"id", :s64, [1, 2]}, + {"score", :f64, [0.1, 0.9]} + ]) + + {RecordBatch.schema(batch), batch} + end + + defp write_parquet!(dir, relative, batch, schema) do + path = Path.join(dir, relative) + File.mkdir_p!(Path.dirname(path)) + assert :ok = Parquet.Writer.to_file(path, schema, [batch]) + path + end + + describe "open/2 validation" do + test "rejects unknown options and bad format before IO" do + assert {:error, msg} = Dataset.open("/tmp", bogus: true) + assert msg =~ "unknown option" + + assert {:error, msg} = Dataset.open("/tmp", format: :csv) + assert msg =~ "format" + end + + test "rejects empty hive schema" do + assert {:error, msg} = Dataset.open("/tmp", partitioning: {:hive, schema: []}) + assert msg =~ "empty" + end + end + + describe "Memory filesystem discovery" do + test "lists hive fragments with exact partition values and sizes" do + assert {:ok, fs} = + Memory.new(%{ + "/data/year=2026/month=1/part-0.parquet" => 128, + "/data/year=2025/month=12/part-0.parquet" => 64, + "/data/.staging/skip.parquet" => 1, + "/data/_tmp/skip.parquet" => 1 + }) + + {schema, _batch} = sample_schema_and_batch() + + assert {:ok, dataset} = + Dataset.open("/data", + filesystem: fs, + schema: schema, + partitioning: {:hive, schema: [{"year", :int32}, {"month", :int32}]} + ) + + fragments = Dataset.fragments(dataset) + assert length(fragments) == 2 + + assert Enum.map(fragments, & &1.path) == [ + "/data/year=2025/month=12/part-0.parquet", + "/data/year=2026/month=1/part-0.parquet" + ] + + assert Enum.map(fragments, & &1.partition_values) == [ + %{"year" => 2025, "month" => 12}, + %{"year" => 2026, "month" => 1} + ] + + assert Enum.map(fragments, & &1.size) == [64, 128] + assert Dataset.schema(dataset) == schema + end + + test "glob and explicit file list" do + assert {:ok, fs} = + Memory.new(%{ + "/data/a.parquet" => 10, + "/data/b.parquet" => 20, + "/data/c.txt" => 1 + }) + + {schema, _} = sample_schema_and_batch() + + assert {:ok, by_glob} = + Dataset.open("/data/*.parquet", filesystem: fs, schema: schema) + + assert Enum.map(Dataset.fragments(by_glob), & &1.path) == [ + "/data/a.parquet", + "/data/b.parquet" + ] + + assert {:ok, by_list} = + Dataset.open(["/data/b.parquet", "/data/a.parquet"], + filesystem: fs, + schema: schema, + root: "/data" + ) + + assert Enum.map(Dataset.fragments(by_list), & &1.path) == [ + "/data/a.parquet", + "/data/b.parquet" + ] + end + end + + describe "Local filesystem" do + @tag :tmp_dir + test "opens hive directory, resolves schema from footer, Fragment.metadata/1", %{ + tmp_dir: dir + } do + {schema, batch} = sample_schema_and_batch() + root = Path.join(dir, "events") + + p1 = write_parquet!(root, "year=2026/month=01/part-0.parquet", batch, schema) + _p2 = write_parquet!(root, "year=2025/month=12/part-0.parquet", batch, schema) + File.mkdir_p!(Path.join(root, ".hidden")) + _ = write_parquet!(root, ".hidden/x.parquet", batch, schema) + + assert {:ok, dataset} = + Dataset.open(root, + format: :parquet, + partitioning: {:hive, schema: [{"year", :int32}, {"month", :int32}]} + ) + + fragments = Dataset.fragments(dataset) + assert length(fragments) == 2 + + assert Enum.map(fragments, & &1.partition_values) == [ + %{"year" => 2025, "month" => 12}, + %{"year" => 2026, "month" => 1} + ] + + resolved = Dataset.schema(dataset) + assert Schema.field_names(resolved) == ["id", "score"] + + # Footer metadata without decoding row groups via Stream.next. + frag = Enum.find(fragments, &(&1.path == Path.expand(p1))) + assert {:ok, meta} = Fragment.metadata(frag) + assert meta.num_rows == 2 + assert meta.num_row_groups >= 1 + end + + @tag :tmp_dir + test "single file and ignore_hidden: false", %{tmp_dir: dir} do + {schema, batch} = sample_schema_and_batch() + path = write_parquet!(dir, "only.parquet", batch, schema) + hidden = write_parquet!(dir, ".secret/x.parquet", batch, schema) + + assert {:ok, dataset} = Dataset.open(path) + assert [frag] = Dataset.fragments(dataset) + assert frag.path == Path.expand(path) + assert frag.partition_values == %{} + + assert {:ok, with_hidden} = Dataset.open(dir, ignore_hidden: false) + paths = Enum.map(Dataset.fragments(with_hidden), & &1.path) + assert Path.expand(path) in paths + assert Path.expand(hidden) in paths + end + + @tag :tmp_dir + test "opens IPC files when format: :ipc", %{tmp_dir: dir} do + {schema, batch} = sample_schema_and_batch() + path = Path.join(dir, "batch.arrow") + assert :ok = IPC.File.write(path, schema, [batch]) + + assert {:ok, dataset} = Dataset.open(dir, format: :ipc) + assert [frag] = Dataset.fragments(dataset) + assert frag.format == :ipc + assert Schema.field_names(Dataset.schema(dataset)) == ["id", "score"] + end + + @tag :tmp_dir + test "errors on malformed hive segment values", %{tmp_dir: dir} do + {schema, batch} = sample_schema_and_batch() + root = Path.join(dir, "bad") + _ = write_parquet!(root, "year=not-a-number/part.parquet", batch, schema) + + assert {:error, msg} = + Dataset.open(root, + partitioning: {:hive, schema: [{"year", :int32}]} + ) + + assert msg =~ "cannot parse" + end + + @tag :tmp_dir + test "covers open error paths and Fragment.metadata/1 for ipc", %{tmp_dir: dir} do + {schema, batch} = sample_schema_and_batch() + + assert {:error, msg} = Dataset.open(:not_a_path) + assert msg =~ "source" + + assert {:error, msg} = Dataset.open(dir, filesystem: :local) + assert msg =~ "filesystem" + + assert {:error, msg} = Dataset.open(dir, ignore_hidden: "yes") + assert msg =~ "ignore_hidden" + + assert {:error, msg} = Dataset.open(dir, root: 123) + assert msg =~ "root" + + assert {:error, msg} = Dataset.open([], schema: schema) + assert msg =~ "empty" + + assert {:error, msg} = Dataset.open(["/no/a.parquet"], schema: schema) + assert msg =~ ~r/does not exist|no parquet/ + + assert {:error, msg} = + Dataset.open(["/a.parquet", "/b/c.parquet"], schema: schema, root: :bad) + + assert msg =~ "root" + + assert {:error, msg} = Dataset.open(dir, partitioning: {:hive, []}) + assert msg =~ ~r/schema|partitioning/ + + empty = Path.join(dir, "empty-hive") + File.mkdir_p!(empty) + + assert {:error, msg} = + Dataset.open(empty, partitioning: {:hive, schema: [{"year", :int32}]}) + + assert msg =~ ~r/no parquet|does not exist/ + + path = Path.join(dir, "batch.arrow") + assert :ok = IPC.File.write(path, schema, [batch]) + assert {:ok, dataset} = Dataset.open(path, format: :ipc) + [frag] = Dataset.fragments(dataset) + assert {:error, msg} = Fragment.metadata(frag) + assert msg =~ "parquet" + + hive_root = Path.join(dir, "hive2") + _ = write_parquet!(hive_root, "year=2026/part.parquet", batch, schema) + + assert {:ok, ds} = + Dataset.open(hive_root, + partitioning: {:hive, [schema: [{"year", :int32}]]} + ) + + assert [%{partition_values: %{"year" => 2026}}] = Dataset.fragments(ds) + + assert {:error, msg} = Dataset.open(Path.join(dir, "missing-dir-xyz")) + assert msg =~ "does not exist" + end + end +end diff --git a/test/ex_arrow/file_system_test.exs b/test/ex_arrow/file_system_test.exs index 221175c..5c5e629 100644 --- a/test/ex_arrow/file_system_test.exs +++ b/test/ex_arrow/file_system_test.exs @@ -205,6 +205,42 @@ defmodule ExArrow.FileSystemTest do refute FileSystem.match_glob?("/a/b/c.txt", "/a/**/*.parquet") assert FileSystem.match_glob?("/a/foo.parquet", "/a/*.parquet") refute FileSystem.match_glob?("/a/b/foo.parquet", "/a/*.parquet") + assert FileSystem.match_glob?("/a/b/c", "/a/**") + assert FileSystem.match_glob?("/a/anything", "/a/*") + refute FileSystem.match_glob?("/a/b", "/a") + end + end + + describe "error paths" do + test "rejects non-keyword opts and Memory put_file guards" do + fs = Local.new() + assert {:error, msg} = FileSystem.list(fs, "/tmp", :not_kw) + assert msg =~ "keyword" + + assert {:error, msg} = FileSystem.glob(fs, "/tmp/*", :not_kw) + assert msg =~ "keyword" + + mem = Memory.new() + assert {:error, msg} = Memory.put_file(mem, :path) + assert msg =~ ~r/UTF-8|string/ + + assert {:error, msg} = Memory.put_file(mem, "/x", :not_kw) + assert msg =~ "keyword" + + assert {:error, msg} = Memory.put_file(mem, "/x", size: -1) + assert msg =~ "size" + + assert {:ok, mem} = Memory.put_file(mem, "/only.parquet", size: 1) + assert {:ok, [%{type: :file}]} = FileSystem.list(mem, "/only.parquet") + assert FileSystem.exists?(mem, "/") + end + + @tag :tmp_dir + test "lists a hidden file path as empty when ignore_hidden", %{tmp_dir: dir} do + hidden = Path.join(dir, ".secret.parquet") + File.write!(hidden, "abc") + fs = Local.new() + assert {:ok, []} = FileSystem.list(fs, hidden, ignore_hidden: true) end end end diff --git a/test/ex_arrow/gen_stage_test.exs b/test/ex_arrow/gen_stage_test.exs index f6012e1..98957c5 100644 --- a/test/ex_arrow/gen_stage_test.exs +++ b/test/ex_arrow/gen_stage_test.exs @@ -1,5 +1,7 @@ defmodule ExArrow.GenStageTest do - use ExUnit.Case, async: true + # Attaches global :telemetry handlers and drives GenStage processes — keep + # serial so mailbox/telemetry traffic from other async tests cannot interfere. + use ExUnit.Case, async: false import ExArrow.TestFixtures alias ExArrow.GenStage.ADBCProducer @@ -56,7 +58,7 @@ defmodule ExArrow.GenStageTest do end # Collect all batches emitted as separate {:batches, [...]} messages. - defp collect_all(timeout_ms \\ 1000) do + defp collect_all(timeout_ms \\ 3000) do collect_all([], timeout_ms) end @@ -120,7 +122,7 @@ defmodule ExArrow.GenStageTest do GenStage.sync_subscribe(consumer, to: producer, max_demand: 10) collect_all() - assert_received {:telem, {:parquet, :binary}}, 200 + assert_receive {:telem, {:parquet, :binary}}, 2000 :telemetry.detach({:ex_arrow_gs, telem_ref}) end diff --git a/test/ex_arrow/telemetry_test.exs b/test/ex_arrow/telemetry_test.exs index 4f0989b..475927a 100644 --- a/test/ex_arrow/telemetry_test.exs +++ b/test/ex_arrow/telemetry_test.exs @@ -1,5 +1,7 @@ defmodule ExArrow.TelemetryTest do - use ExUnit.Case, async: true + # Global :telemetry handlers — must not run in parallel with other tests that + # emit the same events (otherwise assert_received can see foreign measurements). + use ExUnit.Case, async: false alias ExArrow.RecordBatch alias ExArrow.Telemetry @@ -35,14 +37,19 @@ defmodule ExArrow.TelemetryTest do @describetag event: [:ex_arrow, :stream, :batch] test "delivers measurements and metadata to attached handlers", %{ref: ref} do + token = make_ref() + Telemetry.execute( [:ex_arrow, :stream, :batch], %{rows: 10, columns: 3, batch_count: 1}, - %{source: {:parquet, "/tmp/x.parquet"}, schema: nil} + %{source: {:parquet, "/tmp/x.parquet"}, schema: nil, test_token: token} ) if @telemetry_available do - assert_received {:telemetry, [:ex_arrow, :stream, :batch], measurements, metadata} + assert_receive {:telemetry, [:ex_arrow, :stream, :batch], measurements, + %{test_token: ^token} = metadata}, + 1000 + assert measurements[:rows] == 10 assert measurements[:columns] == 3 assert metadata[:source] == {:parquet, "/tmp/x.parquet"} @@ -56,10 +63,18 @@ defmodule ExArrow.TelemetryTest do @describetag event: [:ex_arrow, :parquet, :read] test "emits with source metadata" do - Telemetry.execute([:ex_arrow, :parquet, :read], %{}, %{source: "/data/events.parquet"}) + token = make_ref() + + Telemetry.execute([:ex_arrow, :parquet, :read], %{}, %{ + source: "/data/events.parquet", + test_token: token + }) if @telemetry_available do - assert_received {:telemetry, [:ex_arrow, :parquet, :read], _measurements, metadata} + assert_receive {:telemetry, [:ex_arrow, :parquet, :read], _measurements, + %{test_token: ^token} = metadata}, + 1000 + assert metadata[:source] == "/data/events.parquet" end end From 411c6ec7ddb9fe47f3fbfb0e9dfc8fafb36d18b7 Mon Sep 17 00:00:00 2001 From: thanos Date: Sun, 13 Sep 2026 07:49:05 -0400 Subject: [PATCH 07/11] M5: add Dataset Scanner with partition prune and Stream backend. Lazy scanner compiles Expression filters into partition pruning, Parquet pushdown, and residual Compute.filter; to_stream yields an Agent-backed :dataset stream with exact stats and scan telemetry. - closed #296 - closed #297 - closed #298 - closed #299 - closed #230 - closed #231 - closed #232 - closed #233 --- lib/ex_arrow/compute/expression.ex | 15 +- lib/ex_arrow/dataset.ex | 10 + lib/ex_arrow/scanner.ex | 629 +++++++++++++++++++++++++++++ lib/ex_arrow/scanner/compile.ex | 150 +++++++ lib/ex_arrow/scanner/partition.ex | 125 ++++++ lib/ex_arrow/stream.ex | 39 +- lib/ex_arrow/telemetry.ex | 1 + test/ex_arrow/scanner_test.exs | 350 ++++++++++++++++ 8 files changed, 1304 insertions(+), 15 deletions(-) create mode 100644 lib/ex_arrow/scanner.ex create mode 100644 lib/ex_arrow/scanner/compile.ex create mode 100644 lib/ex_arrow/scanner/partition.ex create mode 100644 test/ex_arrow/scanner_test.exs diff --git a/lib/ex_arrow/compute/expression.ex b/lib/ex_arrow/compute/expression.ex index 3b6e013..42028a3 100644 --- a/lib/ex_arrow/compute/expression.ex +++ b/lib/ex_arrow/compute/expression.ex @@ -149,17 +149,22 @@ defmodule ExArrow.Compute.Expression do Checks that field names exist and that comparisons are type-compatible with the referenced column (and the other side, when both are fields). """ - @spec validate(t(), Schema.t()) :: {:ok, t()} | {:error, String.t()} + @spec validate(t(), Schema.t() | %{optional(String.t()) => term()}) :: + {:ok, t()} | {:error, String.t()} + def validate(%__MODULE__{} = expr, %{} = fields) when not is_struct(fields) do + case validate_node(expr.node, fields) do + :ok -> {:ok, expr} + {:error, _} = err -> err + end + end + def validate(%__MODULE__{} = expr, schema) do fields = schema |> Schema.fields() |> Map.new(fn f -> {f.name, f.type} end) - case validate_node(expr.node, fields) do - :ok -> {:ok, expr} - {:error, _} = err -> err - end + validate(expr, fields) end @doc """ diff --git a/lib/ex_arrow/dataset.ex b/lib/ex_arrow/dataset.ex index 26428ae..1f18ee4 100644 --- a/lib/ex_arrow/dataset.ex +++ b/lib/ex_arrow/dataset.ex @@ -128,6 +128,16 @@ defmodule ExArrow.Dataset do @spec schema(t()) :: Schema.t() def schema(%__MODULE__{schema: schema}), do: schema + @doc """ + Build a lazy `ExArrow.Scanner` over this dataset (no IO). + + See `ExArrow.Scanner.new/2` for options (`:columns`, `:filter`, `:batch_size`). + """ + @spec scanner(t(), keyword()) :: {:ok, ExArrow.Scanner.t()} | {:error, String.t()} + def scanner(%__MODULE__{} = dataset, opts \\ []) when is_list(opts) do + ExArrow.Scanner.new(dataset, opts) + end + # --- options -------------------------------------------------------------- defp validate_opts_keys(opts) do diff --git a/lib/ex_arrow/scanner.ex b/lib/ex_arrow/scanner.ex new file mode 100644 index 0000000..acfed9e --- /dev/null +++ b/lib/ex_arrow/scanner.ex @@ -0,0 +1,629 @@ +defmodule ExArrow.Scanner do + @moduledoc """ + Lazy scan of an `ExArrow.Dataset` with projection, partition pruning, and + filter pushdown. + + Building a scanner does no IO. `to_stream/1` starts an Agent-backed + `ExArrow.Stream` (`backend: :dataset`) that opens fragments on demand. + + ## Pushdown ladder + + 1. **Partition pruning** — predicates on Hive keys are evaluated against + each fragment's `partition_values` (no file open). + 2. **Parquet filters** — remaining pushable predicates become + `Parquet.Reader` `:filters` (row-group stats). + 3. **Residual** — anything left runs through `Compute.filter/2` after decode. + Partition fields in a residual expression are bound to scalars for the + current fragment. + + ## Options + + * `:columns` — list of column names to project (Parquet pushdown / IPC project) + * `:filter` — `ExArrow.Compute.Expression`, legacy Parquet filter tuple, or `nil` + * `:batch_size` — accepted for API stability; reserved (row-group sized batches in 0.9) + + ## Example + + {:ok, dataset} = ExArrow.Dataset.open(root, partitioning: {:hive, schema: [...]}) + {:ok, scanner} = ExArrow.Dataset.scanner(dataset, + columns: ["id"], + filter: ExArrow.Compute.Expression.gte( + ExArrow.Compute.Expression.field("year"), + ExArrow.Compute.Expression.scalar(2026) + ) + ) + {:ok, stream} = ExArrow.Scanner.to_stream(scanner) + batches = Enum.to_list(stream) + ExArrow.Stream.close(stream) + """ + + alias ExArrow.Compute + alias ExArrow.Compute.Expression + alias ExArrow.Dataset + alias ExArrow.Dataset.Fragment + alias ExArrow.IPC + alias ExArrow.Parquet + alias ExArrow.Parquet.Opts, as: ParquetOpts + alias ExArrow.RecordBatch + alias ExArrow.Scanner.Compile + alias ExArrow.Scanner.Partition + alias ExArrow.Schema + alias ExArrow.Stream + alias ExArrow.Telemetry + + @enforce_keys [:dataset, :columns, :filter, :batch_size, :partition_keys] + defstruct [:dataset, :columns, :filter, :batch_size, :partition_keys] + + @type stats :: %{ + fragments_discovered: non_neg_integer(), + fragments_pruned_partition: non_neg_integer(), + fragments_selected: non_neg_integer(), + fragments_scanned: non_neg_integer(), + row_groups_selected: non_neg_integer(), + row_groups_skipped: non_neg_integer(), + rows_emitted: non_neg_integer() + } + + @type t :: %__MODULE__{ + dataset: Dataset.t(), + columns: [String.t()] | nil, + filter: Expression.t() | tuple() | nil, + batch_size: pos_integer() | nil, + partition_keys: [String.t()] + } + + @doc """ + Build a lazy scanner over `dataset`. Performs no IO. + """ + @spec new(Dataset.t(), keyword()) :: {:ok, t()} | {:error, String.t()} + def new(dataset, opts \\ []) + + def new(%Dataset{} = dataset, opts) when is_list(opts) do + with :ok <- validate_opts_keys(opts), + {:ok, columns} <- fetch_columns(opts), + {:ok, filter} <- fetch_filter(opts, dataset), + {:ok, batch_size} <- fetch_batch_size(opts) do + {:ok, + %__MODULE__{ + dataset: dataset, + columns: columns, + filter: filter, + batch_size: batch_size, + partition_keys: partition_keys(dataset) + }} + end + end + + def new(_, _), do: {:error, "scanner requires an ExArrow.Dataset"} + + @doc """ + Start scanning: partition-prune, then return an `ExArrow.Stream` with + `backend: :dataset`. + """ + @spec to_stream(t()) :: {:ok, Stream.t()} | {:error, String.t()} + def to_stream(%__MODULE__{} = scanner) do + with {:ok, {pushed, residual}} <- Compile.compile(scanner.filter, scanner.partition_keys), + {:ok, read_opts} <- build_read_opts(scanner, pushed) do + {selected, pruned} = + Partition.select_fragments(scanner.dataset.fragments, scanner.filter) + + discovered = length(scanner.dataset.fragments) + + stats = %{ + fragments_discovered: discovered, + fragments_pruned_partition: pruned, + fragments_selected: length(selected), + fragments_scanned: 0, + row_groups_selected: 0, + row_groups_skipped: 0, + rows_emitted: 0 + } + + meta = %{ + root: scanner.dataset.root, + format: scanner.dataset.format, + fragments_discovered: discovered, + fragments_pruned_partition: pruned, + fragments_selected: length(selected) + } + + Telemetry.execute([:ex_arrow, :dataset, :scan, :start], %{}, meta) + + {:ok, agent} = + Agent.start_link(fn -> + %{ + fragments: selected, + index: 0, + format: scanner.dataset.format, + columns: scanner.columns, + read_opts: read_opts, + residual: residual, + current_inner: nil, + current_path: nil, + current_fragment: nil, + schema_names: nil, + opened_paths: [], + stats: stats, + scan_meta: meta, + scan_finished: false + } + end) + + {:ok, + %Stream{ + resource: agent, + backend: :dataset, + source: {:dataset, scanner.dataset.root} + }} + end + end + + @doc """ + Scan statistics. + + Pass the scanner for partition-prune preview (no row-group / row counts yet), + or the `:dataset` stream for live / post-scan aggregates. + """ + @spec stats(t() | Stream.t()) :: stats() + def stats(%__MODULE__{} = scanner) do + {selected, pruned} = + Partition.select_fragments(scanner.dataset.fragments, scanner.filter) + + %{ + fragments_discovered: length(scanner.dataset.fragments), + fragments_pruned_partition: pruned, + fragments_selected: length(selected), + fragments_scanned: 0, + row_groups_selected: 0, + row_groups_skipped: 0, + rows_emitted: 0 + } + end + + def stats(%Stream{backend: :dataset, resource: agent}) do + Agent.get(agent, & &1.stats) + end + + def stats(_), do: raise(ArgumentError, "Scanner.stats/1 expects a Scanner or dataset Stream") + + @doc false + @spec dataset_opened_paths(Stream.t()) :: [String.t()] + def dataset_opened_paths(%Stream{backend: :dataset, resource: agent}) do + Agent.get(agent, &Enum.reverse(&1.opened_paths)) + end + + def dataset_opened_paths(_), do: [] + + # --- Stream backend callbacks (invoked from ExArrow.Stream) --------------- + + @doc false + @spec stream_schema(pid()) :: {:ok, Schema.t()} | {:error, String.t()} + def stream_schema(agent) do + case ensure_open(agent) do + {:ok, _} -> + Agent.get(agent, fn state -> + case state.current_inner do + nil -> + {:error, "dataset stream has no open fragment"} + + {:ipc_file, file, _index, _count} -> + IPC.File.schema(file) + + inner -> + Stream.schema(inner) + end + end) + + :exhausted -> + {:error, "dataset stream exhausted"} + + {:error, _} = err -> + err + end + end + + @doc false + @spec stream_next(pid()) :: + {:ok, RecordBatch.t(), String.t()} | :exhausted | {:error, String.t()} + def stream_next(agent), do: do_next(agent) + + @doc false + @spec stream_close(pid()) :: :ok + def stream_close(agent) do + finish_scan(agent) + + if Process.alive?(agent) do + Agent.stop(agent) + end + + :ok + end + + # --- private -------------------------------------------------------------- + + defp do_next(agent) do + case ensure_open(agent) do + :exhausted -> + finish_scan(agent) + :exhausted + + {:error, _} = err -> + finish_scan(agent) + err + + {:ok, _} -> + agent + |> take_inner_batch() + |> handle_inner_result(agent) + end + end + + defp take_inner_batch(agent) do + Agent.get_and_update(agent, fn state -> + path = state.current_path + + case next_inner(state.current_inner) do + :done -> + {{:advance, state}, clear_current(state)} + + {:error, msg} -> + {{:error, prefix_path(path, msg)}, state} + + {:ok, batch, inner2} -> + state = %{state | current_inner: inner2} + + case postprocess(batch, state, state.current_fragment) do + {:ok, _out, 0} -> + {{:skip, state}, state} + + {:ok, out, rows} -> + stats = %{state.stats | rows_emitted: state.stats.rows_emitted + rows} + {{:ok, out, path}, %{state | stats: stats}} + + {:error, msg} -> + {{:error, prefix_path(path, msg)}, state} + end + end + end) + end + + defp next_inner({:ipc_file, _file, index, count}) when index >= count, do: :done + + defp next_inner({:ipc_file, file, index, count}) do + case IPC.File.get_batch(file, index) do + {:ok, batch} -> {:ok, batch, {:ipc_file, file, index + 1, count}} + {:error, msg} -> {:error, msg} + end + end + + defp next_inner(inner) do + case Stream.next(inner) do + nil -> :done + {:error, msg} -> {:error, msg} + batch -> {:ok, batch, inner} + end + end + + defp handle_inner_result({:ok, batch, path}, _agent), do: {:ok, batch, path} + defp handle_inner_result({:error, _} = err, _agent), do: err + defp handle_inner_result({:skip, _state}, agent), do: do_next(agent) + + defp handle_inner_result({:advance, _state}, agent) do + case advance(agent) do + :done -> + finish_scan(agent) + :exhausted + + :ok -> + do_next(agent) + end + end + + defp clear_current(state) do + %{state | current_inner: nil, current_path: nil, current_fragment: nil} + end + + defp postprocess(batch, state, frag) do + with {:ok, batch} <- maybe_project_ipc(batch, state), + {:ok, batch} <- maybe_residual(batch, state, frag) do + {:ok, batch, RecordBatch.num_rows(batch)} + end + end + + defp maybe_project_ipc(batch, %{format: :ipc, columns: cols}) when is_list(cols) do + Compute.project(batch, cols) + end + + defp maybe_project_ipc(batch, _), do: {:ok, batch} + + defp maybe_residual(batch, %{residual: nil}, _), do: {:ok, batch} + + defp maybe_residual(batch, %{residual: residual}, %Fragment{partition_values: pv}) do + bound = Compile.bind_partitions(residual, pv) + Compute.filter(batch, bound) + end + + defp ensure_open(agent) do + Agent.get_and_update(agent, fn state -> + cond do + state.current_inner != nil -> + {{:ok, :open}, state} + + state.index >= length(state.fragments) -> + {:exhausted, state} + + true -> + frag = Enum.at(state.fragments, state.index) + open_fragment(state, frag) + end + end) + end + + defp open_fragment(state, %Fragment{path: path, format: :parquet} = frag) do + case Parquet.Reader.from_file(path, state.read_opts) do + {:error, msg} -> + {{:error, prefix_path(path, msg)}, state} + + {:ok, inner} -> + case accept_schema(state, path, inner) do + {:error, _} = err -> + {err, state} + + {:ok, state2} -> + stats = merge_parquet_stats(state2.stats, inner) + stats = %{stats | fragments_scanned: stats.fragments_scanned + 1} + + {{:ok, :open}, + %{ + state2 + | current_inner: inner, + current_path: path, + current_fragment: frag, + opened_paths: [path | state2.opened_paths], + stats: stats + }} + end + end + end + + defp open_fragment(state, %Fragment{path: path, format: :ipc} = frag) do + case IPC.File.from_file(path) do + {:error, msg} -> + {{:error, prefix_path(path, msg)}, state} + + {:ok, file} -> + case IPC.File.schema(file) do + {:error, msg} -> + {{:error, prefix_path(path, msg)}, state} + + {:ok, sch} -> + names = Schema.field_names(sch) + + names = + if is_list(state.columns) do + state.columns + else + names + end + + case accept_schema_names(state, path, names) do + {:error, _} = err -> + {err, state} + + {:ok, state2} -> + count = IPC.File.batch_count(file) + stats = %{state2.stats | fragments_scanned: state2.stats.fragments_scanned + 1} + + {{:ok, :open}, + %{ + state2 + | current_inner: {:ipc_file, file, 0, count}, + current_path: path, + current_fragment: frag, + opened_paths: [path | state2.opened_paths], + stats: stats + }} + end + end + end + end + + defp accept_schema(state, path, inner) do + case Stream.schema(inner) do + {:error, msg} -> + {:error, prefix_path(path, msg)} + + {:ok, sch} -> + accept_schema_names(state, path, Schema.field_names(sch)) + end + end + + defp accept_schema_names(state, path, names) do + cond do + is_nil(state.schema_names) -> + {:ok, %{state | schema_names: names}} + + state.schema_names == names -> + {:ok, state} + + true -> + {:error, + "schema mismatch in #{path}: expected columns #{inspect(state.schema_names)}, got #{inspect(names)}"} + end + end + + defp merge_parquet_stats(stats, inner) do + rg = Parquet.Reader.read_stats(inner) + + %{ + stats + | row_groups_selected: stats.row_groups_selected + Map.get(rg, :row_groups_selected, 0), + row_groups_skipped: stats.row_groups_skipped + Map.get(rg, :row_groups_skipped, 0) + } + rescue + _ -> stats + end + + defp advance(agent) do + Agent.get_and_update(agent, fn state -> + next_index = state.index + 1 + + if next_index >= length(state.fragments) do + {:done, + %{ + state + | index: next_index, + current_inner: nil, + current_path: nil, + current_fragment: nil + }} + else + {:ok, + %{ + state + | index: next_index, + current_inner: nil, + current_path: nil, + current_fragment: nil + }} + end + end) + end + + defp finish_scan(agent) do + if Process.alive?(agent) do + Agent.get_and_update(agent, fn state -> + if state.scan_finished do + {:ok, state} + else + Telemetry.execute( + [:ex_arrow, :dataset, :scan, :stop], + %{ + fragments_scanned: state.stats.fragments_scanned, + rows_emitted: state.stats.rows_emitted, + row_groups_skipped: state.stats.row_groups_skipped + }, + Map.merge(state.scan_meta, %{stats: state.stats}) + ) + + {:ok, %{state | scan_finished: true}} + end + end) + end + + :ok + end + + defp prefix_path(path, msg) when is_binary(msg) do + if String.contains?(msg, path), do: msg, else: "#{path}: #{msg}" + end + + defp prefix_path(path, msg), do: "#{path}: #{inspect(msg)}" + + defp build_read_opts(%__MODULE__{dataset: %{format: :parquet}} = scanner, pushed) do + opts = + [] + |> then(fn o -> + if scanner.columns, do: Keyword.put(o, :columns, scanner.columns), else: o + end) + |> then(fn o -> if pushed, do: Keyword.put(o, :filters, pushed), else: o end) + + ParquetOpts.validate_read(opts) + end + + defp build_read_opts(%__MODULE__{dataset: %{format: :ipc}}, _pushed), do: {:ok, []} + + defp partition_keys(%Dataset{partitioning: {:hive, schema}}) when is_list(schema) do + Enum.map(schema, fn {name, _type} -> name end) + end + + defp partition_keys(_), do: [] + + defp validate_opts_keys(opts) do + allowed = [:columns, :filter, :batch_size] + unknown = Keyword.keys(opts) -- allowed + + if unknown == [] do + :ok + else + {:error, "unknown option(s): #{inspect(unknown)}"} + end + end + + defp fetch_columns(opts) do + case Keyword.fetch(opts, :columns) do + :error -> + {:ok, nil} + + {:ok, nil} -> + {:ok, nil} + + {:ok, cols} when is_list(cols) -> + if cols != [] and Enum.all?(cols, &is_binary/1) do + {:ok, cols} + else + {:error, "columns must be a non-empty list of strings"} + end + + {:ok, _} -> + {:error, "columns must be a list of strings"} + end + end + + defp fetch_batch_size(opts) do + case Keyword.fetch(opts, :batch_size) do + :error -> + {:ok, nil} + + {:ok, nil} -> + {:ok, nil} + + {:ok, n} when is_integer(n) and n > 0 -> + {:ok, n} + + {:ok, _} -> + {:error, "batch_size must be a positive integer"} + end + end + + defp fetch_filter(opts, dataset) do + case Keyword.fetch(opts, :filter) do + :error -> + {:ok, nil} + + {:ok, nil} -> + {:ok, nil} + + {:ok, %Expression{} = expr} -> + validate_expression(expr, dataset) + + {:ok, tuple} when is_tuple(tuple) -> + case ParquetOpts.validate_read(filters: tuple) do + {:ok, normalised} -> {:ok, Keyword.fetch!(normalised, :filters)} + {:error, _} = err -> err + end + + {:ok, other} -> + {:error, "filter must be an Expression, legacy tuple, or nil, got #{inspect(other)}"} + end + end + + defp validate_expression(expr, dataset) do + fields = + dataset.schema + |> Schema.fields() + |> Map.new(fn f -> {f.name, f.type} end) + |> Map.merge(partition_field_types(dataset)) + + case Expression.validate(expr, fields) do + {:ok, ^expr} -> {:ok, expr} + {:error, _} = err -> err + end + end + + defp partition_field_types(%Dataset{partitioning: {:hive, schema}}) when is_list(schema) do + Map.new(schema) + end + + defp partition_field_types(_), do: %{} +end diff --git a/lib/ex_arrow/scanner/compile.ex b/lib/ex_arrow/scanner/compile.ex new file mode 100644 index 0000000..97fd6be --- /dev/null +++ b/lib/ex_arrow/scanner/compile.ex @@ -0,0 +1,150 @@ +defmodule ExArrow.Scanner.Compile do + @moduledoc false + + # Split a filter into Parquet-pushable predicates (data columns only) and a + # residual Expression. Partition keys are stripped from the pushable AST; + # residual expressions bind partition fields to scalars per fragment at scan. + + alias ExArrow.Compute.Expression + + @compare_ops [:eq, :ne, :gt, :gte, :lt, :lte] + + @spec compile(term() | nil, [String.t()]) :: + {:ok, {term() | nil, Expression.t() | nil}} | {:error, String.t()} + def compile(nil, _partition_keys), do: {:ok, {nil, nil}} + + def compile(%Expression{} = expr, partition_keys) do + keys = MapSet.new(partition_keys) + {pushed0, residual0} = Expression.to_parquet_filters(expr) + pushed = strip_pushed(pushed0, keys) + + residual = + cond do + partition_only_expr?(expr, keys) -> + nil + + not is_nil(residual0) -> + residual0 + + is_nil(pushed) and not is_nil(pushed0) -> + # Strip removed pushdown (partition-only and/or unsafe OR with partition keys). + if pushed_only_partitions?(pushed0, keys), do: nil, else: expr + + is_nil(pushed) and is_nil(pushed0) -> + expr + + true -> + nil + end + + {:ok, {pushed, residual}} + end + + def compile(tuple, partition_keys) when is_tuple(tuple) do + pushed = strip_pushed(tuple, MapSet.new(partition_keys)) + {:ok, {pushed, nil}} + end + + def compile(other, _keys), + do: {:error, "filter must be an Expression, legacy tuple, or nil, got #{inspect(other)}"} + + @spec bind_partitions(Expression.t() | nil, map()) :: Expression.t() | nil + def bind_partitions(nil, _pv), do: nil + + def bind_partitions(%Expression{node: node}, pv) do + %Expression{node: bind_node(node, pv)} + end + + defp bind_node({:field, name}, pv) do + case Map.fetch(pv, name) do + {:ok, v} -> {:scalar, v} + :error -> {:field, name} + end + end + + defp bind_node({:scalar, v}, _pv), do: {:scalar, v} + + defp bind_node({:call, op, args}, pv), + do: {:call, op, Enum.map(args, &bind_node(&1, pv))} + + defp strip_pushed(nil, _keys), do: nil + + defp strip_pushed({:and, kids}, keys) when is_list(kids) do + kids = + kids + |> Enum.map(&strip_pushed(&1, keys)) + |> Enum.reject(&is_nil/1) + + case kids do + [] -> nil + [one] -> one + many -> {:and, many} + end + end + + defp strip_pushed({:or, kids}, keys) when is_list(kids) do + # Partial OR pushdown is unsafe when a branch was partition-only. + if Enum.any?(kids, &references_partition?(&1, keys)) do + nil + else + kids = + kids + |> Enum.map(&strip_pushed(&1, keys)) + |> Enum.reject(&is_nil/1) + + case kids do + [] -> nil + [one] -> one + many -> {:or, many} + end + end + end + + defp strip_pushed({op, col, _value} = pred, keys) + when op in @compare_ops and is_binary(col) do + if MapSet.member?(keys, col), do: nil, else: pred + end + + defp strip_pushed(other, _keys), do: other + + defp references_partition?({op, col, _}, keys) + when op in @compare_ops and is_binary(col), + do: MapSet.member?(keys, col) + + defp references_partition?({:and, kids}, keys), + do: Enum.any?(kids, &references_partition?(&1, keys)) + + defp references_partition?({:or, kids}, keys), + do: Enum.any?(kids, &references_partition?(&1, keys)) + + defp references_partition?(_, _), do: false + + defp pushed_only_partitions?(nil, _), do: true + + defp pushed_only_partitions?({:and, kids}, keys), + do: Enum.all?(kids, &pushed_only_partitions?(&1, keys)) + + defp pushed_only_partitions?({:or, kids}, keys), + do: Enum.all?(kids, &pushed_only_partitions?(&1, keys)) + + defp pushed_only_partitions?({op, col, _}, keys) + when op in @compare_ops and is_binary(col), + do: MapSet.member?(keys, col) + + defp pushed_only_partitions?(_, _), do: false + + defp partition_only_expr?(%Expression{node: node}, keys), + do: partition_only_node?(node, keys) + + defp partition_only_node?({:field, name}, keys), do: MapSet.member?(keys, name) + defp partition_only_node?({:scalar, _}, _keys), do: true + + defp partition_only_node?({:call, op, args}, keys) + when op in @compare_ops or op in [:and, :or], + do: Enum.all?(args, &partition_only_node?(&1, keys)) + + defp partition_only_node?({:call, :not, [inner]}, keys), + do: partition_only_node?(inner, keys) + + defp partition_only_node?(_, _), do: false +end diff --git a/lib/ex_arrow/scanner/partition.ex b/lib/ex_arrow/scanner/partition.ex new file mode 100644 index 0000000..ea2ddc3 --- /dev/null +++ b/lib/ex_arrow/scanner/partition.ex @@ -0,0 +1,125 @@ +defmodule ExArrow.Scanner.Partition do + @moduledoc false + + # Three-valued partition pruning: :true | :false | :unknown. + # Data-column references are :unknown; only partition keys + scalars decide. + + alias ExArrow.Compute.Expression + + @compare_ops [:eq, :ne, :gt, :gte, :lt, :lte] + + @spec select_fragments([ExArrow.Dataset.Fragment.t()], term() | nil) :: + {[ExArrow.Dataset.Fragment.t()], non_neg_integer()} + def select_fragments(fragments, nil), do: {fragments, 0} + + def select_fragments(fragments, filter) do + {kept, pruned} = + Enum.reduce(fragments, {[], 0}, fn frag, {acc, pruned} -> + if may_match?(filter, frag.partition_values) do + {[frag | acc], pruned} + else + {acc, pruned + 1} + end + end) + + {Enum.reverse(kept), pruned} + end + + @spec may_match?(term(), map()) :: boolean() + def may_match?(%Expression{node: node}, partition_values) do + eval(node, partition_values) != false + end + + def may_match?(tuple, partition_values) when is_tuple(tuple) do + eval_legacy(tuple, partition_values) != false + end + + def may_match?(_, _), do: true + + # --- Expression AST ------------------------------------------------------- + + defp eval({:field, name}, pv) do + case Map.fetch(pv, name) do + {:ok, v} -> {:value, v} + :error -> :unknown + end + end + + defp eval({:scalar, v}, _pv), do: {:value, v} + + defp eval({:call, op, [left, right]}, pv) when op in @compare_ops do + case {eval(left, pv), eval(right, pv)} do + {{:value, a}, {:value, b}} -> + if compare(op, a, b), do: true, else: false + + _ -> + :unknown + end + end + + defp eval({:call, :and, [left, right]}, pv) do + case {eval(left, pv), eval(right, pv)} do + {false, _} -> false + {_, false} -> false + {true, true} -> true + _ -> :unknown + end + end + + defp eval({:call, :or, [left, right]}, pv) do + case {eval(left, pv), eval(right, pv)} do + {true, _} -> true + {_, true} -> true + {false, false} -> false + _ -> :unknown + end + end + + defp eval({:call, :not, [inner]}, pv) do + case eval(inner, pv) do + true -> false + false -> true + _ -> :unknown + end + end + + defp eval(_, _), do: :unknown + + # --- legacy Parquet filter tuples ----------------------------------------- + + defp eval_legacy({:and, kids}, pv) when is_list(kids) do + Enum.reduce_while(kids, true, fn kid, acc -> + case {acc, eval_legacy(kid, pv)} do + {_, false} -> {:halt, false} + {true, true} -> {:cont, true} + _ -> {:cont, :unknown} + end + end) + end + + defp eval_legacy({:or, kids}, pv) when is_list(kids) do + Enum.reduce_while(kids, false, fn kid, acc -> + case {acc, eval_legacy(kid, pv)} do + {_, true} -> {:halt, true} + {false, false} -> {:cont, false} + _ -> {:cont, :unknown} + end + end) + end + + defp eval_legacy({op, col, value}, pv) when op in @compare_ops and is_binary(col) do + case Map.fetch(pv, col) do + {:ok, actual} -> if compare(op, actual, value), do: true, else: false + :error -> :unknown + end + end + + defp eval_legacy(_, _), do: :unknown + + defp compare(:eq, a, b), do: a == b + defp compare(:ne, a, b), do: a != b + defp compare(:gt, a, b), do: a > b + defp compare(:gte, a, b), do: a >= b + defp compare(:lt, a, b), do: a < b + defp compare(:lte, a, b), do: a <= b +end diff --git a/lib/ex_arrow/stream.ex b/lib/ex_arrow/stream.ex index c85c287..92cb625 100644 --- a/lib/ex_arrow/stream.ex +++ b/lib/ex_arrow/stream.ex @@ -2,14 +2,16 @@ defmodule ExArrow.Stream do @moduledoc """ Opaque handle to a native Arrow record-batch stream. - Provides a unified iterator interface over four backing sources: + Provides a unified iterator interface over five backing sources: - | Backend | Created by | - |--------------|-----------------------------------------------------------------| - | `:ipc` | `ExArrow.IPC.Reader` — Arrow IPC stream or file format | - | `:parquet` | `ExArrow.Parquet.Reader` — lazy row-group Parquet reader | - | `:adbc` | `ExArrow.ADBC.Statement.execute/1` — SQL result streams | - | `:flight_sql`| `ExArrow.FlightSQL.Client.stream_query/2` — Flight SQL streams | + | Backend | Created by | + |-----------------|-----------------------------------------------------------------| + | `:ipc` | `ExArrow.IPC.Reader` — Arrow IPC stream or file format | + | `:parquet` | `ExArrow.Parquet.Reader` — lazy row-group Parquet reader | + | `:parquet_multi`| `ExArrow.Stream.from_parquet_files/2` — multi-file Parquet | + | `:dataset` | `ExArrow.Scanner.to_stream/1` — Dataset fragment scan | + | `:adbc` | `ExArrow.ADBC.Statement.execute/1` — SQL result streams | + | `:flight_sql` | `ExArrow.FlightSQL.Client.stream_query/2` — Flight SQL streams | Plain Flight `do_get` results also use the `:ipc` backend (the Flight client returns an IPC stream resource). @@ -64,7 +66,7 @@ defmodule ExArrow.Stream do @opaque t :: %__MODULE__{ resource: reference() | pid() | nil, - backend: :ipc | :adbc | :parquet | :parquet_multi | :flight_sql, + backend: :ipc | :adbc | :parquet | :parquet_multi | :dataset | :flight_sql, source: term() } defstruct [:resource, :source, backend: :ipc] @@ -222,8 +224,9 @@ defmodule ExArrow.Stream do @doc """ Release resources held by a stream. - For `:parquet_multi` streams this stops the backing `Agent` (and drops the - open Parquet handle). Other backends are GC-safe and this is a no-op. + For `:parquet_multi` and `:dataset` streams this stops the backing `Agent` + (and drops any open file handle). Other backends are GC-safe and this is a + no-op. """ @spec close(t()) :: :ok def close(%__MODULE__{resource: agent, backend: :parquet_multi}) do @@ -231,6 +234,10 @@ defmodule ExArrow.Stream do :ok end + def close(%__MODULE__{resource: agent, backend: :dataset}) do + ExArrow.Scanner.stream_close(agent) + end + def close(%__MODULE__{}), do: :ok @doc false @@ -387,6 +394,10 @@ defmodule ExArrow.Stream do end end + def schema(%__MODULE__{resource: agent, backend: :dataset}) do + ExArrow.Scanner.stream_schema(agent) + end + def schema(%__MODULE__{resource: ref, backend: :flight_sql}) do case native().flight_sql_stream_schema(ref) do {:error, msg} -> {:error, msg} @@ -462,6 +473,14 @@ defmodule ExArrow.Stream do end end + def next(%__MODULE__{resource: agent, backend: :dataset} = stream) do + case ExArrow.Scanner.stream_next(agent) do + :exhausted -> nil + {:error, _} = err -> err + {:ok, batch, path} -> emit_batch(%{stream | source: {:dataset, path}}, batch) + end + end + def next(%__MODULE__{resource: ref, backend: :flight_sql} = stream) do case native().flight_sql_stream_next(ref) do :done -> nil diff --git a/lib/ex_arrow/telemetry.ex b/lib/ex_arrow/telemetry.ex index f6ce249..e315b0f 100644 --- a/lib/ex_arrow/telemetry.ex +++ b/lib/ex_arrow/telemetry.ex @@ -27,6 +27,7 @@ defmodule ExArrow.Telemetry do | `[:ex_arrow, :flight_sql, :query]` | A Flight SQL query stream is opened | | `[:ex_arrow, :parquet, :read]` | A Parquet reader stream is opened | | `[:ex_arrow, :parquet, :write]` | Batches are written to Parquet | + | `[:ex_arrow, :dataset, :scan]` | Dataset scanner span (`:start` / `:stop`)| | `[:ex_arrow, :stream, :batch]` | A single batch is yielded from a stream | | `[:ex_arrow, :pipeline, :batch]` | A pipeline stage processes a batch | diff --git a/test/ex_arrow/scanner_test.exs b/test/ex_arrow/scanner_test.exs new file mode 100644 index 0000000..d228657 --- /dev/null +++ b/test/ex_arrow/scanner_test.exs @@ -0,0 +1,350 @@ +defmodule ExArrow.ScannerTest do + use ExUnit.Case, async: false + + alias ExArrow.Compute.Expression, as: E + alias ExArrow.Dataset + alias ExArrow.Native + alias ExArrow.Parquet + alias ExArrow.RecordBatch + alias ExArrow.Scanner + alias ExArrow.Stream + + defp s64_column(batch, name) do + ref = RecordBatch.resource_ref(batch) + {:ok, {binary, "s64", _n}} = Native.record_batch_column_buffer(ref, name) + for <>, do: v + end + + defp write_parquet!(path, schema, batches, opts \\ []) do + File.mkdir_p!(Path.dirname(path)) + assert :ok = Parquet.Writer.to_file(path, schema, batches, opts) + path + end + + defp hive_dataset!(root) do + assert {:ok, batch} = + RecordBatch.from_lists([ + {"id", :s64, [1, 2, 100, 101]}, + {"score", :f64, [0.0, 0.5, 0.9, 1.0]} + ]) + + schema = RecordBatch.schema(batch) + + _ = + write_parquet!( + Path.join(root, "year=2025/month=12/part-0.parquet"), + schema, + [batch], + row_group_size: 2 + ) + + _ = + write_parquet!( + Path.join(root, "year=2026/month=01/part-0.parquet"), + schema, + [batch], + row_group_size: 2 + ) + + assert {:ok, dataset} = + Dataset.open(root, + format: :parquet, + partitioning: {:hive, schema: [{"year", :int32}, {"month", :int32}]} + ) + + {dataset, schema} + end + + describe "scanner/2 and stats preview" do + @tag :tmp_dir + test "builds without IO and previews partition prune counts", %{tmp_dir: dir} do + root = Path.join(dir, "events") + {dataset, _schema} = hive_dataset!(root) + + filter = + E.and_( + E.gte(E.field("year"), E.scalar(2026)), + E.gt(E.field("id"), E.scalar(50)) + ) + + assert {:ok, scanner} = + Dataset.scanner(dataset, columns: ["id"], filter: filter) + + preview = Scanner.stats(scanner) + assert preview.fragments_discovered == 2 + assert preview.fragments_pruned_partition == 1 + assert preview.fragments_selected == 1 + assert preview.fragments_scanned == 0 + assert preview.row_groups_skipped == 0 + assert preview.rows_emitted == 0 + end + + test "rejects bad options" do + assert {:error, msg} = Scanner.new(:not_a_dataset, []) + assert msg =~ "Dataset" + end + end + + describe "partition prune + parquet pushdown + exact rows" do + @tag :tmp_dir + @tag :nif + test "prunes fragments and row groups with exact stats", %{tmp_dir: dir} do + root = Path.join(dir, "events") + {dataset, _schema} = hive_dataset!(root) + + filter = + E.and_( + E.gte(E.field("year"), E.scalar(2026)), + E.gt(E.field("id"), E.scalar(50)) + ) + + assert {:ok, scanner} = Dataset.scanner(dataset, columns: ["id"], filter: filter) + assert {:ok, stream} = Scanner.to_stream(scanner) + + batches = Enum.to_list(stream) + ids = Enum.flat_map(batches, &s64_column(&1, "id")) + assert ids == [100, 101] + + stats = Scanner.stats(stream) + assert stats.fragments_discovered == 2 + assert stats.fragments_pruned_partition == 1 + assert stats.fragments_selected == 1 + assert stats.fragments_scanned == 1 + assert stats.row_groups_skipped == 1 + assert stats.row_groups_selected == 1 + assert stats.rows_emitted == 2 + + assert Stream.next(stream) == nil + assert Stream.next(stream) == nil + + agent = stream.resource + assert Process.alive?(agent) + assert :ok = Stream.close(stream) + refute Process.alive?(agent) + end + + @tag :tmp_dir + @tag :nif + test "residual not_ filter binds after decode", %{tmp_dir: dir} do + root = Path.join(dir, "events") + {dataset, _schema} = hive_dataset!(root) + + filter = + E.and_( + E.eq(E.field("year"), E.scalar(2026)), + E.not_(E.eq(E.field("id"), E.scalar(100))) + ) + + assert {:ok, scanner} = Dataset.scanner(dataset, filter: filter) + assert {:ok, stream} = Scanner.to_stream(scanner) + + ids = + stream + |> Enum.to_list() + |> Enum.flat_map(&s64_column(&1, "id")) + + assert ids == [1, 2, 101] + + stats = Scanner.stats(stream) + assert stats.fragments_pruned_partition == 1 + assert stats.fragments_scanned == 1 + assert stats.rows_emitted == 3 + + Stream.close(stream) + end + + @tag :tmp_dir + @tag :nif + test "early Enum.take does not open later fragments", %{tmp_dir: dir} do + root = Path.join(dir, "events") + {dataset, _schema} = hive_dataset!(root) + + assert {:ok, scanner} = Dataset.scanner(dataset) + assert {:ok, stream} = Scanner.to_stream(scanner) + + assert [%RecordBatch{}] = Enum.take(stream, 1) + opened = Scanner.dataset_opened_paths(stream) + assert length(opened) == 1 + + Stream.close(stream) + end + end + + describe "telemetry" do + @tag :tmp_dir + @tag :nif + test "emits dataset scan span and batch source", %{tmp_dir: dir} do + root = Path.join(dir, "events") + {dataset, _schema} = hive_dataset!(root) + + parent = self() + handler_id = "scanner-telem-#{System.unique_integer([:positive])}" + + :telemetry.attach_many( + handler_id, + [ + [:ex_arrow, :dataset, :scan, :start], + [:ex_arrow, :dataset, :scan, :stop], + [:ex_arrow, :stream, :batch] + ], + fn event, measurements, metadata, _ -> + send(parent, {:telem, event, measurements, metadata}) + end, + nil + ) + + on_exit(fn -> :telemetry.detach(handler_id) end) + + assert {:ok, scanner} = Dataset.scanner(dataset, columns: ["id"]) + assert {:ok, stream} = Scanner.to_stream(scanner) + _ = Enum.to_list(stream) + Stream.close(stream) + + assert_receive {:telem, [:ex_arrow, :dataset, :scan, :start], _, meta} + assert meta.fragments_discovered == 2 + + assert_receive {:telem, [:ex_arrow, :stream, :batch], %{rows: rows}, + %{source: {:dataset, path}}} + + assert rows > 0 + assert is_binary(path) + + assert_receive {:telem, [:ex_arrow, :dataset, :scan, :stop], _, _} + end + end + + describe "errors include fragment path" do + @tag :tmp_dir + @tag :nif + test "schema mismatch names the path", %{tmp_dir: dir} do + assert {:ok, a} = RecordBatch.from_lists([{"id", :s64, [1]}]) + assert {:ok, b} = RecordBatch.from_lists([{"x", :s64, [1]}]) + sa = RecordBatch.schema(a) + sb = RecordBatch.schema(b) + + p1 = write_parquet!(Path.join(dir, "a.parquet"), sa, [a]) + _p2 = write_parquet!(Path.join(dir, "b.parquet"), sb, [b]) + + assert {:ok, dataset} = Dataset.open(dir) + assert {:ok, scanner} = Dataset.scanner(dataset) + assert {:ok, stream} = Scanner.to_stream(scanner) + + assert %RecordBatch{} = Stream.next(stream) + assert {:error, msg} = Stream.next(stream) + assert msg =~ "schema mismatch" + assert msg =~ "b.parquet" or msg =~ Path.expand(Path.join(dir, "b.parquet")) + + Stream.close(stream) + _ = p1 + end + end + + describe "Compile / Partition helpers" do + test "partition may_match? three-valued AND/OR" do + alias ExArrow.Scanner.Partition + + expr = E.and_(E.eq(E.field("year"), E.scalar(2026)), E.gt(E.field("id"), E.scalar(0))) + assert Partition.may_match?(expr, %{"year" => 2026}) + refute Partition.may_match?(expr, %{"year" => 2025}) + + or_expr = E.or_(E.eq(E.field("year"), E.scalar(2026)), E.gt(E.field("id"), E.scalar(0))) + assert Partition.may_match?(or_expr, %{"year" => 2025}) + + assert Partition.may_match?({:gte, "year", 2026}, %{"year" => 2026}) + refute Partition.may_match?({:gte, "year", 2026}, %{"year" => 2025}) + end + + test "compile strips partition keys from pushed filters" do + alias ExArrow.Scanner.Compile + + expr = + E.and_( + E.gte(E.field("year"), E.scalar(2026)), + E.gt(E.field("id"), E.scalar(50)) + ) + + assert {:ok, {pushed, residual}} = Compile.compile(expr, ["year", "month"]) + assert pushed == {:gt, "id", 50} + assert residual == nil + + or_expr = + E.or_( + E.eq(E.field("year"), E.scalar(2026)), + E.gt(E.field("id"), E.scalar(50)) + ) + + assert {:ok, {nil, %E{} = residual}} = Compile.compile(or_expr, ["year"]) + assert E.to_string(residual) =~ "or_" + end + end + + describe "validation" do + @tag :tmp_dir + test "unknown filter field errors before scan", %{tmp_dir: dir} do + root = Path.join(dir, "events") + {dataset, _} = hive_dataset!(root) + + assert {:error, msg} = + Dataset.scanner(dataset, filter: E.eq(E.field("nope"), E.scalar(1))) + + assert msg =~ "unknown field" + end + + @tag :tmp_dir + test "batch_size must be positive", %{tmp_dir: dir} do + root = Path.join(dir, "events") + {dataset, _} = hive_dataset!(root) + + assert {:error, msg} = Dataset.scanner(dataset, batch_size: 0) + assert msg =~ "batch_size" + + assert {:ok, _} = Dataset.scanner(dataset, batch_size: 1024) + end + end + + describe "legacy filters and prune-all" do + @tag :tmp_dir + @tag :nif + test "accepts legacy filter tuples and can prune every fragment", %{tmp_dir: dir} do + root = Path.join(dir, "events") + {dataset, _} = hive_dataset!(root) + + assert {:ok, scanner} = + Dataset.scanner(dataset, filter: {:and, [{:eq, "year", 1999}, {:gt, "id", 0}]}) + + preview = Scanner.stats(scanner) + assert preview.fragments_pruned_partition == 2 + assert preview.fragments_selected == 0 + + assert {:ok, stream} = Scanner.to_stream(scanner) + assert Enum.to_list(stream) == [] + assert Scanner.stats(stream).fragments_scanned == 0 + Stream.close(stream) + end + end + + describe "IPC dataset scan" do + @tag :tmp_dir + @tag :nif + test "projects columns from IPC fragments", %{tmp_dir: dir} do + assert {:ok, batch} = + RecordBatch.from_lists([ + {"id", :s64, [1, 2]}, + {"score", :f64, [0.1, 0.2]} + ]) + + schema = RecordBatch.schema(batch) + path = Path.join(dir, "batch.arrow") + assert :ok = ExArrow.IPC.File.write(path, schema, [batch]) + + assert {:ok, dataset} = Dataset.open(dir, format: :ipc) + assert {:ok, scanner} = Dataset.scanner(dataset, columns: ["id"]) + assert {:ok, stream} = Scanner.to_stream(scanner) + + [out] = Enum.to_list(stream) + assert RecordBatch.column_names(out) == ["id"] + assert s64_column(out, "id") == [1, 2] + Stream.close(stream) + end + end +end From c138284f07826c6c4cf079ee6812fc9b513d0f4a Mon Sep 17 00:00:00 2001 From: thanos Date: Sun, 13 Sep 2026 08:48:00 -0400 Subject: [PATCH 08/11] M7/M8: add Dataset fixtures, guide, Livebook, and scan bench. Check in a PyArrow hive fixture with exact Scanner prune stats, expand Expression tests, and document Dataset/Scanner via guide, README, notebook, and a labeled pushdown-ladder benchmark. - closed #304 - closed #305 - closed #306 - closed #307 - closed #308 - closed #309 - closed #310 - closed #311 - closed #312 - closed #313 - closed #314 - closed #315 - closed #316 --- README.md | 37 ++++ bench/dataset_scan_bench.exs | 105 ++++++++++ docs/benchmarks.md | 11 ++ docs/overview.md | 21 ++ guides/11_datasets.md | 128 ++++++++++++ livebook/06_datasets.livemd | 182 ++++++++++++++++++ livebook/README.md | 3 +- mix.exs | 9 + script/generate_hive_events_fixture.py | 59 ++++++ test/ex_arrow/compute/expression_test.exs | 60 +++++- test/ex_arrow/hive_events_fixture_test.exs | 87 +++++++++ test/ex_arrow/scanner_test.exs | 28 +++ test/fixtures/README.md | 9 + .../year=2025/month=12/part-0.parquet | Bin 0 -> 1041 bytes .../year=2026/month=01/part-0.parquet | Bin 0 -> 1599 bytes .../year=2026/month=02/part-0.parquet | Bin 0 -> 1041 bytes 16 files changed, 735 insertions(+), 4 deletions(-) create mode 100644 bench/dataset_scan_bench.exs create mode 100644 guides/11_datasets.md create mode 100644 livebook/06_datasets.livemd create mode 100755 script/generate_hive_events_fixture.py create mode 100644 test/ex_arrow/hive_events_fixture_test.exs create mode 100644 test/fixtures/hive_events/year=2025/month=12/part-0.parquet create mode 100644 test/fixtures/hive_events/year=2026/month=01/part-0.parquet create mode 100644 test/fixtures/hive_events/year=2026/month=02/part-0.parquet diff --git a/README.md b/README.md index bb808df..3aa835a 100644 --- a/README.md +++ b/README.md @@ -415,6 +415,7 @@ without `:telemetry`, the Flow/GenStage/Broadway modules return - [08 Arrow and GenStage](guides/08_arrow_and_genstage.md) - [09 Arrow and Broadway](guides/09_arrow_and_broadway.md) - [10 Arrow pipeline patterns](guides/10_arrow_pipeline_patterns.md) +- [11 Datasets and scanners](guides/11_datasets.md) ### New benchmarks @@ -435,6 +436,7 @@ Interactive notebooks (open in [Livebook](https://livebook.dev)): - **[03 ADBC](livebook/03_adbc.livemd)** — Database, Connection, Statement, Stream (`:adbc_package` in Livebook). - **[04 ADBC integration](livebook/04_adbc_integration.livemd)** — Connection pooling with NimblePool. - **[05 Parquet](livebook/05_parquet.livemd)** — Pushdown reads, compressed writes, multi-file directories, PyArrow side-by-side. +- **[06 Datasets](livebook/06_datasets.livemd)** — Hive Dataset open, Expression scanner, prune stats, PyArrow `dataset` side-by-side. See [livebook/README.md](livebook/README.md) for run instructions. Notebooks use Hex `~> 0.8.0` by default; opening from `livebook/` in a clone builds from source. @@ -763,6 +765,40 @@ Full guide: [docs/parquet_guide.md](docs/parquet_guide.md). Livebook: --- +## Datasets and scanners + +Discover Hive-partitioned Parquet (or IPC) trees, then scan with projection +and Expression filters. Partition pruning, Parquet row-group pushdown, and +residual `Compute.filter/2` form a pushdown ladder. + +```elixir +alias ExArrow.Compute.Expression, as: E + +{:ok, dataset} = + ExArrow.Dataset.open("/data/events", + partitioning: {:hive, schema: [{"year", :int32}, {"month", :int32}]} + ) + +filter = + E.and_( + E.gte(E.field("year"), E.scalar(2026)), + E.gt(E.field("amount"), E.scalar(0.0)) + ) + +{:ok, scanner} = + ExArrow.Dataset.scanner(dataset, columns: ["id", "amount"], filter: filter) + +{:ok, stream} = ExArrow.Scanner.to_stream(scanner) +batches = Enum.to_list(stream) +ExArrow.Stream.close(stream) +ExArrow.Scanner.stats(stream) +``` + +Full guide: [guides/11_datasets.md](guides/11_datasets.md). Livebook: +`livebook/06_datasets.livemd`. + +--- + ## Arrow compute kernels All operations run entirely in native memory. Results are new @@ -1106,6 +1142,7 @@ The CI workflow posts a PR alert comment when any scenario regresses more than - [Arrow and GenStage](guides/08_arrow_and_genstage.md) — demand-driven producers with backpressure - [Arrow and Broadway](guides/09_arrow_and_broadway.md) — ingestion pipelines with `BatchBuilder` and sinks - [Arrow Pipeline Patterns](guides/10_arrow_pipeline_patterns.md) — composable `ExArrow.Pipeline` transforms and sinks +- [Datasets and Scanners](guides/11_datasets.md) — Dataset, Fragment, Hive partitioning, Scanner pushdown ladder - [Memory model](docs/memory_model.md) — handles, copying rules, NIF scheduling - [IPC guide](docs/ipc_guide.md) — stream vs file, types, limitations - [Parquet guide](docs/parquet_guide.md) — read/write Parquet, streaming, comparison with IPC diff --git a/bench/dataset_scan_bench.exs b/bench/dataset_scan_bench.exs new file mode 100644 index 0000000..7e9b4a8 --- /dev/null +++ b/bench/dataset_scan_bench.exs @@ -0,0 +1,105 @@ +# Dataset scan pushdown ladder — timing helper for announcements. +# Usage: mix run bench/dataset_scan_bench.exs +# +# Each label states exactly what the branch measures (F-013). +# Synthetic layout: >= 8 Hive partitions, 1M+ rows total. + +alias ExArrow.Compute.Expression, as: E +alias ExArrow.Dataset +alias ExArrow.Parquet +alias ExArrow.RecordBatch +alias ExArrow.Scanner +alias ExArrow.Stream + +partitions = 8 +rows_per_part = 150_000 +total_rows = partitions * rows_per_part + +root = Path.join(System.tmp_dir!(), "ex_arrow_dataset_scan_bench") +File.rm_rf!(root) + +IO.puts("Writing #{total_rows} rows across #{partitions} hive partitions under #{root}...") + +Enum.each(0..(partitions - 1), fn p -> + year = 2020 + rem(p, 4) + month = rem(p, 12) + 1 + n = rows_per_part + + ids = for i <- 1..n, into: <<>>, do: <> + amounts = for i <- 1..n, into: <<>>, do: <> + + {:ok, batch} = + RecordBatch.from_columns(["id", "amount"], [ids, amounts], ["s64", "f64"], n) + + schema = RecordBatch.schema(batch) + path = Path.join(root, "year=#{year}/month=#{month}/part-0.parquet") + File.mkdir_p!(Path.dirname(path)) + :ok = Parquet.Writer.to_file(path, schema, [batch], row_group_size: 50_000) +end) + +{:ok, dataset} = + Dataset.open(root, partitioning: {:hive, schema: [{"year", :int32}, {"month", :int32}]}) + +measure = fn label, fun -> + {us, result} = :timer.tc(fun) + IO.puts("#{label}: #{Float.round(us / 1000, 1)} ms -> #{inspect(result)}") +end + +measure.("full scan (all fragments, no filter, all columns)", fn -> + {:ok, scanner} = Dataset.scanner(dataset) + {:ok, stream} = Scanner.to_stream(scanner) + rows = Enum.sum(Enum.map(Enum.to_list(stream), &RecordBatch.num_rows/1)) + Stream.close(stream) + rows +end) + +measure.("projection only (columns: id — no filter)", fn -> + {:ok, scanner} = Dataset.scanner(dataset, columns: ["id"]) + {:ok, stream} = Scanner.to_stream(scanner) + rows = Enum.sum(Enum.map(Enum.to_list(stream), &RecordBatch.num_rows/1)) + Stream.close(stream) + rows +end) + +year_filter = E.gte(E.field("year"), E.scalar(2022)) + +measure.("partition-pruned (year >= 2022; opens fewer fragments)", fn -> + {:ok, scanner} = Dataset.scanner(dataset, filter: year_filter) + {:ok, stream} = Scanner.to_stream(scanner) + rows = Enum.sum(Enum.map(Enum.to_list(stream), &RecordBatch.num_rows/1)) + stats = Scanner.stats(stream) + Stream.close(stream) + {rows, stats.fragments_pruned_partition, stats.fragments_scanned} +end) + +rg_filter = + E.and_( + E.gte(E.field("year"), E.scalar(2022)), + E.gt(E.field("amount"), E.scalar(140_000.0)) + ) + +measure.("row-group-pruned (partition + amount > 140000 Parquet filters)", fn -> + {:ok, scanner} = Dataset.scanner(dataset, columns: ["id"], filter: rg_filter) + {:ok, stream} = Scanner.to_stream(scanner) + rows = Enum.sum(Enum.map(Enum.to_list(stream), &RecordBatch.num_rows/1)) + stats = Scanner.stats(stream) + Stream.close(stream) + {rows, stats.row_groups_skipped, stats.row_groups_selected} +end) + +residual_filter = + E.and_( + E.gte(E.field("year"), E.scalar(2022)), + E.not_(E.eq(E.field("id"), E.scalar(1))) + ) + +measure.("expression-residual (partition prune + not_ residual after decode)", fn -> + {:ok, scanner} = Dataset.scanner(dataset, columns: ["id"], filter: residual_filter) + {:ok, stream} = Scanner.to_stream(scanner) + rows = Enum.sum(Enum.map(Enum.to_list(stream), &RecordBatch.num_rows/1)) + stats = Scanner.stats(stream) + Stream.close(stream) + {rows, stats.rows_emitted, stats.fragments_scanned} +end) + +IO.puts("done.") diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 437cfdf..0fc0d25 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -131,6 +131,17 @@ MIX_ENV=dev mix run bench/parquet_pushdown_bench.exs It prints elapsed milliseconds and the row-group selection stats after open. Not part of `bench/run_all.exs` (no HTML/JSON output). +### v0.9.0 Dataset scan ladder (`bench/dataset_scan_bench.exs`) + +Standalone `:timer.tc/1` helper over a synthetic Hive layout (8 partitions, +1.2M rows). Labels state exactly what each branch measures: full scan, +projection-only, partition-pruned, row-group-pruned, and expression-residual. +Not part of `bench/run_all.exs` (writes under `/tmp`, longer runtime). + +```bash +MIX_ENV=dev mix run bench/dataset_scan_bench.exs +``` + ## Published results Benchmark results from every push to `main` are stored in the `gh-pages` diff --git a/docs/overview.md b/docs/overview.md index e57ba67..fb62b48 100644 --- a/docs/overview.md +++ b/docs/overview.md @@ -3,6 +3,26 @@ The main overview, installation, quick start, and usage examples live in the [README on GitHub](https://github.com/thanos/ex_arrow/blob/main/README.md). +## What's changed in v0.9.0 + +v0.9.0 adds a Dataset / Scanner layer for discovering Hive-partitioned +Parquet (and IPC) trees and scanning them with analyzable Expression filters. +Partition pruning, Parquet row-group pushdown, and residual +`Compute.filter/2` form a pushdown ladder. See the +[Datasets guide](../guides/11_datasets.md) and `livebook/06_datasets.livemd`. + +New / changed: + +- **`ExArrow.Dataset`** — open a directory, file, glob, or path list; + Hive `partition_values`; schema from footer without decoding pages. +- **`ExArrow.Scanner`** — lazy `scanner/2` / `to_stream/1` with `:columns` + and `:filter` (`Expression` or legacy tuple); `stats/1` with exact prune + counts. +- **`ExArrow.Compute.Expression`** — builders, `validate/2`, + `to_parquet_filters/1` (pushable vs residual). +- **`ExArrow.FileSystem`** — Local and Memory backends for discovery. +- **`RecordBatch.from_lists/1` / `from_map/1`** — ergonomic batch construction. + ## What's changed in v0.8.0 v0.8.0 makes Parquet a first-class citizen for larger-than-memory workloads. @@ -68,6 +88,7 @@ New guides: [05 Arrow pipelines overview](05_arrow_pipelines_overview.md), | Arrow Flight SQL remote query client | [Flight SQL guide](flight_sql_guide.md) | | ADBC database connectivity | [ADBC guide](adbc_guide.md) | | Parquet read and write | [Parquet guide](parquet_guide.md) | +| Datasets and scanners | [Datasets guide](../guides/11_datasets.md) | | Compute kernels (filter, project, sort) | [Compute guide](compute_guide.md) | | C Data Interface (CDI) | [CDI guide](cdi_guide.md) | | Nx tensor bridge | [Nx guide](nx_guide.md) | diff --git a/guides/11_datasets.md b/guides/11_datasets.md new file mode 100644 index 0000000..2c0eb88 --- /dev/null +++ b/guides/11_datasets.md @@ -0,0 +1,128 @@ +# Datasets and Scanners + +ExArrow's Dataset layer is an **IO, discovery, pruning, and streaming-execution** +API, not a DataFrame API. You discover files (fragments), optionally interpret +Hive partition paths, then scan with projection and filters that push down as +far as possible before decoding row groups. + +If you need column reshaping, joins, or group-by, use Explorer (or another +DataFrame tool) **after** ExArrow has streamed the batches you care about. + +## Concepts + +| Term | Meaning in ExArrow | +|------|--------------------| +| **Dataset** | Result of discovering files under a root (or an explicit path list). Holds fragments + schema. Does not decode data pages. | +| **Fragment** | One readable unit, usually a Parquet (or IPC file) path plus optional Hive `partition_values`. | +| **Partitioning** | How path segments map to columns. 0.9 supports `:none` and `{:hive, schema: [{name, type}, ...]}`. | +| **Scanner** | Lazy plan over a Dataset: columns, filter, batch options. No IO until `to_stream/1`. | +| **Expression** | Analyzable filter AST (`ExArrow.Compute.Expression`). Not an Elixir closure. | + +## Open a Dataset + +```elixir +{:ok, dataset} = + ExArrow.Dataset.open("/data/events", + format: :parquet, + partitioning: {:hive, schema: [{"year", :int32}, {"month", :int32}]}, + ignore_hidden: true + ) + +ExArrow.Dataset.fragments(dataset) +ExArrow.Dataset.schema(dataset) +``` + +`open/2` accepts a directory, a single file, a glob (`*` / `**`), or a list of +paths. Pass `:filesystem` (`ExArrow.FileSystem.Local` or `Memory`) for tests +without touching the OS. Pass `:schema` to skip footer resolution (required +for Memory-only discovery when files are not OS-readable). + +Schema is resolved from the first fragment's Parquet footer / IPC file +metadata without consuming batches. + +## Scan + +```elixir +alias ExArrow.Compute.Expression, as: E + +filter = + E.and_( + E.gte(E.field("year"), E.scalar(2026)), + E.gt(E.field("amount"), E.scalar(0.0)) + ) + +{:ok, scanner} = + ExArrow.Dataset.scanner(dataset, + columns: ["id", "amount"], + filter: filter + ) + +{:ok, stream} = ExArrow.Scanner.to_stream(scanner) +batches = Enum.to_list(stream) +ExArrow.Stream.close(stream) + +ExArrow.Scanner.stats(stream) +``` + +Fragments are scanned in **path-sorted** order, one at a time. Early +`Enum.take/2` does not open later fragments. Call `ExArrow.Stream.close/1` when +abandoning a partially consumed scan from a long-lived process. + +`:batch_size` is accepted for API stability but reserved in 0.9 (batches +follow Parquet row-group sizing). + +## Pushdown ladder + +Strongest first: + +1. **Partition pruning (Elixir)** — predicates on Hive keys are evaluated + against each fragment's `partition_values`. Non-matching fragments are + never opened. +2. **Parquet filters (Rust / parquet-rs)** — pushable field-vs-scalar + predicates on **data** columns become `Parquet.Reader` `:filters` + (row-group statistics). Hive keys are stripped from this AST because they + are not columns in the file. +3. **Residual (Rust compute)** — anything left (`not_/1`, temporal scalars + the Parquet reader cannot bind yet, field-vs-field, OR mixes with + partition keys) runs through `Compute.filter/2` after decode. Partition + fields in a residual expression are bound to scalars for the current + fragment. + +`Scanner.stats/1` reports exact fragment prune/scan counts and aggregated +row-group selected/skipped counts (no `>= 1` hedges). + +## Expression vs callback + +Spec §13.3: Dataset filters must be **data** (Expression ASTs), not closures. +Closures cannot be analyzed for pushdown. Build with `field/1`, `scalar/1`, +and `eq/2` … `not_/1`. Validate with `Expression.validate/2` against a schema +or a field-name map (useful when merging Hive partition types). + +Legacy Parquet filter tuples (`{:gt, "col", value}`, `{:and, [...]}`) remain +accepted on the scanner and on `Parquet.Reader`. + +## Ordering + +0.9 scans fragments sequentially in lexicographic path order. Parallel +readahead and `:ordered` options are out of scope (stretch). Do not assume +global sorted-by-column order across fragments unless your layout guarantees +it. + +## PyArrow `pa.dataset` migration + +| PyArrow | ExArrow 0.9 | +|---------|-------------| +| `ds.dataset(path, format="parquet", partitioning="hive")` | `Dataset.open(path, partitioning: {:hive, schema: [...]})` | +| `dataset.to_table(filter=..., columns=...)` | `Dataset.scanner` + `Scanner.to_stream` + `Enum.to_list` | +| `pc.field("x") > 0` expressions | `E.gt(E.field("x"), E.scalar(0))` | +| Fragment metadata / files | `Dataset.fragments/1`, `Fragment.metadata/1` (Parquet) | +| Dataset write | Out of scope (later release) | +| S3 / fsspec | Out of scope (planned filesystem work) | + +## Related + +- Guide: this file (`guides/11_datasets.md`) +- Livebook: `livebook/06_datasets.livemd` +- Parquet pushdown background: `docs/parquet_guide.md`, `livebook/05_parquet.livemd` +- Modules: `ExArrow.Dataset`, `ExArrow.Dataset.Fragment`, `ExArrow.Scanner`, + `ExArrow.Compute.Expression`, `ExArrow.FileSystem` diff --git a/livebook/06_datasets.livemd b/livebook/06_datasets.livemd new file mode 100644 index 0000000..0a440ab --- /dev/null +++ b/livebook/06_datasets.livemd @@ -0,0 +1,182 @@ +# ExArrow — Datasets and Scanners (v0.9) + +```elixir +deps = [ + {:pythonx, "~> 0.4.2"}, + {:kino_pythonx, "~> 0.1.0"}, + {:kino, "~> 0.19.0"} +] + +# Opened from livebook/ in the repo → local source; otherwise Hex (precompiled NIF). +local? = File.exists?(Path.join(__DIR__, "../native/ex_arrow_native/Cargo.toml")) + +{ex_arrow_dep, extra_deps, config} = + if local? do + System.put_env("EX_ARROW_BUILD", "1") + + ex_arrow_beam = Path.join(__DIR__, "../_build/dev/lib/ex_arrow/ebin") + + if File.dir?(ex_arrow_beam) do + File.rm_rf!(ex_arrow_beam) + end + + { + {:ex_arrow, path: Path.expand("..", __DIR__)}, + [{:rustler, "~> 0.36", optional: true}], + [rustler_precompiled: [force_build: [ex_arrow: true]]] + } + else + { + {:ex_arrow, "~> 0.8.0"}, + [], + [] + } + end + +Mix.install(deps ++ [ex_arrow_dep] ++ extra_deps, config: config) +``` + +```pyproject.toml +[project] +name = "project" +version = "0.0.0" +requires-python = "==3.13.*" +dependencies = ["pyarrow"] +``` + +## Overview + +This notebook mirrors PyArrow `dataset` discovery + scan with ExArrow: + +* Hive-partitioned directory open +* Expression filters with partition prune + Parquet pushdown +* Scanner stats (exact fragment / row-group counts) +* Projection via `:columns` + +Run cells **top to bottom**. Prefer the checked-in fixture under +`test/fixtures/hive_events` when working from a git clone. + +--- + +## Locate the fixture (or write a tiny Hive tree) + +```elixir +repo_fixture = + Path.expand("../test/fixtures/hive_events", __DIR__) + +root = + if File.dir?(repo_fixture) do + repo_fixture + else + dir = Path.join(System.tmp_dir!(), "ex_arrow_hive_demo") + File.rm_rf!(dir) + + {:ok, batch} = + ExArrow.RecordBatch.from_lists([ + {"id", :s64, [1, 2]}, + {"amount", :f64, [10.0, 100.0]}, + {"account_id", :utf8, ["a", "b"]} + ]) + + schema = ExArrow.RecordBatch.schema(batch) + + for {rel, ids, amounts} <- [ + {"year=2025/month=12/part-0.parquet", [1], [10.0]}, + {"year=2026/month=01/part-0.parquet", [2], [100.0]} + ] do + {:ok, b} = + ExArrow.RecordBatch.from_lists([ + {"id", :s64, ids}, + {"amount", :f64, amounts}, + {"account_id", :utf8, List.duplicate("a", length(ids))} + ]) + + path = Path.join(dir, rel) + File.mkdir_p!(Path.dirname(path)) + :ok = ExArrow.Parquet.Writer.to_file(path, schema, [b]) + end + + dir + end + +{root, File.ls!(root)} +``` + +--- + +## PyArrow: open + filter + +```python +import pyarrow.dataset as ds + +hive_root = root.decode("utf-8") if isinstance(root, (bytes, bytearray)) else root + +dataset = ds.dataset(hive_root, format="parquet", partitioning="hive") +table = dataset.to_table(filter=(ds.field("year") >= 2026) & (ds.field("amount") > 50), columns=["id", "amount"]) +{"rows": table.num_rows, "ids": table.column("id").to_pylist()} +``` + +--- + +## ExArrow: Dataset + Scanner + +```elixir +alias ExArrow.Compute.Expression, as: E + +{:ok, dataset} = + ExArrow.Dataset.open(root, + partitioning: {:hive, schema: [{"year", :int32}, {"month", :int32}]} + ) + +filter = + E.and_( + E.gte(E.field("year"), E.scalar(2026)), + E.gt(E.field("amount"), E.scalar(50.0)) + ) + +{:ok, scanner} = + ExArrow.Dataset.scanner(dataset, columns: ["id", "amount"], filter: filter) + +{:ok, stream} = ExArrow.Scanner.to_stream(scanner) +batches = Enum.to_list(stream) +stats = ExArrow.Scanner.stats(stream) +:ok = ExArrow.Stream.close(stream) + +%{ + fragments: length(ExArrow.Dataset.fragments(dataset)), + batches: length(batches), + rows: Enum.map(batches, &ExArrow.RecordBatch.num_rows/1) |> Enum.sum(), + stats: stats +} +``` + +On the checked-in fixture, expect one emitted row (`id = 5`), one fragment +pruned by year, and exact row-group skip counts in `stats`. + +--- + +## Expression vs legacy tuple + +```elixir +# Fully pushable data predicate as a legacy tuple (still supported): +{:ok, scanner2} = + ExArrow.Dataset.scanner(dataset, filter: {:gt, "amount", 0.0}) + +{:ok, stream2} = ExArrow.Scanner.to_stream(scanner2) +rows2 = + stream2 + |> Enum.to_list() + |> Enum.map(&ExArrow.RecordBatch.num_rows/1) + |> Enum.sum() + +:ok = ExArrow.Stream.close(stream2) +%{rows_amount_gt_0: rows2} +``` + +Prefer `ExArrow.Compute.Expression` when the filter mixes partition keys, +`not_/1`, or anything that needs a residual after Parquet pushdown. + +## See also + +* Guide: `guides/11_datasets.md` +* Parquet pushdown notebook: `livebook/05_parquet.livemd` diff --git a/livebook/README.md b/livebook/README.md index b9a2a08..3907856 100644 --- a/livebook/README.md +++ b/livebook/README.md @@ -12,8 +12,9 @@ Tutorial notebooks for the **ex_arrow** library, suitable for an introductory Me | **03_adbc.livemd** | ADBC: `:adbc_package` backend, Database → Connection → Statement → Stream, metadata APIs (native driver), Explorer roundtrip. | | **04_adbc_integration.livemd** | **adbc_package** backend with connection pooling (NimblePool), concurrent queries. | | **05_parquet.livemd** | v0.8 Parquet pushdown, compressed write, multi-file directory, IPC file writer (PyArrow side-by-side). | +| **06_datasets.livemd** | v0.9 Dataset / Scanner: Hive open, Expression filters, prune stats (PyArrow `dataset` side-by-side). | -Together they demonstrate ExArrow functionality: IPC (stream + file), Flight (client + server + Flight SQL), ADBC (Arrow result streams), Parquet pushdown/multi-file, and the pipeline DSL (`ExArrow.Stream`, `ExArrow.Batch`, `ExArrow.Pipeline`, telemetry). +Together they demonstrate ExArrow functionality: IPC (stream + file), Flight (client + server + Flight SQL), ADBC (Arrow result streams), Parquet pushdown/multi-file, Datasets/Scanners, and the pipeline DSL (`ExArrow.Stream`, `ExArrow.Batch`, `ExArrow.Pipeline`, telemetry). ## How to run diff --git a/mix.exs b/mix.exs index 1a0b254..c49aaaa 100644 --- a/mix.exs +++ b/mix.exs @@ -119,6 +119,7 @@ defmodule ExArrow.MixProject do "guides/08_arrow_and_genstage.md", "guides/09_arrow_and_broadway.md", "guides/10_arrow_pipeline_patterns.md", + "guides/11_datasets.md", "docs/overview.md", "docs/memory_model.md", "docs/ipc_guide.md", @@ -136,6 +137,14 @@ defmodule ExArrow.MixProject do IPC: [ExArrow.IPC.Reader, ExArrow.IPC.Writer, ExArrow.IPC.File], Parquet: [ExArrow.Parquet.Reader, ExArrow.Parquet.Writer, ExArrow.Parquet.Metadata], "Compute kernels": [ExArrow.Compute, ExArrow.Compute.Expression], + Dataset: [ + ExArrow.Dataset, + ExArrow.Dataset.Fragment, + ExArrow.Scanner, + ExArrow.FileSystem, + ExArrow.FileSystem.Local, + ExArrow.FileSystem.Memory + ], "Batch operations": [ExArrow.Batch], Pipeline: [ ExArrow.Pipeline, diff --git a/script/generate_hive_events_fixture.py b/script/generate_hive_events_fixture.py new file mode 100755 index 0000000..4696e6d --- /dev/null +++ b/script/generate_hive_events_fixture.py @@ -0,0 +1,59 @@ +#!/usr/bin/env python3 +"""Generate the checked-in hive-partitioned Parquet fixture for ExArrow tests. + +Requires: pip install pyarrow + +Usage (from repo root): + + python3 script/generate_hive_events_fixture.py + +Writes under test/fixtures/hive_events/ with Hive layout: + + year=YYYY/month=MM/part-0.parquet + +Columns in each file: id (int64), amount (float64), account_id (utf8). +Partition keys live only in the path (not in the file schema). +""" + +from __future__ import annotations + +import shutil +from pathlib import Path + +import pyarrow as pa +import pyarrow.parquet as pq + +ROOT = Path(__file__).resolve().parents[1] / "test" / "fixtures" / "hive_events" + +# (year, month, rows) — rows are (id, amount, account_id) +PARTS = [ + (2025, 12, [(1, 10.0, "a"), (2, 0.0, "b")]), + (2026, 1, [(3, 25.5, "a"), (4, 0.0, "c"), (5, 100.0, "a")]), + (2026, 2, [(6, 50.0, "b"), (7, -1.0, "a")]), +] + + +def main() -> None: + if ROOT.exists(): + shutil.rmtree(ROOT) + + for year, month, rows in PARTS: + table = pa.table( + { + "id": pa.array([r[0] for r in rows], type=pa.int64()), + "amount": pa.array([r[1] for r in rows], type=pa.float64()), + "account_id": pa.array([r[2] for r in rows], type=pa.string()), + } + ) + dest = ROOT / f"year={year}" / f"month={month:02d}" + dest.mkdir(parents=True, exist_ok=True) + path = dest / "part-0.parquet" + # Force multiple row groups when there are enough rows (stats pruning). + row_group_size = 2 if len(rows) > 2 else max(len(rows), 1) + pq.write_table(table, path, row_group_size=row_group_size, compression="snappy") + meta = pq.read_metadata(path) + print(f"wrote {path} rows={meta.num_rows} row_groups={meta.num_row_groups}") + + +if __name__ == "__main__": + main() diff --git a/test/ex_arrow/compute/expression_test.exs b/test/ex_arrow/compute/expression_test.exs index cb94fd3..70db781 100644 --- a/test/ex_arrow/compute/expression_test.exs +++ b/test/ex_arrow/compute/expression_test.exs @@ -23,12 +23,28 @@ defmodule ExArrow.Compute.ExpressionTest do assert to_string(E.scalar(true)) == "scalar(true)" assert to_string(E.scalar("x")) == "scalar(\"x\")" assert to_string(E.scalar(~D[2026-01-01])) =~ "2026-01-01" + assert to_string(E.scalar(~N[2026-01-01 12:00:00])) =~ "2026-01-01" end test "scalar/1 rejects invalid UTF-8" do assert_raise ArgumentError, fn -> E.scalar(<<0xFF>>) end end + test "every comparison constructor builds the expected op" do + f = E.field("x") + s = E.scalar(1) + + assert %E{node: {:call, :eq, _}} = E.eq(f, s) + assert %E{node: {:call, :ne, _}} = E.ne(f, s) + assert %E{node: {:call, :gt, _}} = E.gt(f, s) + assert %E{node: {:call, :gte, _}} = E.gte(f, s) + assert %E{node: {:call, :lt, _}} = E.lt(f, s) + assert %E{node: {:call, :lte, _}} = E.lte(f, s) + assert %E{node: {:call, :and, _}} = E.and_(E.eq(f, s), E.ne(f, s)) + assert %E{node: {:call, :or, _}} = E.or_(E.eq(f, s), E.ne(f, s)) + assert %E{node: {:call, :not, _}} = E.not_(E.eq(f, s)) + end + test "comparisons and boolean composition render" do expr = E.and_( @@ -122,9 +138,25 @@ defmodule ExArrow.Compute.ExpressionTest do assert {nil, %E{}} = E.to_parquet_filters(expr) end - test "field-vs-field comparison is residual" do - expr = E.gt(E.field("a"), E.field("b")) - assert {nil, %E{}} = E.to_parquet_filters(expr) + test "OR pushes when both sides are pushable" do + expr = E.or_(E.eq(E.field("id"), E.scalar(1)), E.eq(E.field("id"), E.scalar(2))) + assert {{:or, [{:eq, "id", 1}, {:eq, "id", 2}]}, nil} = E.to_parquet_filters(expr) + end + + test "validate/2 accepts a field-name map (partition merge)" do + fields = %{"year" => :int32, "amount" => :float64} + + assert {:ok, _} = + E.validate( + E.and_( + E.gte(E.field("year"), E.scalar(2026)), + E.gt(E.field("amount"), E.scalar(0.0)) + ), + fields + ) + + assert {:error, msg} = E.validate(E.eq(E.field("missing"), E.scalar(1)), fields) + assert msg =~ "unknown field" end end @@ -215,6 +247,28 @@ defmodule ExArrow.Compute.ExpressionTest do assert {:call, :gte, [_, {:scalar, {:date32, _}}]} = E.encode_for_nif(E.gte(E.field("d"), E.scalar(~D[2026-01-01]))) + + naive = ~N[2026-03-01 08:30:00] + + assert {:call, :lt, [_, {:scalar, {:timestamp_micros, _}}]} = + E.encode_for_nif(E.lt(E.field("ts"), E.scalar(naive))) + end + + test "to_parquet_filters rejects non-utf8 string? covered via scalar builder" do + # Invalid UTF-8 cannot be built; empty string is pushable. + assert {{:eq, "name", ""}, nil} = E.to_parquet_filters(E.eq(E.field("name"), E.scalar(""))) + end + + test "validate rejects bad widths for unsigned and float32" do + schema = + schema_for([ + {"u8", :u8, [1]}, + {"f32", :f32, [1.0]} + ]) + + assert {:error, _} = E.validate(E.eq(E.field("u8"), E.scalar(-1)), schema) + assert {:error, _} = E.validate(E.eq(E.field("u8"), E.scalar(300)), schema) + assert {:ok, _} = E.validate(E.eq(E.field("f32"), E.scalar(1)), schema) end end end diff --git a/test/ex_arrow/hive_events_fixture_test.exs b/test/ex_arrow/hive_events_fixture_test.exs new file mode 100644 index 0000000..942fa13 --- /dev/null +++ b/test/ex_arrow/hive_events_fixture_test.exs @@ -0,0 +1,87 @@ +defmodule ExArrow.Fixtures.HiveEventsTest do + use ExUnit.Case, async: true + + alias ExArrow.Compute.Expression, as: E + alias ExArrow.Dataset + alias ExArrow.Native + alias ExArrow.RecordBatch + alias ExArrow.Scanner + alias ExArrow.Stream + + @fixture Path.expand("../fixtures/hive_events", __DIR__) + + defp s64_column(batch, name) do + ref = RecordBatch.resource_ref(batch) + {:ok, {binary, "s64", _n}} = Native.record_batch_column_buffer(ref, name) + for <>, do: v + end + + defp f64_column(batch, name) do + ref = RecordBatch.resource_ref(batch) + {:ok, {binary, "f64", _n}} = Native.record_batch_column_buffer(ref, name) + for <>, do: v + end + + @tag :nif + test "opens PyArrow hive fixture with exact partition values and schema" do + assert File.dir?(@fixture) + + assert {:ok, dataset} = + Dataset.open(@fixture, + partitioning: {:hive, schema: [{"year", :int32}, {"month", :int32}]} + ) + + frags = Dataset.fragments(dataset) + assert length(frags) == 3 + + assert Enum.map(frags, & &1.partition_values) == [ + %{"year" => 2025, "month" => 12}, + %{"year" => 2026, "month" => 1}, + %{"year" => 2026, "month" => 2} + ] + + assert ExArrow.Schema.field_names(Dataset.schema(dataset)) == [ + "id", + "amount", + "account_id" + ] + end + + @tag :nif + test "scan with partition prune + projection yields exact ids" do + assert {:ok, dataset} = + Dataset.open(@fixture, + partitioning: {:hive, schema: [{"year", :int32}, {"month", :int32}]} + ) + + filter = + E.and_( + E.gte(E.field("year"), E.scalar(2026)), + E.gt(E.field("amount"), E.scalar(50.0)) + ) + + assert {:ok, scanner} = Dataset.scanner(dataset, columns: ["id", "amount"], filter: filter) + assert {:ok, stream} = Scanner.to_stream(scanner) + + batches = Enum.to_list(stream) + ids = Enum.flat_map(batches, &s64_column(&1, "id")) + amounts = Enum.flat_map(batches, &f64_column(&1, "amount")) + + # Only id=5 (amount 100) survives year>=2026 and amount>50. + assert ids == [5] + assert amounts == [100.0] + + stats = Scanner.stats(stream) + assert stats.fragments_discovered == 3 + assert stats.fragments_pruned_partition == 1 + assert stats.fragments_selected == 2 + assert stats.fragments_scanned == 2 + # 2026/01 has 2 row groups; amount>50 skips the first (max 25.5). + # 2026/02 has 1 row group (max 50.0) skipped entirely. + assert stats.row_groups_skipped == 2 + assert stats.row_groups_selected == 1 + assert stats.rows_emitted == 1 + + Stream.close(stream) + end +end diff --git a/test/ex_arrow/scanner_test.exs b/test/ex_arrow/scanner_test.exs index d228657..69f1af5 100644 --- a/test/ex_arrow/scanner_test.exs +++ b/test/ex_arrow/scanner_test.exs @@ -254,6 +254,24 @@ defmodule ExArrow.ScannerTest do refute Partition.may_match?({:gte, "year", 2026}, %{"year" => 2025}) end + test "not_/1 and legacy and/or lists for partition prune" do + alias ExArrow.Scanner.Partition + + expr = E.not_(E.eq(E.field("year"), E.scalar(2025))) + assert Partition.may_match?(expr, %{"year" => 2026}) + refute Partition.may_match?(expr, %{"year" => 2025}) + + assert Partition.may_match?( + {:or, [{:eq, "year", 2026}, {:eq, "year", 2025}]}, + %{"year" => 2025} + ) + + refute Partition.may_match?( + {:and, [{:eq, "year", 2026}, {:eq, "month", 1}]}, + %{"year" => 2026, "month" => 2} + ) + end + test "compile strips partition keys from pushed filters" do alias ExArrow.Scanner.Compile @@ -276,6 +294,16 @@ defmodule ExArrow.ScannerTest do assert {:ok, {nil, %E{} = residual}} = Compile.compile(or_expr, ["year"]) assert E.to_string(residual) =~ "or_" end + + test "bind_partitions replaces hive fields with scalars" do + alias ExArrow.Scanner.Compile + + expr = E.and_(E.eq(E.field("year"), E.scalar(2026)), E.gt(E.field("id"), E.scalar(0))) + bound = Compile.bind_partitions(expr, %{"year" => 2026}) + assert to_string(bound) =~ "scalar(2026)" + assert to_string(bound) =~ "field(\"id\")" + assert Compile.bind_partitions(nil, %{}) == nil + end end describe "validation" do diff --git a/test/fixtures/README.md b/test/fixtures/README.md index 45a541b..b929026 100644 --- a/test/fixtures/README.md +++ b/test/fixtures/README.md @@ -21,6 +21,15 @@ field indices; used to regression-test statistics pruning. Regenerate with PyArrow (two `write_table` calls so each becomes a row group). +- **Hive-partitioned events (interop):** `hive_events/` is a small PyArrow + dataset under `year=*/month=*/part-0.parquet` with columns `id`, `amount`, + `account_id`. Used by Dataset/Scanner tests for exact partition prune and + row-group skip counts. Regenerate with: + + ```sh + python3 script/generate_hive_events_fixture.py + ``` + - **Cross-language corpus (v0.8+):** optional suite under `test/fixtures/arrow_testing/` populated by `script/fetch_arrow_testing.sh` from diff --git a/test/fixtures/hive_events/year=2025/month=12/part-0.parquet b/test/fixtures/hive_events/year=2025/month=12/part-0.parquet new file mode 100644 index 0000000000000000000000000000000000000000..c849dfb28d52b453b8403463a5f093229d9dbcff GIT binary patch literal 1041 zcmb7E&2G~`5FR@#%S1V(6}wtXKKPJYMWv-F3KA%Xu0u-^>0cy>!UZ`_6oS+yq)pR9 z4;**|jywVf9)Sno$dMy=UV)iiuM{N(iBHzf&hIxfvVEdjRnB)VBqGG z17GFhKKjet<6bCz>yf^V6|jDbb!7!h+=(chIZnY{WHJPmT_XXOnQVmg^Oz6Bv~0bF z=`NRcrecNqOzG$@ay)`{rz@RwxkAvaf%Go2xBat}d%)S6E8Sraco8pNdN!J#Uji?o zQcSaqv^T@v8Z<+s5iAIUdJG|FK@GfeK$+YXOni654<|l|{r}rzK>G*o(vHd~>i4`L z28L8bQp`#6uI7k7yYWv4sW1>x*|j(L`y&5MRm!R6X4Z5!;-eG)D#Tw4;oM4>YT;sR z{9FefVnH+kZTMjXL`GOe;?hZN9qJ zDU12fOn&vM*RE~r2-Tr3(B3gBwur@E(i}L?0nLRtB@%Y#HJOYEsj+Egi8f{g$Dn6z}XTl(KuzGgNNmz=A)14O8SM9l2Sfi+wUsnpA|HV)hMc8vKg^ z9V91e>YOD|m00pSOczT-eKh&stQPCSXiXis%c0W6B5c53^WQW4bVg|vk6)q`BT6gD zoBhc!0y?m_AeL+AMWeCPyc5l){gvF1cD*CO_T&QgK!}m>I$~3Zj;N`S@H`>5Ww0j1 zeYwM3?{JeLh=A>=RhFuSvNRk#uc4f(669$E93lMX!xztPJs!3;*ZWW44)KkM%H3_9 zFJ7sX$^3g!e(AW?EU&2-!j~+-??+dl2*n=BKA@j1*%#QJ1c_{;25|sYeszrevNccf zfqKv!%q()nd1``d7-oe%qsclJ=;KmFCK}=eZmg2&t zkKocraN!%&2XN`qrF;7be)nE(LZF)Xg`9KG?|04#a3q^$&UlHhHK1sfs95#L;KodI z;;UTTMZCy8?nQxbBY|&o1#Mj8U0FeknLIZn=9WHJQx9iy)_?h-^1s(RZC zV`89EB>5T1Y^ph;&u;wF0TmK~`i^~FUHkb zVeH`I6B8`<#TQ0l6cl8$dDwbv&K-Pn?9wd_CoD}efHH}z_B7R*sUQBQtj7I<5#OCdLHz{ za;Tr0eRUiiEhs3B(($`V?|d^GjW*NqcsMOx^u}j{;c4l9Wovt@99HpFH6QwoAL$YP G4E_KisH}_t literal 0 HcmV?d00001 From b37ef17001bfd5de168902f49cd1b71b76fbc12c Mon Sep 17 00:00:00 2001 From: thanos Date: Sun, 13 Sep 2026 09:29:07 -0400 Subject: [PATCH 09/11] Docs: expand Dataset/Scanner/Expression/FileSystem API documentation. Document struct fields, function parameters and options, and typical usage examples; add doctests for Expression builders and Memory/Local filesystem. --- lib/ex_arrow/compute/expression.ex | 197 ++++++++++++++++++-- lib/ex_arrow/dataset.ex | 207 +++++++++++++++++++--- lib/ex_arrow/dataset/fragment.ex | 50 +++++- lib/ex_arrow/file_system.ex | 72 ++++++-- lib/ex_arrow/file_system/local.ex | 12 ++ lib/ex_arrow/file_system/memory.ex | 62 ++++++- lib/ex_arrow/scanner.ex | 166 ++++++++++++++--- test/ex_arrow/compute/expression_test.exs | 2 + test/ex_arrow/file_system_test.exs | 3 + 9 files changed, 686 insertions(+), 85 deletions(-) diff --git a/lib/ex_arrow/compute/expression.ex b/lib/ex_arrow/compute/expression.ex index 42028a3..bd9e5a5 100644 --- a/lib/ex_arrow/compute/expression.ex +++ b/lib/ex_arrow/compute/expression.ex @@ -1,37 +1,75 @@ defmodule ExArrow.Compute.Expression do @moduledoc """ - Analyzable compute expression AST for filters and (later) Dataset scanners. + Analyzable compute expression AST for filters and Dataset scanners. Builders are the canonical API for 0.9. Macro sugar (`expr do ... end`) is - out of scope. Expressions are data: they can be validated against a schema, - printed for diagnostics, and partially compiled to the Parquet filter tuple - AST used since v0.8.0. They are not Elixir closures. + out of scope. Expressions are **data**: they can be validated against a + schema, printed for diagnostics, partially compiled to the Parquet filter + tuple AST (v0.8.0), and evaluated as residuals via + `ExArrow.Compute.filter/2`. They are not Elixir closures. - ## Example + ## Typical usage alias ExArrow.Compute.Expression, as: E filter = E.and_( - E.gte(E.field("date"), E.scalar(~D[2026-01-01])), + E.gte(E.field("year"), E.scalar(2026)), E.ne(E.field("amount"), E.scalar(0)) ) + {:ok, ^filter} = E.validate(filter, schema) {pushed, residual} = E.to_parquet_filters(filter) + + Use Expressions with `ExArrow.Dataset.scanner/2` so the Scanner can prune + partitions, push Parquet filters, and apply residuals. + + ## Examples + + iex> alias ExArrow.Compute.Expression, as: E + iex> to_string(E.eq(E.field("id"), E.scalar(1))) + "eq(field(\\"id\\"), scalar(1))" + + iex> alias ExArrow.Compute.Expression, as: E + iex> E.expression?(E.field("x")) + true """ alias ExArrow.Schema + @typedoc """ + An expression tree. + + ## Fields + + * `:node` — internal AST (`t:expr_node/0`). Prefer builders (`field/1`, + `scalar/1`, `eq/2`, ...) over constructing nodes by hand. + """ @type t :: %__MODULE__{node: expr_node()} defstruct [:node] + @typedoc """ + Internal AST node. + + * `{:field, name}` — column reference + * `{:scalar, value}` — literal (`t:scalar/0`) + * `{:call, op, args}` — comparison or boolean operator + """ @type expr_node :: {:field, String.t()} | {:scalar, scalar()} | {:call, op(), [expr_node()]} + @typedoc "Comparison and boolean operators in the AST." @type op :: :eq | :ne | :gt | :gte | :lt | :lte | :and | :or | :not + @typedoc """ + Literal values accepted by `scalar/1`. + + Temporal values (`Date`, `NaiveDateTime`, `DateTime`) validate against + date/timestamp columns but are **residual** for Parquet pushdown in 0.9 + (they are not bound into the v0.8 filter tuple AST yet). + """ @type scalar :: integer() | float() @@ -45,6 +83,20 @@ defmodule ExArrow.Compute.Expression do @doc """ Reference a column by name. + + ## Parameters + + * `name` — UTF-8 string or atom (atoms are converted with `Atom.to_string/1`) + + ## Examples + + iex> alias ExArrow.Compute.Expression, as: E + iex> to_string(E.field("amount")) + "field(\\"amount\\")" + + iex> alias ExArrow.Compute.Expression, as: E + iex> E.field(:amount) == E.field("amount") + true """ @spec field(String.t() | atom()) :: t() def field(name) when is_binary(name), do: %__MODULE__{node: {:field, name}} @@ -54,8 +106,22 @@ defmodule ExArrow.Compute.Expression do @doc """ A scalar literal. - Supported values: integer, float, boolean, UTF-8 string, `Date`, - `NaiveDateTime`, and `DateTime`. + ## Parameters + + * `value` — integer, float, boolean, UTF-8 string, `Date`, + `NaiveDateTime`, or `DateTime` + + Raises `ArgumentError` for invalid UTF-8 or unsupported terms. + + ## Examples + + iex> alias ExArrow.Compute.Expression, as: E + iex> to_string(E.scalar(42)) + "scalar(42)" + + iex> alias ExArrow.Compute.Expression, as: E + iex> to_string(E.scalar(true)) + "scalar(true)" """ @spec scalar(scalar()) :: t() def scalar(%Date{} = d), do: %__MODULE__{node: {:scalar, d}} @@ -77,37 +143,59 @@ defmodule ExArrow.Compute.Expression do do: raise(ArgumentError, "unsupported scalar: #{inspect(other)}") @doc """ - Equality comparison. + Equality comparison (`left == right`). + + ## Parameters + + * `left`, `right` — field or scalar expressions + + ## Examples + + iex> alias ExArrow.Compute.Expression, as: E + iex> to_string(E.eq(E.field("ok"), E.scalar(true))) + "eq(field(\\"ok\\"), scalar(true))" """ @spec eq(t(), t()) :: t() def eq(%__MODULE__{} = l, %__MODULE__{} = r), do: call(:eq, [l, r]) @doc """ - Inequality comparison. + Inequality comparison (`left != right`). + + ## Examples + + iex> alias ExArrow.Compute.Expression, as: E + iex> to_string(E.ne(E.field("amount"), E.scalar(0))) + "ne(field(\\"amount\\"), scalar(0))" """ @spec ne(t(), t()) :: t() def ne(%__MODULE__{} = l, %__MODULE__{} = r), do: call(:ne, [l, r]) @doc """ - Greater-than comparison. + Greater-than comparison (`left > right`). + + ## Examples + + iex> alias ExArrow.Compute.Expression, as: E + iex> to_string(E.gt(E.field("score"), E.scalar(0.9))) + "gt(field(\\"score\\"), scalar(0.9))" """ @spec gt(t(), t()) :: t() def gt(%__MODULE__{} = l, %__MODULE__{} = r), do: call(:gt, [l, r]) @doc """ - Greater-than-or-equal comparison. + Greater-than-or-equal comparison (`left >= right`). """ @spec gte(t(), t()) :: t() def gte(%__MODULE__{} = l, %__MODULE__{} = r), do: call(:gte, [l, r]) @doc """ - Less-than comparison. + Less-than comparison (`left < right`). """ @spec lt(t(), t()) :: t() def lt(%__MODULE__{} = l, %__MODULE__{} = r), do: call(:lt, [l, r]) @doc """ - Less-than-or-equal comparison. + Less-than-or-equal comparison (`left <= right`). """ @spec lte(t(), t()) :: t() def lte(%__MODULE__{} = l, %__MODULE__{} = r), do: call(:lte, [l, r]) @@ -116,6 +204,13 @@ defmodule ExArrow.Compute.Expression do Boolean AND of two expressions. Named `and_/2` because `and/2` is a Kernel special form. + + ## Examples + + iex> alias ExArrow.Compute.Expression, as: E + iex> expr = E.and_(E.gt(E.field("a"), E.scalar(0)), E.lt(E.field("a"), E.scalar(10))) + iex> to_string(expr) =~ "and_(" + true """ @spec and_(t(), t()) :: t() def and_(%__MODULE__{} = l, %__MODULE__{} = r), do: call(:and, [l, r]) @@ -132,22 +227,64 @@ defmodule ExArrow.Compute.Expression do Boolean NOT. Named `not_/1` because `not/1` is a Kernel special form. + + Always residual for Parquet pushdown (the v0.8 filter AST has no NOT). + + ## Examples + + iex> alias ExArrow.Compute.Expression, as: E + iex> {nil, residual} = E.to_parquet_filters(E.not_(E.eq(E.field("ok"), E.scalar(true)))) + iex> to_string(residual) =~ "not_(" + true """ @spec not_(t()) :: t() def not_(%__MODULE__{} = e), do: call(:not, [e]) @doc """ Returns `true` if `term` is an `ExArrow.Compute.Expression`. + + ## Examples + + iex> ExArrow.Compute.Expression.expression?(ExArrow.Compute.Expression.field("x")) + true + + iex> ExArrow.Compute.Expression.expression?(:nope) + false """ @spec expression?(term()) :: boolean() def expression?(%__MODULE__{}), do: true def expression?(_), do: false @doc """ - Type-check `expr` against `schema`. + Type-check `expr` against a schema or a field-name map. Checks that field names exist and that comparisons are type-compatible with the referenced column (and the other side, when both are fields). + + ## Parameters + + * `expr` — expression to validate + * `schema_or_fields` — either: + + * an `ExArrow.Schema.t()`, or + * a `%{String.t() => type_atom}` map (useful when merging Hive + partition types into the file schema for Scanner validation) + + ## Returns + + * `{:ok, expr}` when valid + * `{:error, message}` for unknown fields or type mismatches + + ## Examples + + {:ok, batch} = ExArrow.RecordBatch.from_lists([{"amount", :s64, [1]}]) + schema = ExArrow.RecordBatch.schema(batch) + alias ExArrow.Compute.Expression, as: E + {:ok, _} = E.validate(E.gt(E.field("amount"), E.scalar(0)), schema) + + # Partition keys for Dataset.scanner/2: + fields = Map.merge(%{"amount" => :int64}, %{"year" => :int32}) + {:ok, _} = E.validate(E.gte(E.field("year"), E.scalar(2026)), fields) """ @spec validate(t(), Schema.t() | %{optional(String.t()) => term()}) :: {:ok, t()} | {:error, String.t()} @@ -171,7 +308,13 @@ defmodule ExArrow.Compute.Expression do Split `expr` into a Parquet-pushable filter AST and an optional residual expression. - Returns `{pushed, residual}` where: + ## Parameters + + * `expr` — expression to split + + ## Returns + + `{pushed, residual}` where: - `pushed` is `nil` or a v0.8.0 filter tuple (`{:eq|:ne|:gt|:gte|:lt|:lte, col, value}` / `{:and|:or, [...]}`) @@ -181,6 +324,22 @@ defmodule ExArrow.Compute.Expression do AND may push one side and residual the other. OR is pushed only when both sides are fully pushable; otherwise the whole OR is residual. + + ## Examples + + iex> alias ExArrow.Compute.Expression, as: E + iex> E.to_parquet_filters(E.gt(E.field("score"), E.scalar(0.9))) + {{:gt, "score", 0.9}, nil} + + iex> alias ExArrow.Compute.Expression, as: E + iex> {pushed, residual} = + ...> E.to_parquet_filters( + ...> E.and_(E.gt(E.field("amount"), E.scalar(0)), E.gte(E.field("day"), E.scalar(~D[2026-01-01]))) + ...> ) + iex> pushed + {:gt, "amount", 0} + iex> match?(%E{}, residual) + true """ @spec to_parquet_filters(t()) :: {term() | nil, t() | nil} def to_parquet_filters(%__MODULE__{node: node}) do @@ -195,6 +354,12 @@ defmodule ExArrow.Compute.Expression do @doc """ Render `expr` as a diagnostic string. + + ## Examples + + iex> alias ExArrow.Compute.Expression, as: E + iex> ExArrow.Compute.Expression.to_string(E.lt(E.field("x"), E.scalar(3))) + "lt(field(\\"x\\"), scalar(3))" """ @spec to_string(t()) :: String.t() def to_string(%__MODULE__{node: node}), do: render(node) diff --git a/lib/ex_arrow/dataset.ex b/lib/ex_arrow/dataset.ex index 1f18ee4..9a4fe5f 100644 --- a/lib/ex_arrow/dataset.ex +++ b/lib/ex_arrow/dataset.ex @@ -2,12 +2,13 @@ defmodule ExArrow.Dataset do @moduledoc """ Dataset discovery over Parquet (and IPC) files. - A Dataset is a discovered set of fragments — usually files under a directory, - optionally with Hive partition values parsed from the path. Scanning is - handled by `ExArrow.Scanner` (v0.9.0 M5); this module only discovers and - describes fragments. + A Dataset is the result of finding files and describing them as fragments. + It does **not** decode row groups. Use `ExArrow.Scanner` to project, filter, + and stream batches. - ## Example + ## Typical workflow + + alias ExArrow.Compute.Expression, as: E {:ok, dataset} = ExArrow.Dataset.open("/data/events", @@ -15,26 +16,45 @@ defmodule ExArrow.Dataset do partitioning: {:hive, schema: [{"year", :int32}, {"month", :int32}]} ) - ExArrow.Dataset.fragments(dataset) - ExArrow.Dataset.schema(dataset) + fragments = ExArrow.Dataset.fragments(dataset) + schema = ExArrow.Dataset.schema(dataset) + + filter = + E.and_( + E.gte(E.field("year"), E.scalar(2026)), + E.gt(E.field("amount"), E.scalar(0.0)) + ) + + {:ok, scanner} = + ExArrow.Dataset.scanner(dataset, columns: ["id", "amount"], filter: filter) - ## Sources + {:ok, stream} = ExArrow.Scanner.to_stream(scanner) + batches = Enum.to_list(stream) + :ok = ExArrow.Stream.close(stream) - `open/2` accepts: + ## Sources for `open/2` - - a directory path (recursive discovery) - - a single file path - - a glob pattern (`*` / `**`) - - an explicit list of file paths + - a **directory** path (recursive discovery of matching files) + - a **single file** path + - a **glob** pattern (`*` within a segment, `**` across segments) + - an explicit **list** of file paths - ## Options + ## Options for `open/2` * `:format` — `:parquet` (default) or `:ipc` - * `:partitioning` — `:none` (default) or `{:hive, schema: [{name, type}, ...]}` - * `:filesystem` — `ExArrow.FileSystem` handle (default `FileSystem.Local.new()`) - * `:ignore_hidden` — skip `.` / `_`-prefixed path components (default `true`) - * `:schema` — optional `ExArrow.Schema` to skip footer schema resolution - * `:root` — dataset root for Hive relative paths (inferred when omitted) + * `:partitioning` — `:none` (default) or + `{:hive, schema: [{name, type}, ...]}` (see `t:partition_schema/0`) + * `:filesystem` — `ExArrow.FileSystem` handle (default + `ExArrow.FileSystem.Local.new/0`) + * `:ignore_hidden` — skip path components whose basename starts with + `.` or `_` (default `true`) + * `:schema` — optional `ExArrow.Schema.t()` to skip footer / IPC schema + resolution (required for Memory-only discovery when paths are not + OS-readable) + * `:root` — dataset root used when parsing Hive relative paths + (inferred from the source when omitted) + + See also: `guides/11_datasets.md`, `livebook/06_datasets.livemd`. """ alias ExArrow.Dataset.Fragment @@ -49,7 +69,13 @@ defmodule ExArrow.Dataset do @enforce_keys [:format, :partitioning, :filesystem, :ignore_hidden, :root, :fragments, :schema] defstruct [:format, :partitioning, :filesystem, :ignore_hidden, :root, :fragments, :schema] - @typedoc "Arrow-ish type atom used when coercing Hive `key=value` path segments." + @typedoc """ + Arrow-ish type atom used when coercing Hive `key=value` path segments. + + Integers are range-checked for the named width. `:date32` accepts ISO-8601 + date strings. `:utf8` URL-decodes the value. `:boolean` accepts + `true`/`false`/`1`/`0` (case-insensitive). + """ @type partition_type :: :int8 | :int16 @@ -65,12 +91,40 @@ defmodule ExArrow.Dataset do | :boolean | :date32 - @typedoc "Ordered list of `{name, type}` pairs for `{:hive, schema: ...}`." + @typedoc """ + Ordered list of `{column_name, type}` pairs for Hive partitioning. + + Example: `[{"year", :int32}, {"month", :int32}]` matches paths like + `.../year=2026/month=01/part-0.parquet`. + """ @type partition_schema :: [{String.t(), partition_type()}] + @typedoc """ + How fragment paths contribute partition columns. + + * `:none` — no path parsing; every fragment has `partition_values: %{}` + * `{:hive, schema}` — parse `key=value` segments under the dataset root + using `schema` (see `t:partition_schema/0`) + """ @type partitioning :: :none | {:hive, partition_schema()} + + @typedoc "On-disk format of every fragment in this dataset." @type format :: :parquet | :ipc + @typedoc """ + A discovered Dataset. + + ## Fields + + * `:format` — `:parquet` or `:ipc` (from `open/2`) + * `:partitioning` — `:none` or `{:hive, schema}` used at open time + * `:filesystem` — discovery backend (`Local` or `Memory`) + * `:ignore_hidden` — whether hidden path components were skipped + * `:root` — root path for Hive relative parsing (often absolute on Local) + * `:fragments` — path-sorted `ExArrow.Dataset.Fragment` list + * `:schema` — Arrow schema from the first fragment footer / IPC metadata, + or the caller-supplied `:schema` option + """ @type t :: %__MODULE__{ format: format(), partitioning: partitioning(), @@ -85,6 +139,55 @@ defmodule ExArrow.Dataset do @doc """ Discover fragments for `source` and resolve the dataset schema. + + Performs discovery IO (list/glob/exists) and, unless `:schema` is passed, + opens the first fragment's footer (Parquet) or IPC file metadata. Does + **not** decode data pages. + + ## Parameters + + * `source` — directory, file path, glob string, or list of file paths + * `opts` — see the module documentation (format, partitioning, filesystem, + ignore_hidden, schema, root) + + ## Returns + + * `{:ok, dataset}` on success + * `{:error, message}` for validation failures, missing paths, empty + discovery, malformed Hive segments, or schema resolution errors + + ## Examples + + Open a Hive-partitioned directory: + + {:ok, dataset} = + ExArrow.Dataset.open("/data/events", + partitioning: {:hive, schema: [{"year", :int32}, {"month", :int32}]} + ) + + Open an explicit file list with a known schema (no footer read): + + {:ok, dataset} = + ExArrow.Dataset.open( + ["/data/a.parquet", "/data/b.parquet"], + schema: schema, + root: "/data" + ) + + Discover via Memory filesystem (tests): + + {:ok, fs} = + ExArrow.FileSystem.Memory.new(%{ + "/data/year=2026/part-0.parquet" => 128 + }) + + {:ok, dataset} = + ExArrow.Dataset.open("/data", + filesystem: fs, + schema: schema, + partitioning: {:hive, schema: [{"year", :int32}]}, + root: "/data" + ) """ @spec open(String.t() | [String.t()], keyword()) :: {:ok, t()} | {:error, String.t()} def open(source, opts \\ []) @@ -117,21 +220,75 @@ defmodule ExArrow.Dataset do def open(_source, _opts), do: {:error, "source must be a path string or a list of paths"} @doc """ - Return discovered fragments in path-sorted order. + Return discovered fragments in lexicographic path order. + + ## Parameters + + * `dataset` — an `ExArrow.Dataset.t()` from `open/2` + + ## Examples + + frags = ExArrow.Dataset.fragments(dataset) + Enum.map(frags, & &1.partition_values) + # => [%{"year" => 2025, "month" => 12}, %{"year" => 2026, "month" => 1}] """ @spec fragments(t()) :: [Fragment.t()] def fragments(%__MODULE__{fragments: fragments}), do: fragments @doc """ - Return the dataset schema resolved at open time (footer / IPC metadata only). + Return the Arrow schema resolved at open time. + + Comes from the first fragment's Parquet footer / IPC file metadata, or from + the `:schema` option passed to `open/2`. No data pages are read. + + ## Parameters + + * `dataset` — an `ExArrow.Dataset.t()` from `open/2` + + ## Examples + + schema = ExArrow.Dataset.schema(dataset) + ExArrow.Schema.field_names(schema) + # => ["id", "amount", "account_id"] """ @spec schema(t()) :: Schema.t() def schema(%__MODULE__{schema: schema}), do: schema @doc """ - Build a lazy `ExArrow.Scanner` over this dataset (no IO). + Build a lazy `ExArrow.Scanner` over this dataset. + + Performs **no IO**. Validation of `:columns` / `:filter` / `:batch_size` + happens here; file opens start in `ExArrow.Scanner.to_stream/1`. + + ## Parameters + + * `dataset` — discovered dataset + * `opts` — scanner options: + + * `:columns` — non-empty list of column name strings to project, or + omit for all columns + * `:filter` — `ExArrow.Compute.Expression.t()`, legacy Parquet filter + tuple (`{:gt, "col", value}`, `{:and, [...]}`, ...), or omit/`nil` + * `:batch_size` — positive integer accepted for API stability; reserved + in 0.9 (batches follow Parquet row-group sizing) + + ## Returns + + * `{:ok, scanner}` when options validate + * `{:error, message}` for unknown options, bad columns, or filter + validation failures (including unknown Expression fields) + + ## Examples + + alias ExArrow.Compute.Expression, as: E + + {:ok, scanner} = + ExArrow.Dataset.scanner(dataset, + columns: ["id"], + filter: E.gte(E.field("year"), E.scalar(2026)) + ) - See `ExArrow.Scanner.new/2` for options (`:columns`, `:filter`, `:batch_size`). + {:ok, stream} = ExArrow.Scanner.to_stream(scanner) """ @spec scanner(t(), keyword()) :: {:ok, ExArrow.Scanner.t()} | {:error, String.t()} def scanner(%__MODULE__{} = dataset, opts \\ []) when is_list(opts) do diff --git a/lib/ex_arrow/dataset/fragment.ex b/lib/ex_arrow/dataset/fragment.ex index b11549b..6a65fb8 100644 --- a/lib/ex_arrow/dataset/fragment.ex +++ b/lib/ex_arrow/dataset/fragment.ex @@ -2,6 +2,25 @@ defmodule ExArrow.Dataset.Fragment do @moduledoc """ One readable unit in a Dataset: a file path plus optional Hive partition values discovered from the path. + + Fragments are produced by `ExArrow.Dataset.open/2`. They describe *what* + can be scanned; they do not hold open file handles. Scanning opens each + path on demand via `ExArrow.Scanner`. + + ## Example + + {:ok, dataset} = + ExArrow.Dataset.open("/data/events", + partitioning: {:hive, schema: [{"year", :int32}]} + ) + + [frag | _] = ExArrow.Dataset.fragments(dataset) + frag.path + frag.partition_values + # => %{"year" => 2026} + + {:ok, meta} = ExArrow.Dataset.Fragment.metadata(frag) + meta.num_rows """ alias ExArrow.Parquet.Metadata @@ -9,8 +28,22 @@ defmodule ExArrow.Dataset.Fragment do @enforce_keys [:path, :format, :partition_values, :size] defstruct [:path, :format, :partition_values, :size] + @typedoc "On-disk format of this fragment (matches the parent Dataset)." @type format :: :parquet | :ipc + @typedoc """ + A discovered fragment. + + ## Fields + + * `:path` — absolute or dataset-relative file path (Local discovery + typically expands to an absolute path) + * `:format` — `:parquet` or `:ipc` + * `:partition_values` — map of Hive column name => coerced term + (empty map when partitioning is `:none`) + * `:size` — file size in bytes as reported by the filesystem (may be `0` + when unknown) + """ @type t :: %__MODULE__{ path: String.t(), format: format(), @@ -19,9 +52,22 @@ defmodule ExArrow.Dataset.Fragment do } @doc """ - Read Parquet footer metadata for this fragment (no data pages). + Read Parquet footer metadata for this fragment without decoding data pages. + + ## Parameters + + * `fragment` — must have `format: :parquet` and an OS-readable `:path` + + ## Returns + + * `{:ok, %ExArrow.Parquet.Metadata{}}` with row-group and column stats + * `{:error, message}` for IPC fragments, missing files, or read failures + + ## Examples - Only supported for `:parquet` fragments with an OS-readable path. + {:ok, meta} = ExArrow.Dataset.Fragment.metadata(frag) + meta.num_row_groups + meta.num_rows """ @spec metadata(t()) :: {:ok, Metadata.t()} | {:error, String.t()} def metadata(%__MODULE__{format: :parquet, path: path}) when is_binary(path) do diff --git a/lib/ex_arrow/file_system.ex b/lib/ex_arrow/file_system.ex index 5cb8045..763d154 100644 --- a/lib/ex_arrow/file_system.ex +++ b/lib/ex_arrow/file_system.ex @@ -20,25 +20,38 @@ defmodule ExArrow.FileSystem do starts with `.` or `_` is skipped. That matches Dataset's `:ignore_hidden` option (dotfiles and `_`-prefixed Hive / staging dirs). - ## Example + ## Typical usage fs = ExArrow.FileSystem.Local.new() - {:ok, entries} = ExArrow.FileSystem.list(fs, "/data/events") + {:ok, entries} = ExArrow.FileSystem.list(fs, "/data/events", recursive: true) {:ok, paths} = ExArrow.FileSystem.glob(fs, "/data/events/**/*.parquet") true = ExArrow.FileSystem.exists?(fs, "/data/events") + + {:ok, dataset} = ExArrow.Dataset.open("/data/events", filesystem: fs) """ @typedoc "Filesystem handle (struct whose module implements this behaviour)." @type t :: struct() - @typedoc "One discovered path." + @typedoc """ + One discovered path. + + ## Keys + + * `:path` — absolute or normalized path string + * `:type` — `:file` or `:directory` + * `:size` — byte size for files; `0` for directories (and when unknown) + """ @type entry :: %{ path: String.t(), type: :file | :directory, size: non_neg_integer() } + @typedoc "Option for `list/3`." @type list_opt :: {:recursive, boolean()} | {:ignore_hidden, boolean()} + + @typedoc "Option for `glob/3`." @type glob_opt :: {:ignore_hidden, boolean()} @callback list(t(), String.t(), keyword()) :: {:ok, [entry()]} | {:error, String.t()} @@ -48,11 +61,26 @@ defmodule ExArrow.FileSystem do @doc """ List entries under `path`. - ## Options + ## Parameters + + * `fs` — filesystem handle (`Local` or `Memory`) + * `path` — directory or file to list + * `opts`: - * `:recursive` — when `true` (default), walk the whole tree; when `false`, - only immediate children - * `:ignore_hidden` — when `true` (default), skip `.` / `_`-prefixed names + * `:recursive` — when `true` (default), walk the whole tree; when + `false`, only immediate children + * `:ignore_hidden` — when `true` (default), skip `.` / `_`-prefixed names + + ## Returns + + * `{:ok, entries}` — list of `t:entry/0` maps, typically path-sorted + * `{:error, message}` — missing path, invalid opts, or backend failure + + ## Examples + + fs = ExArrow.FileSystem.Local.new() + {:ok, entries} = ExArrow.FileSystem.list(fs, "/data/events", recursive: false) + Enum.map(entries, &{&1.type, &1.path}) """ @spec list(t(), String.t(), [list_opt()]) :: {:ok, [entry()]} | {:error, String.t()} def list(fs, path, opts \\ []) @@ -75,10 +103,24 @@ defmodule ExArrow.FileSystem do Patterns use `/` separators. `*` matches within one path segment; `**` matches across segments (including zero segments). - ## Options + ## Parameters - * `:ignore_hidden` — when `true` (default), skip matches with a `.` / - `_`-prefixed path component + * `fs` — filesystem handle + * `pattern` — glob string (for example `"/data/**/*.parquet"`) + * `opts`: + + * `:ignore_hidden` — when `true` (default), skip matches with a `.` / + `_`-prefixed path component + + ## Returns + + * `{:ok, paths}` — sorted list of matching **file** paths + * `{:error, message}` — invalid pattern or opts + + ## Examples + + fs = ExArrow.FileSystem.Local.new() + {:ok, paths} = ExArrow.FileSystem.glob(fs, "/data/events/year=*/**/*.parquet") """ @spec glob(t(), String.t(), [glob_opt()]) :: {:ok, [String.t()]} | {:error, String.t()} def glob(fs, pattern, opts \\ []) @@ -97,6 +139,16 @@ defmodule ExArrow.FileSystem do @doc """ Return whether `path` exists as a file or directory. + + ## Parameters + + * `fs` — filesystem handle + * `path` — path string (non-binaries return `false`) + + ## Examples + + fs = ExArrow.FileSystem.Local.new() + ExArrow.FileSystem.exists?(fs, "/data/events") """ @spec exists?(t(), String.t()) :: boolean() def exists?(%mod{} = fs, path) when is_binary(path), do: mod.exists?(fs, path) diff --git a/lib/ex_arrow/file_system/local.ex b/lib/ex_arrow/file_system/local.ex index 73e0f96..d9e0223 100644 --- a/lib/ex_arrow/file_system/local.ex +++ b/lib/ex_arrow/file_system/local.ex @@ -4,6 +4,13 @@ defmodule ExArrow.FileSystem.Local do Paths are expanded with `Path.expand/1` before use. File contents are not read here; Dataset / Parquet NIFs open paths returned by discovery. + + ## Example + + fs = ExArrow.FileSystem.Local.new() + {:ok, entries} = ExArrow.FileSystem.list(fs, "/data/events") + {:ok, paths} = ExArrow.FileSystem.glob(fs, "/data/**/*.parquet") + ExArrow.FileSystem.exists?(fs, "/data/events") """ @behaviour ExArrow.FileSystem @@ -12,10 +19,15 @@ defmodule ExArrow.FileSystem.Local do defstruct [] + @typedoc "Empty handle; all state is the OS filesystem." @type t :: %__MODULE__{} @doc """ Build a local filesystem handle. + + ## Examples + + iex> %ExArrow.FileSystem.Local{} = ExArrow.FileSystem.Local.new() """ @spec new() :: t() def new, do: %__MODULE__{} diff --git a/lib/ex_arrow/file_system/memory.ex b/lib/ex_arrow/file_system/memory.ex index 15575af..0359f00 100644 --- a/lib/ex_arrow/file_system/memory.ex +++ b/lib/ex_arrow/file_system/memory.ex @@ -6,15 +6,21 @@ defmodule ExArrow.FileSystem.Memory do also registers parent directories so `list/3` can walk a Hive-style tree without touching the OS. + ## Fields + + * `:entries` — `%{path => ExArrow.FileSystem.entry()}` + ## Examples - fs = ExArrow.FileSystem.Memory.new() - {:ok, fs} = ExArrow.FileSystem.Memory.put_file(fs, "/data/a.parquet", size: 128) + iex> {:ok, fs} = ExArrow.FileSystem.Memory.new(%{"/data/a.parquet" => 128}) + iex> ExArrow.FileSystem.exists?(fs, "/data/a.parquet") + true - {:ok, fs} = - ExArrow.FileSystem.Memory.new(%{ - "/data/year=2026/part-0.parquet" => 128 - }) + iex> fs = ExArrow.FileSystem.Memory.new() + iex> {:ok, fs} = ExArrow.FileSystem.Memory.put_file(fs, "/data/year=2026/part.parquet", size: 64) + iex> {:ok, paths} = ExArrow.FileSystem.glob(fs, "/data/**/*.parquet") + iex> paths + ["/data/year=2026/part.parquet"] """ @behaviour ExArrow.FileSystem @@ -23,11 +29,24 @@ defmodule ExArrow.FileSystem.Memory do defstruct entries: %{} + @typedoc "Internal path => entry map." @type entry_map :: %{optional(String.t()) => FileSystem.entry()} + + @typedoc """ + Memory filesystem handle. + + ## Fields + + * `:entries` — normalized paths to `t:ExArrow.FileSystem.entry/0` maps + """ @type t :: %__MODULE__{entries: entry_map()} @doc """ Build an empty memory filesystem. + + ## Examples + + iex> %ExArrow.FileSystem.Memory{entries: %{}} = ExArrow.FileSystem.Memory.new() """ @spec new() :: t() def new, do: %__MODULE__{} @@ -35,7 +54,21 @@ defmodule ExArrow.FileSystem.Memory do @doc """ Build a memory filesystem from a path → size map or `{path, size}` list. - Returns `{:ok, fs}` or `{:error, message}`. + ## Parameters + + * `seed` — `%{path => size}` or `[{path, size}, ...]` where `size` is a + non-negative integer (file byte size) + + ## Returns + + * `{:ok, fs}` on success + * `{:error, message}` for invalid paths or sizes + + ## Examples + + iex> {:ok, fs} = ExArrow.FileSystem.Memory.new(%{"/data/a.parquet" => 10}) + iex> ExArrow.FileSystem.exists?(fs, "/data") + true """ @spec new(map() | [{String.t(), non_neg_integer()}]) :: {:ok, t()} | {:error, String.t()} @@ -46,6 +79,21 @@ defmodule ExArrow.FileSystem.Memory do Register a file at `path` with `size` (default `0`). Creates missing parent directories. Returns `{:ok, fs}` or `{:error, msg}`. + + ## Parameters + + * `fs` — memory filesystem + * `path` — absolute-style path string (normalized with a leading `/`) + * `opts`: + + * `:size` — non-negative integer byte size (default `0`) + + ## Examples + + iex> fs = ExArrow.FileSystem.Memory.new() + iex> {:ok, fs} = ExArrow.FileSystem.Memory.put_file(fs, "/data/x.parquet", size: 32) + iex> ExArrow.FileSystem.exists?(fs, "/data/x.parquet") + true """ @spec put_file(t(), String.t(), keyword()) :: {:ok, t()} | {:error, String.t()} def put_file(fs, path, opts \\ []) diff --git a/lib/ex_arrow/scanner.ex b/lib/ex_arrow/scanner.ex index acfed9e..f82d33e 100644 --- a/lib/ex_arrow/scanner.ex +++ b/lib/ex_arrow/scanner.ex @@ -3,38 +3,60 @@ defmodule ExArrow.Scanner do Lazy scan of an `ExArrow.Dataset` with projection, partition pruning, and filter pushdown. - Building a scanner does no IO. `to_stream/1` starts an Agent-backed - `ExArrow.Stream` (`backend: :dataset`) that opens fragments on demand. + Building a scanner (`new/2` / `ExArrow.Dataset.scanner/2`) does **no IO**. + `to_stream/1` starts an Agent-backed `ExArrow.Stream` (`backend: :dataset`) + that opens fragments on demand in path-sorted order. ## Pushdown ladder 1. **Partition pruning** — predicates on Hive keys are evaluated against each fragment's `partition_values` (no file open). - 2. **Parquet filters** — remaining pushable predicates become - `Parquet.Reader` `:filters` (row-group stats). - 3. **Residual** — anything left runs through `Compute.filter/2` after decode. - Partition fields in a residual expression are bound to scalars for the - current fragment. + 2. **Parquet filters** — remaining pushable predicates on data columns + become `Parquet.Reader` `:filters` (row-group statistics). Hive keys are + stripped from this AST because they are not columns in the file. + 3. **Residual** — anything left runs through `ExArrow.Compute.filter/2` + after decode. Partition fields in a residual expression are bound to + scalars for the current fragment. ## Options - * `:columns` — list of column names to project (Parquet pushdown / IPC project) - * `:filter` — `ExArrow.Compute.Expression`, legacy Parquet filter tuple, or `nil` - * `:batch_size` — accepted for API stability; reserved (row-group sized batches in 0.9) + * `:columns` — list of column names to project (Parquet pushdown / IPC + `Compute.project/2`) + * `:filter` — `ExArrow.Compute.Expression.t()`, legacy Parquet filter + tuple, or `nil` + * `:batch_size` — accepted for API stability; reserved in 0.9 (batches + follow Parquet row-group sizing) - ## Example + ## Typical usage - {:ok, dataset} = ExArrow.Dataset.open(root, partitioning: {:hive, schema: [...]}) - {:ok, scanner} = ExArrow.Dataset.scanner(dataset, - columns: ["id"], - filter: ExArrow.Compute.Expression.gte( - ExArrow.Compute.Expression.field("year"), - ExArrow.Compute.Expression.scalar(2026) + alias ExArrow.Compute.Expression, as: E + + {:ok, dataset} = + ExArrow.Dataset.open("/data/events", + partitioning: {:hive, schema: [{"year", :int32}, {"month", :int32}]} + ) + + {:ok, scanner} = + ExArrow.Dataset.scanner(dataset, + columns: ["id", "amount"], + filter: + E.and_( + E.gte(E.field("year"), E.scalar(2026)), + E.gt(E.field("amount"), E.scalar(0.0)) + ) ) - ) + + # Optional: preview how many fragments survive partition prune (no IO) + preview = ExArrow.Scanner.stats(scanner) + {:ok, stream} = ExArrow.Scanner.to_stream(scanner) batches = Enum.to_list(stream) - ExArrow.Stream.close(stream) + stats = ExArrow.Scanner.stats(stream) + :ok = ExArrow.Stream.close(stream) + + Early `Enum.take/2` does not open later fragments. Call + `ExArrow.Stream.close/1` when abandoning a partially consumed scan from a + long-lived process. """ alias ExArrow.Compute @@ -54,6 +76,20 @@ defmodule ExArrow.Scanner do @enforce_keys [:dataset, :columns, :filter, :batch_size, :partition_keys] defstruct [:dataset, :columns, :filter, :batch_size, :partition_keys] + @typedoc """ + Cumulative scan statistics. + + ## Keys + + * `:fragments_discovered` — fragments on the Dataset before prune + * `:fragments_pruned_partition` — dropped by Hive-key evaluation + * `:fragments_selected` — kept after partition prune + * `:fragments_scanned` — actually opened during `to_stream/1` + * `:row_groups_selected` / `:row_groups_skipped` — sums of + `Parquet.Reader.read_stats/1` across opened Parquet fragments + * `:rows_emitted` — rows yielded after residual filter (empty batches + are skipped and do not count) + """ @type stats :: %{ fragments_discovered: non_neg_integer(), fragments_pruned_partition: non_neg_integer(), @@ -64,6 +100,18 @@ defmodule ExArrow.Scanner do rows_emitted: non_neg_integer() } + @typedoc """ + A lazy scan plan over a Dataset. + + ## Fields + + * `:dataset` — source `ExArrow.Dataset.t()` + * `:columns` — projection list, or `nil` for all columns + * `:filter` — Expression, legacy filter tuple, or `nil` + * `:batch_size` — reserved positive integer, or `nil` + * `:partition_keys` — Hive column names used when compiling filters + (derived from the dataset partitioning schema) + """ @type t :: %__MODULE__{ dataset: Dataset.t(), columns: [String.t()] | nil, @@ -73,7 +121,29 @@ defmodule ExArrow.Scanner do } @doc """ - Build a lazy scanner over `dataset`. Performs no IO. + Build a lazy scanner over `dataset`. + + Performs **no IO**. Prefer `ExArrow.Dataset.scanner/2`, which delegates here. + + ## Parameters + + * `dataset` — an `ExArrow.Dataset.t()` + * `opts` — keyword list: + + * `:columns` — non-empty `[String.t()]` to project, or omit/`nil` + * `:filter` — `Expression.t()`, legacy tuple, or omit/`nil` + * `:batch_size` — positive integer (reserved; validated only) + + ## Returns + + * `{:ok, scanner}` when options validate against the dataset schema + (Expression fields may include Hive partition columns) + * `{:error, message}` for bad options or filter validation failures + + ## Examples + + {:ok, scanner} = ExArrow.Scanner.new(dataset, columns: ["id"]) + {:ok, scanner} = ExArrow.Scanner.new(dataset, filter: {:gt, "amount", 0.0}) """ @spec new(Dataset.t(), keyword()) :: {:ok, t()} | {:error, String.t()} def new(dataset, opts \\ []) @@ -97,8 +167,33 @@ defmodule ExArrow.Scanner do def new(_, _), do: {:error, "scanner requires an ExArrow.Dataset"} @doc """ - Start scanning: partition-prune, then return an `ExArrow.Stream` with - `backend: :dataset`. + Start scanning: partition-prune, then return an `ExArrow.Stream`. + + The stream has `backend: :dataset` and implements `Enumerable`. Fragments + open lazily; EOF is idempotent (`next/1` returns `nil` forever after + exhaustion). + + ## Parameters + + * `scanner` — from `new/2` / `Dataset.scanner/2` + + ## Returns + + * `{:ok, stream}` — Agent-backed stream; call `ExArrow.Stream.close/1` + when done or when abandoning early + * `{:error, message}` — filter compile / Parquet opts validation failure + + ## Examples + + {:ok, stream} = ExArrow.Scanner.to_stream(scanner) + batch = ExArrow.Stream.next(stream) + batches = Enum.to_list(stream) + :ok = ExArrow.Stream.close(stream) + + Telemetry: emits `[:ex_arrow, :dataset, :scan, :start]` here and + `[:ex_arrow, :dataset, :scan, :stop]` when the scan finishes or is closed. + Per-batch events use `[:ex_arrow, :stream, :batch]` with + `source: {:dataset, current_fragment_path}`. """ @spec to_stream(t()) :: {:ok, Stream.t()} | {:error, String.t()} def to_stream(%__MODULE__{} = scanner) do @@ -159,10 +254,31 @@ defmodule ExArrow.Scanner do end @doc """ - Scan statistics. + Return scan statistics. + + ## Parameters + + * `scanner_or_stream` — either: + + * an `ExArrow.Scanner.t()` — **preview** after partition prune only + (`fragments_scanned`, row-group, and `rows_emitted` are `0`) + * an `ExArrow.Stream.t()` with `backend: :dataset` — live or post-scan + aggregates from the Agent + + ## Returns - Pass the scanner for partition-prune preview (no row-group / row counts yet), - or the `:dataset` stream for live / post-scan aggregates. + A `t:stats/0` map. Raises `ArgumentError` for other arguments. + + ## Examples + + preview = ExArrow.Scanner.stats(scanner) + preview.fragments_pruned_partition + + {:ok, stream} = ExArrow.Scanner.to_stream(scanner) + _ = Enum.to_list(stream) + stats = ExArrow.Scanner.stats(stream) + stats.row_groups_skipped + stats.rows_emitted """ @spec stats(t() | Stream.t()) :: stats() def stats(%__MODULE__{} = scanner) do diff --git a/test/ex_arrow/compute/expression_test.exs b/test/ex_arrow/compute/expression_test.exs index 70db781..90585db 100644 --- a/test/ex_arrow/compute/expression_test.exs +++ b/test/ex_arrow/compute/expression_test.exs @@ -1,6 +1,8 @@ defmodule ExArrow.Compute.ExpressionTest do use ExUnit.Case, async: true + doctest ExArrow.Compute.Expression + alias ExArrow.Compute.Expression, as: E alias ExArrow.Parquet.Opts alias ExArrow.RecordBatch diff --git a/test/ex_arrow/file_system_test.exs b/test/ex_arrow/file_system_test.exs index 5c5e629..1f2fc41 100644 --- a/test/ex_arrow/file_system_test.exs +++ b/test/ex_arrow/file_system_test.exs @@ -1,6 +1,9 @@ defmodule ExArrow.FileSystemTest do use ExUnit.Case, async: true + doctest ExArrow.FileSystem.Local + doctest ExArrow.FileSystem.Memory + alias ExArrow.FileSystem alias ExArrow.FileSystem.Local alias ExArrow.FileSystem.Memory From 3ffcda6946f103f96e2dc08929cff1b955d87b11 Mon Sep 17 00:00:00 2001 From: thanos Date: Sun, 13 Sep 2026 11:26:42 -0400 Subject: [PATCH 10/11] =?UTF-8?q?Fix=20v0.9.0=20review=20findings=20for=20?= =?UTF-8?q?Dataset/Scanner=20correctness.=20Guard=20Int=E2=86=92Float64=20?= =?UTF-8?q?residual=20filters,=20compare=20fragment=20schemas=20by=20name?= =?UTF-8?q?=20and=20type,=20return=20{:error,=5F}=20from=20Scanner.stats?= =?UTF-8?q?=20after=20close,=20and=20remove=20O(n=C2=B2)=20fragment=20list?= =?UTF-8?q?=20patterns=20plus=20related=20test=20and=20doc=20gaps.?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- lib/ex_arrow/compute/expression.ex | 4 ++ lib/ex_arrow/dataset.ex | 56 +++++++++------ lib/ex_arrow/dataset/fragment.ex | 6 +- lib/ex_arrow/file_system/local.ex | 2 +- lib/ex_arrow/parquet/reader.ex | 6 ++ lib/ex_arrow/record_batch.ex | 6 +- lib/ex_arrow/scanner.ex | 84 ++++++++++------------ native/ex_arrow_native/src/compute.rs | 10 ++- test/ex_arrow/compute_filter_expr_test.exs | 70 ++++++++++++++++++ test/ex_arrow/dataset_test.exs | 2 +- test/ex_arrow/scanner_test.exs | 78 ++++++++++++++++++++ 11 files changed, 248 insertions(+), 76 deletions(-) diff --git a/lib/ex_arrow/compute/expression.ex b/lib/ex_arrow/compute/expression.ex index bd9e5a5..9ee3e2d 100644 --- a/lib/ex_arrow/compute/expression.ex +++ b/lib/ex_arrow/compute/expression.ex @@ -325,6 +325,10 @@ defmodule ExArrow.Compute.Expression do AND may push one side and residual the other. OR is pushed only when both sides are fully pushable; otherwise the whole OR is residual. + Note: `ExArrow.Parquet.Reader` `:filters` accepts an Expression only when + it is fully pushable (`residual` is `nil`). For mixed pushable/residual + filters, use `ExArrow.Dataset.scanner/2`. + ## Examples iex> alias ExArrow.Compute.Expression, as: E diff --git a/lib/ex_arrow/dataset.ex b/lib/ex_arrow/dataset.ex index 9a4fe5f..318a597 100644 --- a/lib/ex_arrow/dataset.ex +++ b/lib/ex_arrow/dataset.ex @@ -440,12 +440,18 @@ defmodule ExArrow.Dataset do end defp attach_sizes(paths, filesystem) do - Enum.reduce_while(paths, {:ok, []}, fn path, {:ok, acc} -> - case lookup_size(filesystem, path) do - {:ok, size} -> {:cont, {:ok, acc ++ [{path, size}]}} - {:error, _} = err -> {:halt, err} - end - end) + result = + Enum.reduce_while(paths, {:ok, []}, fn path, {:ok, acc} -> + case lookup_size(filesystem, path) do + {:ok, size} -> {:cont, {:ok, [{path, size} | acc]}} + {:error, _} = err -> {:halt, err} + end + end) + + case result do + {:ok, acc} -> {:ok, Enum.reverse(acc)} + {:error, _} = err -> err + end end defp lookup_size(filesystem, path) do @@ -536,22 +542,28 @@ defmodule ExArrow.Dataset do end defp build_fragments(sized_paths, root, format, {:hive, schema}) do - Enum.reduce_while(sized_paths, {:ok, []}, fn {path, size}, {:ok, acc} -> - case Hive.parse_path(path, root, schema) do - {:ok, values} -> - frag = %Fragment{ - path: path, - format: format, - partition_values: values, - size: size - } - - {:cont, {:ok, acc ++ [frag]}} - - {:error, _} = err -> - {:halt, err} - end - end) + result = + Enum.reduce_while(sized_paths, {:ok, []}, fn {path, size}, {:ok, acc} -> + case Hive.parse_path(path, root, schema) do + {:ok, values} -> + frag = %Fragment{ + path: path, + format: format, + partition_values: values, + size: size + } + + {:cont, {:ok, [frag | acc]}} + + {:error, _} = err -> + {:halt, err} + end + end) + + case result do + {:ok, acc} -> {:ok, Enum.reverse(acc)} + {:error, _} = err -> err + end end # --- schema --------------------------------------------------------------- diff --git a/lib/ex_arrow/dataset/fragment.ex b/lib/ex_arrow/dataset/fragment.ex index 6a65fb8..f967f6c 100644 --- a/lib/ex_arrow/dataset/fragment.ex +++ b/lib/ex_arrow/dataset/fragment.ex @@ -41,8 +41,10 @@ defmodule ExArrow.Dataset.Fragment do * `:format` — `:parquet` or `:ipc` * `:partition_values` — map of Hive column name => coerced term (empty map when partitioning is `:none`) - * `:size` — file size in bytes as reported by the filesystem (may be `0` - when unknown) + * `:size` — file size in bytes as reported by the filesystem. Current + backends (`Local`, `Memory`) always resolve a real size; a future + object-store backend may report `0` when size is unavailable without + an extra round-trip """ @type t :: %__MODULE__{ path: String.t(), diff --git a/lib/ex_arrow/file_system/local.ex b/lib/ex_arrow/file_system/local.ex index d9e0223..9fff0a3 100644 --- a/lib/ex_arrow/file_system/local.ex +++ b/lib/ex_arrow/file_system/local.ex @@ -121,7 +121,7 @@ defmodule ExArrow.FileSystem.Local do if recursive do case collect_dir(full, true, ignore_hidden) do - {:ok, child} -> {:ok, acc ++ [entry | child]} + {:ok, child} -> {:ok, Enum.reverse(child, [entry | acc])} {:error, _} = err -> err end else diff --git a/lib/ex_arrow/parquet/reader.ex b/lib/ex_arrow/parquet/reader.ex index 48939dd..22f25fb 100644 --- a/lib/ex_arrow/parquet/reader.ex +++ b/lib/ex_arrow/parquet/reader.ex @@ -27,6 +27,12 @@ defmodule ExArrow.Parquet.Reader do {:and, [{:gte, "id", 10}, {:lt, "id", 100}]} {:or, [{:eq, "name", "alice"}, {:eq, "name", "bob"}]} + An `ExArrow.Compute.Expression` is also accepted here, but it must be + **fully** Parquet-pushable (`Expression.to_parquet_filters/1` returns a + tuple with `residual == nil`). Filters that need residual evaluation + (for example `not_/1`, temporal scalars, or mixed partition keys) must + use `ExArrow.Dataset.scanner/2` instead. + Supported comparison ops: `:eq`, `:ne`, `:gt`, `:gte`, `:lt`, `:lte`. Values may be integers, floats, UTF-8 strings, or booleans. diff --git a/lib/ex_arrow/record_batch.ex b/lib/ex_arrow/record_batch.ex index 44e851c..d958f95 100644 --- a/lib/ex_arrow/record_batch.ex +++ b/lib/ex_arrow/record_batch.ex @@ -87,9 +87,9 @@ defmodule ExArrow.RecordBatch do ## Nullability - `from_columns/4` and `from_lists/1` produce non-nullable columns - (`Field.nullable = false`). `from_lists/1` rejects `nil` cells in 0.9; - null-bitmap support arrives with the core-model release. + `from_columns/4`, `from_lists/1`, and `from_map/1` produce non-nullable + columns (`Field.nullable = false`). `from_lists/1` and `from_map/1` reject + `nil` cells in 0.9; null-bitmap support arrives with the core-model release. """ alias ExArrow.Native alias ExArrow.Schema diff --git a/lib/ex_arrow/scanner.ex b/lib/ex_arrow/scanner.ex index f82d33e..b84c40c 100644 --- a/lib/ex_arrow/scanner.ex +++ b/lib/ex_arrow/scanner.ex @@ -227,8 +227,7 @@ defmodule ExArrow.Scanner do {:ok, agent} = Agent.start_link(fn -> %{ - fragments: selected, - index: 0, + remaining: selected, format: scanner.dataset.format, columns: scanner.columns, read_opts: read_opts, @@ -236,7 +235,7 @@ defmodule ExArrow.Scanner do current_inner: nil, current_path: nil, current_fragment: nil, - schema_names: nil, + schema_sig: nil, opened_paths: [], stats: stats, scan_meta: meta, @@ -267,7 +266,9 @@ defmodule ExArrow.Scanner do ## Returns - A `t:stats/0` map. Raises `ArgumentError` for other arguments. + A `t:stats/0` map for an open scanner or live stream. For a `:dataset` + stream whose Agent has already been stopped via `ExArrow.Stream.close/1`, + returns `{:error, "stream is closed"}` instead of raising. ## Examples @@ -279,8 +280,10 @@ defmodule ExArrow.Scanner do stats = ExArrow.Scanner.stats(stream) stats.row_groups_skipped stats.rows_emitted + :ok = ExArrow.Stream.close(stream) + {:error, "stream is closed"} = ExArrow.Scanner.stats(stream) """ - @spec stats(t() | Stream.t()) :: stats() + @spec stats(t() | Stream.t()) :: stats() | {:error, String.t()} def stats(%__MODULE__{} = scanner) do {selected, pruned} = Partition.select_fragments(scanner.dataset.fragments, scanner.filter) @@ -297,7 +300,11 @@ defmodule ExArrow.Scanner do end def stats(%Stream{backend: :dataset, resource: agent}) do - Agent.get(agent, & &1.stats) + if Process.alive?(agent) do + Agent.get(agent, & &1.stats) + else + {:error, "stream is closed"} + end end def stats(_), do: raise(ArgumentError, "Scanner.stats/1 expects a Scanner or dataset Stream") @@ -465,11 +472,11 @@ defmodule ExArrow.Scanner do state.current_inner != nil -> {{:ok, :open}, state} - state.index >= length(state.fragments) -> + state.remaining == [] -> {:exhausted, state} true -> - frag = Enum.at(state.fragments, state.index) + [frag | _] = state.remaining open_fragment(state, frag) end end) @@ -513,16 +520,7 @@ defmodule ExArrow.Scanner do {{:error, prefix_path(path, msg)}, state} {:ok, sch} -> - names = Schema.field_names(sch) - - names = - if is_list(state.columns) do - state.columns - else - names - end - - case accept_schema_names(state, path, names) do + case accept_schema_sig(state, path, schema_signature(sch)) do {:error, _} = err -> {err, state} @@ -550,21 +548,27 @@ defmodule ExArrow.Scanner do {:error, prefix_path(path, msg)} {:ok, sch} -> - accept_schema_names(state, path, Schema.field_names(sch)) + accept_schema_sig(state, path, schema_signature(sch)) end end - defp accept_schema_names(state, path, names) do + defp schema_signature(schema) do + schema + |> Schema.fields() + |> Enum.map(fn f -> {f.name, f.type} end) + end + + defp accept_schema_sig(state, path, sig) do cond do - is_nil(state.schema_names) -> - {:ok, %{state | schema_names: names}} + is_nil(state.schema_sig) -> + {:ok, %{state | schema_sig: sig}} - state.schema_names == names -> + state.schema_sig == sig -> {:ok, state} true -> {:error, - "schema mismatch in #{path}: expected columns #{inspect(state.schema_names)}, got #{inspect(names)}"} + "schema mismatch in #{path}: expected #{inspect(state.schema_sig)}, got #{inspect(sig)}"} end end @@ -576,32 +580,20 @@ defmodule ExArrow.Scanner do | row_groups_selected: stats.row_groups_selected + Map.get(rg, :row_groups_selected, 0), row_groups_skipped: stats.row_groups_skipped + Map.get(rg, :row_groups_skipped, 0) } - rescue - _ -> stats end defp advance(agent) do Agent.get_and_update(agent, fn state -> - next_index = state.index + 1 - - if next_index >= length(state.fragments) do - {:done, - %{ - state - | index: next_index, - current_inner: nil, - current_path: nil, - current_fragment: nil - }} - else - {:ok, - %{ - state - | index: next_index, - current_inner: nil, - current_path: nil, - current_fragment: nil - }} + case state.remaining do + [] -> + {:done, clear_current(state)} + + [_opened | rest] -> + if rest == [] do + {:done, clear_current(%{state | remaining: rest})} + else + {:ok, clear_current(%{state | remaining: rest})} + end end end) end diff --git a/native/ex_arrow_native/src/compute.rs b/native/ex_arrow_native/src/compute.rs index 50ba86a..ebf0761 100644 --- a/native/ex_arrow_native/src/compute.rs +++ b/native/ex_arrow_native/src/compute.rs @@ -489,7 +489,15 @@ fn make_scalar_array(value: &ScalarValue, data_type: &DataType) -> Result { - Ok(Arc::new(Float64Array::from(vec![*v as f64])) as ArrayRef) + let f = *v as f64; + // Use i128 on the round-trip so saturating f64→i64 casts (e.g. + // i64::MAX) cannot mask precision loss. + if (f as i128) != i128::from(*v) { + return Err(format!( + "filter value {v} is not exactly representable as Float64" + )); + } + Ok(Arc::new(Float64Array::from(vec![f])) as ArrayRef) } (ScalarValue::Float(v), DataType::Float32) => { let f = *v as f32; diff --git a/test/ex_arrow/compute_filter_expr_test.exs b/test/ex_arrow/compute_filter_expr_test.exs index 4bca3eb..18ea6fe 100644 --- a/test/ex_arrow/compute_filter_expr_test.exs +++ b/test/ex_arrow/compute_filter_expr_test.exs @@ -145,4 +145,74 @@ defmodule ExArrow.ComputeFilterExprTest do assert msg =~ "out of range" end + + @tag :nif + test "eq against Float64 column errors on a non-representable large integer" do + assert {:ok, batch} = RecordBatch.from_lists([{"amount", :f64, [1.0, 2.0]}]) + + assert {:error, msg} = + Compute.filter( + batch, + E.eq(E.field("amount"), E.scalar(9_223_372_036_854_775_807)) + ) + + assert msg =~ "not exactly representable" + end + + @tag :nif + test "eq against Float64 column succeeds for an exactly representable integer" do + # 2^52 is exactly representable as Float64. + n = 4_503_599_627_370_496 + + assert {:ok, batch} = + RecordBatch.from_lists([{"amount", :f64, [n * 1.0, 1.0]}]) + + assert {:ok, filtered} = + Compute.filter(batch, E.eq(E.field("amount"), E.scalar(n))) + + assert RecordBatch.num_rows(filtered) == 1 + end + + @tag :nif + test "eq against Float32 column errors on non-representable float and succeeds for int" do + assert {:ok, batch} = RecordBatch.from_lists([{"x", :f32, [1.0, 2.0]}]) + + assert {:error, msg} = + Compute.filter(batch, E.eq(E.field("x"), E.scalar(1.0e40))) + + assert msg =~ "not exactly representable" + + assert {:ok, filtered} = Compute.filter(batch, E.eq(E.field("x"), E.scalar(2))) + assert RecordBatch.num_rows(filtered) == 1 + end + + @tag :nif + test "filters Int8/UInt8/Date64 and TimestampMillis columns" do + assert {:ok, i8} = RecordBatch.from_lists([{"x", :s8, [1, 2, 3]}]) + assert {:ok, f} = Compute.filter(i8, E.gt(E.field("x"), E.scalar(1))) + assert RecordBatch.num_rows(f) == 2 + + assert {:ok, u8} = RecordBatch.from_lists([{"x", :u8, [1, 2, 3]}]) + assert {:ok, f2} = Compute.filter(u8, E.eq(E.field("x"), E.scalar(2))) + assert RecordBatch.num_rows(f2) == 1 + + millis = [ + Date.diff(~D[2025-01-01], ~D[1970-01-01]) * 86_400_000, + Date.diff(~D[2026-01-01], ~D[1970-01-01]) * 86_400_000 + ] + + assert {:ok, d64} = RecordBatch.from_lists([{"d", :date64, millis}]) + assert {:ok, f3} = Compute.filter(d64, E.gte(E.field("d"), E.scalar(~D[2026-01-01]))) + assert RecordBatch.num_rows(f3) == 1 + + assert {:ok, ts} = + RecordBatch.from_lists([ + {"t", :timestamp_millis, [1_000, 2_000, 3_000]} + ]) + + assert {:ok, f4} = + Compute.filter(ts, E.gte(E.field("t"), E.scalar(2_000))) + + assert RecordBatch.num_rows(f4) == 2 + end end diff --git a/test/ex_arrow/dataset_test.exs b/test/ex_arrow/dataset_test.exs index 31b563a..9acadd9 100644 --- a/test/ex_arrow/dataset_test.exs +++ b/test/ex_arrow/dataset_test.exs @@ -143,7 +143,7 @@ defmodule ExArrow.DatasetTest do frag = Enum.find(fragments, &(&1.path == Path.expand(p1))) assert {:ok, meta} = Fragment.metadata(frag) assert meta.num_rows == 2 - assert meta.num_row_groups >= 1 + assert meta.num_row_groups == 1 end @tag :tmp_dir diff --git a/test/ex_arrow/scanner_test.exs b/test/ex_arrow/scanner_test.exs index 69f1af5..787880e 100644 --- a/test/ex_arrow/scanner_test.exs +++ b/test/ex_arrow/scanner_test.exs @@ -237,6 +237,63 @@ defmodule ExArrow.ScannerTest do Stream.close(stream) _ = p1 end + + @tag :tmp_dir + @tag :nif + test "type mismatch across fragments with same column names is rejected", %{tmp_dir: dir} do + assert {:ok, a} = + RecordBatch.from_lists([{"id", :s64, [1, 2]}, {"amount", :f64, [1.0, 2.0]}]) + + assert {:ok, b} = + RecordBatch.from_lists([{"id", :s64, [3, 4]}, {"amount", :utf8, ["x", "y"]}]) + + :ok = Parquet.Writer.to_file(Path.join(dir, "a.parquet"), RecordBatch.schema(a), [a]) + :ok = Parquet.Writer.to_file(Path.join(dir, "b.parquet"), RecordBatch.schema(b), [b]) + + assert {:ok, dataset} = Dataset.open(dir) + assert {:ok, scanner} = Dataset.scanner(dataset) + assert {:ok, stream} = Scanner.to_stream(scanner) + + assert %RecordBatch{} = Stream.next(stream) + assert {:error, msg} = Stream.next(stream) + assert msg =~ "schema mismatch" + assert msg =~ "amount" + Stream.close(stream) + end + + @tag :tmp_dir + @tag :nif + test "stats/1 after close returns an error instead of crashing", %{tmp_dir: dir} do + {dataset, _} = hive_dataset!(Path.join(dir, "events")) + assert {:ok, scanner} = Dataset.scanner(dataset) + assert {:ok, stream} = Scanner.to_stream(scanner) + _ = Enum.to_list(stream) + assert %{rows_emitted: _} = Scanner.stats(stream) + assert :ok = Stream.close(stream) + assert {:error, "stream is closed"} = Scanner.stats(stream) + end + + @tag :tmp_dir + @tag :nif + test "IPC type mismatch across fragments is rejected", %{tmp_dir: dir} do + assert {:ok, a} = + RecordBatch.from_lists([{"id", :s64, [1]}, {"v", :f64, [1.0]}]) + + assert {:ok, b} = + RecordBatch.from_lists([{"id", :s64, [2]}, {"v", :utf8, ["x"]}]) + + assert :ok = ExArrow.IPC.File.write(Path.join(dir, "a.arrow"), RecordBatch.schema(a), [a]) + assert :ok = ExArrow.IPC.File.write(Path.join(dir, "b.arrow"), RecordBatch.schema(b), [b]) + + assert {:ok, dataset} = Dataset.open(dir, format: :ipc) + assert {:ok, scanner} = Dataset.scanner(dataset) + assert {:ok, stream} = Scanner.to_stream(scanner) + + assert %RecordBatch{} = Stream.next(stream) + assert {:error, msg} = Stream.next(stream) + assert msg =~ "schema mismatch" + Stream.close(stream) + end end describe "Compile / Partition helpers" do @@ -304,6 +361,27 @@ defmodule ExArrow.ScannerTest do assert to_string(bound) =~ "field(\"id\")" assert Compile.bind_partitions(nil, %{}) == nil end + + test "compile partition-only expression drops residual and pushed" do + alias ExArrow.Scanner.Compile + + expr = E.eq(E.field("year"), E.scalar(2026)) + assert {:ok, {nil, nil}} = Compile.compile(expr, ["year", "month"]) + + and_only = + E.and_( + E.eq(E.field("year"), E.scalar(2026)), + E.eq(E.field("month"), E.scalar(1)) + ) + + assert {:ok, {nil, nil}} = Compile.compile(and_only, ["year", "month"]) + + assert {:ok, {nil, nil}} = + Compile.compile({:and, [{:eq, "year", 2026}, {:eq, "month", 1}]}, [ + "year", + "month" + ]) + end end describe "validation" do From 06e7d2f698923db6838d3d92b72c165eadda5bd8 Mon Sep 17 00:00:00 2001 From: thanos Date: Sun, 13 Sep 2026 15:00:52 -0400 Subject: [PATCH 11/11] M9: prepare v0.9.0 release (Dataset & Scanner). Bump mix/crate/Livebook pins to 0.9.0, add CHANGELOG and release notes, refresh README/docs, and clear cargo clippy so the quality gate is green. --- CHANGELOG.md | 41 ++++++++++++++ README.md | 62 +++++++++++++++++++-- docs/RELEASE_NOTES_0.9.0.md | 81 ++++++++++++++++++++++++++++ docs/parquet_guide.md | 4 ++ docs/release_checklist.md | 5 +- livebook/00_quickstart-tester.livemd | 2 +- livebook/00_quickstart.exs | 2 +- livebook/00_quickstart.livemd | 2 +- livebook/01_ipc.livemd | 2 +- livebook/02_flight.livemd | 2 +- livebook/03_adbc.livemd | 2 +- livebook/04_adbc_integration.livemd | 2 +- livebook/05_parquet.livemd | 2 +- livebook/06_datasets.livemd | 2 +- livebook/README.md | 2 +- mix.exs | 4 +- native/ex_arrow_native/Cargo.lock | 2 +- native/ex_arrow_native/Cargo.toml | 2 +- native/ex_arrow_native/src/flight.rs | 12 ++--- native/ex_arrow_native/src/ipc.rs | 8 +-- 20 files changed, 211 insertions(+), 30 deletions(-) create mode 100644 docs/RELEASE_NOTES_0.9.0.md diff --git a/CHANGELOG.md b/CHANGELOG.md index a0cf297..653e87e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,6 +5,47 @@ All notable changes to this project will be documented in this file. The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/), and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). +## [0.9.0] - 2026-09-13 + +### Added + +- **`ExArrow.Dataset` / `ExArrow.Dataset.Fragment`**: discover Parquet or IPC + trees from a directory, single file, glob, or explicit path list via + `Dataset.open/2`. Hive partitioning parses typed keys from path segments; + each fragment carries `path`, `format`, `size`, and `partition_values`. + Schema is resolved from the first fragment footer without decoding pages. +- **`ExArrow.FileSystem`**: behaviour with `Local` (OS) and `Memory` + (in-process tree) backends for list/glob/exists used by Dataset discovery. +- **`ExArrow.Scanner`**: lazy scan plan from `Dataset.scanner/2` with + `:columns`, `:filter`, and `:batch_size`. `Scanner.to_stream/1` yields an + Agent-backed `:dataset` stream; `Scanner.stats/1` reports exact fragment + and row-group prune counts (preview before IO, or live after scan). Closed + streams return `{:error, "stream is closed"}` instead of exiting. +- **`ExArrow.Compute.Expression`**: analyzable filter AST with `field/1`, + `scalar/1`, comparisons, `and_/2`, `or_/2`, `not_/1`, plus `validate/2`, + `to_string/1`, and `to_parquet_filters/1` (pushable subset vs residual). + Temporal scalars (`Date`, `NaiveDateTime`, `DateTime`) are supported for + residual evaluation. +- **Residual `Compute.filter/2`**: new NIF evaluates an Expression to a + boolean mask and filters a `RecordBatch` (also via `Batch.filter/2`). + Integer→float casts reject values that are not exactly representable. +- **`RecordBatch.from_lists/1` and `from_map/1`**: ergonomic constructors + for non-nullable columnar batches from Elixir lists/maps. +- **Docs / fixtures**: Datasets guide (`guides/11_datasets.md`), Livebook + `06_datasets.livemd`, PyArrow Hive fixture with exact Scanner prune + stats, and `bench/dataset_scan_bench.exs` (labeled pushdown ladder). + +### Changed + +- **Native stack**: arrow-rs / parquet / arrow-flight **59.3.0** (from 56); + tonic **0.14**; adbc_core / adbc_driver_manager **0.24**. +- **Parquet `:filters`**: continues to accept the v0.8.0 tuple AST; an + `Expression` is accepted only when fully Parquet-pushable (use Scanner + when a residual remains). +- Cross-fragment Scanner schema checks compare **name and type** (not names + alone). Dataset fragment list building uses linear prepend/reverse + accumulation. + ## [0.8.0] - 2026-08-21 ### Added diff --git a/README.md b/README.md index 3aa835a..ad0bc0f 100644 --- a/README.md +++ b/README.md @@ -9,6 +9,12 @@ Native Apache Arrow for the BEAM: IPC streaming, Arrow Flight, Arrow Flight SQL, ADBC database bindings, and Arrow-native pipelines. Column data lives in Rust buffers; Elixir holds lightweight opaque handles. Precompiled NIFs for Linux, macOS, and Windows — no Rust required to use. +> **v0.9.0 — Dataset and Scanner.** Discover Hive-partitioned Parquet/IPC trees, +> filter with `ExArrow.Compute.Expression`, and scan through partition prune → +> row-group pushdown → residual `Compute.filter/2`. See +> [Datasets and scanners](#datasets-and-scanners) and the +> [Datasets guide](https://ex-arrow.hexdocs.pm/guides/11_datasets.html). +> > **v0.8.0 — Larger-than-memory Parquet.** Column/predicate/row-group pushdown, > write compression options, footer metadata, and multi-file directory streams. > See [Parquet: read and write](#parquet-read-and-write) and the @@ -31,6 +37,7 @@ Native Apache Arrow for the BEAM: IPC streaming, Arrow Flight, Arrow Flight SQL, - [Requirements](#requirements) - [Installation](#installation) - [Quick start](#quick-start) +- [What's changed in v0.9.0](#whats-changed-in-v090) - [What's changed in v0.8.0](#whats-changed-in-v080) - [What's changed in v0.7.0](#whats-changed-in-v070) - [Livebook tutorials](#livebook-tutorials) @@ -55,6 +62,7 @@ Native Apache Arrow for the BEAM: IPC streaming, Arrow Flight, Arrow Flight SQL, - [Shipped (v0.5.0)](#shipped-v050) - [Shipped (v0.7.0)](#shipped-v070) - [Shipped (v0.8.0)](#shipped-v080) + - [Shipped (v0.9.0)](#shipped-v090) - [FAQ](#faq) - [License](#license) @@ -215,7 +223,7 @@ Add the dependency: ```elixir def deps do - [{:ex_arrow, "~> 0.8"}] + [{:ex_arrow, "~> 0.9"}] end ``` @@ -243,11 +251,11 @@ For **path dependencies** in Livebook (`Mix.install`), open notebooks from is detected) or use the Hex package: ```elixir -Mix.install([{:ex_arrow, "~> 0.8.0"}, {:rustler, "~> 0.36", optional: true}]) +Mix.install([{:ex_arrow, "~> 0.9.0"}, {:rustler, "~> 0.36", optional: true}]) ``` Alternatively, use the published Hex package so the precompiled NIF is used -and no Rust is needed: `Mix.install([{:ex_arrow, "~> 0.8.0"}])`. +and no Rust is needed: `Mix.install([{:ex_arrow, "~> 0.9.0"}])`. --- @@ -329,6 +337,39 @@ batch = ExArrow.Stream.next(stream) --- +## What's changed in v0.9.0 + +v0.9.0 adds a **Dataset / Scanner** layer on top of v0.8.0 Parquet pushdown: +discover partitioned trees, compile analyzable Expression filters, and scan +with a three-level pushdown ladder (Hive partition prune → Parquet row-group +pushdown → residual `Compute.filter/2`). + +### Dataset, Scanner, Expression + +| API | Purpose | +|-----|---------| +| `ExArrow.Dataset.open/2` | Open a directory, file, glob, or path list; Hive `partition_values` | +| `ExArrow.Dataset.Fragment` | One file fragment with path, format, size, partition map | +| `ExArrow.Dataset.scanner/2` | Lazy scan plan: `:columns`, `:filter`, `:batch_size` | +| `ExArrow.Scanner.to_stream/1` | Agent-backed `:dataset` stream; `Stream.close/1` for early abandon | +| `ExArrow.Scanner.stats/1` | Exact fragment / row-group prune counts (preview or live) | +| `ExArrow.Compute.Expression` | Builders, `validate/2`, `to_parquet_filters/1` (pushable vs residual) | +| `ExArrow.Compute.filter/2` | Residual evaluation of an Expression against a RecordBatch | +| `ExArrow.FileSystem` | `Local` and `Memory` discovery backends | +| `RecordBatch.from_lists/1`, `from_map/1` | Ergonomic batch construction | + +Native stack: arrow-rs / parquet / arrow-flight **59.3.0** (from 56). + +### Docs, Livebook, bench + +- Datasets Livebook (`livebook/06_datasets.livemd`) and + [Datasets guide](guides/11_datasets.md) +- Pushdown-ladder timing helper: `bench/dataset_scan_bench.exs` + +See [CHANGELOG.md](CHANGELOG.md) for the full list. + +--- + ## What's changed in v0.8.0 v0.8.0 focuses on **larger-than-memory Parquet**: read only the columns, row @@ -438,7 +479,7 @@ Interactive notebooks (open in [Livebook](https://livebook.dev)): - **[05 Parquet](livebook/05_parquet.livemd)** — Pushdown reads, compressed writes, multi-file directories, PyArrow side-by-side. - **[06 Datasets](livebook/06_datasets.livemd)** — Hive Dataset open, Expression scanner, prune stats, PyArrow `dataset` side-by-side. -See [livebook/README.md](livebook/README.md) for run instructions. Notebooks use Hex `~> 0.8.0` by default; opening from `livebook/` in a clone builds from source. +See [livebook/README.md](livebook/README.md) for run instructions. Notebooks use Hex `~> 0.9.0` by default; opening from `livebook/` in a clone builds from source. --- @@ -1118,6 +1159,7 @@ HTML reports are written to `bench/output/` (gitignored). | `v070_stream_flow_pipeline_bench.exs` | Parquet/IPC stream drains, Flow execution, Pipeline map_batches + write_parquet at 1K/100K/1M rows | | `v070_record_batch_vs_maps_bench.exs` | Arrow `RecordBatch` vs `list(map())` for build, transform, and drain at 1K/100K/1M rows | | `parquet_pushdown_bench.exs` | Pushdown filter vs full read + project (v0.8.0 rough timing helper) | +| `dataset_scan_bench.exs` | Dataset pushdown ladder: full / project / partition / row-group / residual (v0.9.0) | ### Published results @@ -1340,6 +1382,18 @@ welcome for any of them. - **Docs / CI** — Parquet Livebook, Mix.install smoke, Livebook pin checks; optional arrow-testing corpus (local / workflow_dispatch). +### Shipped (v0.9.0) + +- **Dataset / Fragment / Hive discovery** — `Dataset.open/2` over Local or + Memory filesystems; typed Hive partition values; schema from footer. +- **Scanner** — partition prune, Parquet pushdown, residual Expression filter; + `:dataset` stream backend; exact `Scanner.stats/1`. +- **`ExArrow.Compute.Expression`** — builders, schema validate, pushable vs + residual compile; `Compute.filter/2` residual NIF. +- **`RecordBatch.from_lists/1` / `from_map/1`** — list/map batch construction. +- **arrow-rs 59.3.0** — coordinated parquet / Flight / ADBC companion bumps. +- **Docs** — Datasets guide, Livebook 06, dataset scan bench. + ### Longer-term - **Streaming writes to Delta Lake** — sink for data pipeline nodes. diff --git a/docs/RELEASE_NOTES_0.9.0.md b/docs/RELEASE_NOTES_0.9.0.md new file mode 100644 index 0000000..0908e05 --- /dev/null +++ b/docs/RELEASE_NOTES_0.9.0.md @@ -0,0 +1,81 @@ +# ExArrow 0.9.0 — Release notes + +**Release date:** 2026-09-13 +**Package:** [Hex](https://hex.pm/packages/ex_arrow) | **Docs:** [ex-arrow.hexdocs.pm](https://ex-arrow.hexdocs.pm) | **Source:** [GitHub](https://github.com/thanos/ex_arrow) + +--- + +## Summary + +ExArrow 0.9.0 finishes the Dataset half of the original roadmap: discover +Hive-partitioned Parquet (and IPC) trees, filter with a first-class +`Compute.Expression` AST, and scan through a three-level pushdown ladder — +partition prune → Parquet row-group pushdown → residual `Compute.filter/2`. + +v0.8.0 Parquet pushdown on a single file or path list remains; Dataset +generalises that into discovered, partitioned layouts with exact +`Scanner.stats/1` for verifying what was pruned. + +**Requirements:** Elixir ~> 1.14 (CI covers 1.18/OTP 27, 1.19/OTP 28, +1.20/OTP 29). Native stack: arrow-rs / parquet / Flight **59.3.0**. + +--- + +## What's included + +**Dataset discovery** +- `ExArrow.Dataset.open/2` over a directory, file, glob, or path list. +- Hive partitioning with typed keys (`{:hive, schema: [...]}`). +- `ExArrow.Dataset.Fragment` with path, format, size, `partition_values`. +- `ExArrow.FileSystem.Local` and `Memory` backends. + +**Scanner** +- `Dataset.scanner/2` + `Scanner.to_stream/1` (`:dataset` stream backend). +- Options: `:columns`, `:filter` (`Expression` or legacy tuple), `:batch_size`. +- `Scanner.stats/1` — exact fragment / row-group counts; closed stream → + `{:error, "stream is closed"}`. + +**Expressions and residual filter** +- `ExArrow.Compute.Expression` builders, `validate/2`, `to_parquet_filters/1`. +- `Compute.filter/2` / `Batch.filter/2` evaluate residuals after decode. +- Int→Float64/Float32 casts reject non-representable literals. + +**Ergonomics** +- `RecordBatch.from_lists/1`, `from_map/1`. + +**Docs** +- [Datasets guide](../guides/11_datasets.md), `livebook/06_datasets.livemd`, + `bench/dataset_scan_bench.exs`, checked-in PyArrow Hive fixture. + +--- + +## Installation + +```elixir +def deps do + [{:ex_arrow, "~> 0.9.0"}] +end +``` + +Precompiled NIFs download from GitHub releases after the `v0.9.0` tag assets +are published. To build from source: `EX_ARROW_BUILD=1 mix compile`. + +--- + +## Out of scope (deferred) + +Dataset writes, object-store filesystems, CSV/JSON fragments, `expr do` +macro sugar, page-level Parquet filtering, full compute catalog / aggregates, +and the core Arrow model release. + +--- + +## Changelog + +See [CHANGELOG.md](https://github.com/thanos/ex_arrow/blob/v0.9.0/CHANGELOG.md) for the full 0.9.0 entry. + +--- + +## Feedback + +Issues and discussions: [GitHub Issues](https://github.com/thanos/ex_arrow/issues). diff --git a/docs/parquet_guide.md b/docs/parquet_guide.md index 8e4ba0a..dc5387d 100644 --- a/docs/parquet_guide.md +++ b/docs/parquet_guide.md @@ -10,6 +10,10 @@ schema + batches pattern on the write side. > `ExArrow.Parquet.Metadata`, and multi-file streams > (`from_parquet_files/2`, `from_parquet_dir/2`). Preferred entry point: > `ExArrow.Stream.from_parquet/2`. +> +> **v0.9.0**: for Hive-partitioned trees and Expression filters that need a +> residual, prefer `ExArrow.Dataset` / `ExArrow.Scanner` (see the +> [Datasets guide](../guides/11_datasets.md)). --- diff --git a/docs/release_checklist.md b/docs/release_checklist.md index fd18acf..6286b5f 100644 --- a/docs/release_checklist.md +++ b/docs/release_checklist.md @@ -100,9 +100,10 @@ The `package` in `mix.exs` already includes `checksum-*.exs`, so this file will ## Compatibility notes - **Apache Arrow / Flight** - - `arrow`, `arrow-ipc`, `arrow-schema`, `arrow-array`, `arrow-flight`: version **56**. + - `arrow`, `arrow-ipc`, `arrow-schema`, `arrow-array`, `arrow-flight`, + `parquet`: version **59** (v0.9.0; previously 56). - **ADBC** - - `adbc_core` and `adbc_driver_manager`: version **0.22**. + - `adbc_core` and `adbc_driver_manager`: version **0.24**. - **BEAM** - Designed for Elixir `~> 1.14`; CI exercises Elixir 1.18/OTP 27, 1.19/OTP 28, and 1.20/OTP 29. diff --git a/livebook/00_quickstart-tester.livemd b/livebook/00_quickstart-tester.livemd index edc2af4..3bfad8a 100644 --- a/livebook/00_quickstart-tester.livemd +++ b/livebook/00_quickstart-tester.livemd @@ -39,7 +39,7 @@ local? = File.exists?(Path.join(__DIR__, "../native/ex_arrow_native/Cargo.toml") } else { - {:ex_arrow, "~> 0.8.0"}, + {:ex_arrow, "~> 0.9.0"}, [], [adbc: [drivers: [:sqlite]]] } diff --git a/livebook/00_quickstart.exs b/livebook/00_quickstart.exs index 0ccd5a9..d14a46b 100644 --- a/livebook/00_quickstart.exs +++ b/livebook/00_quickstart.exs @@ -25,7 +25,7 @@ local? = File.exists?(Path.join(__DIR__, "../native/ex_arrow_native/Cargo.toml") } else { - {:ex_arrow, "~> 0.8.0"}, + {:ex_arrow, "~> 0.9.0"}, [], [adbc: [drivers: [:sqlite]]] } diff --git a/livebook/00_quickstart.livemd b/livebook/00_quickstart.livemd index 2fc894d..8370531 100644 --- a/livebook/00_quickstart.livemd +++ b/livebook/00_quickstart.livemd @@ -35,7 +35,7 @@ local? = File.exists?(Path.join(__DIR__, "../native/ex_arrow_native/Cargo.toml") } else { - {:ex_arrow, "~> 0.8.0"}, + {:ex_arrow, "~> 0.9.0"}, [], [adbc: [drivers: [:sqlite]]] } diff --git a/livebook/01_ipc.livemd b/livebook/01_ipc.livemd index 56e4284..e058654 100644 --- a/livebook/01_ipc.livemd +++ b/livebook/01_ipc.livemd @@ -31,7 +31,7 @@ local? = File.exists?(Path.join(__DIR__, "../native/ex_arrow_native/Cargo.toml") } else { - {:ex_arrow, "~> 0.8.0"}, + {:ex_arrow, "~> 0.9.0"}, [], [adbc: [drivers: [:sqlite]]] } diff --git a/livebook/02_flight.livemd b/livebook/02_flight.livemd index 011fcff..1e8e27f 100644 --- a/livebook/02_flight.livemd +++ b/livebook/02_flight.livemd @@ -31,7 +31,7 @@ local? = File.exists?(Path.join(__DIR__, "../native/ex_arrow_native/Cargo.toml") } else { - {:ex_arrow, "~> 0.8.0"}, + {:ex_arrow, "~> 0.9.0"}, [], [adbc: [drivers: [:sqlite]]] } diff --git a/livebook/03_adbc.livemd b/livebook/03_adbc.livemd index 1db1426..5b8242e 100644 --- a/livebook/03_adbc.livemd +++ b/livebook/03_adbc.livemd @@ -31,7 +31,7 @@ local? = File.exists?(Path.join(__DIR__, "../native/ex_arrow_native/Cargo.toml") } else { - {:ex_arrow, "~> 0.8.0"}, + {:ex_arrow, "~> 0.9.0"}, [], [adbc: [drivers: [:sqlite]]] } diff --git a/livebook/04_adbc_integration.livemd b/livebook/04_adbc_integration.livemd index 6e69010..2efd7eb 100644 --- a/livebook/04_adbc_integration.livemd +++ b/livebook/04_adbc_integration.livemd @@ -31,7 +31,7 @@ local? = File.exists?(Path.join(__DIR__, "../native/ex_arrow_native/Cargo.toml") } else { - {:ex_arrow, "~> 0.8.0"}, + {:ex_arrow, "~> 0.9.0"}, [], [adbc: [drivers: [:sqlite]]] } diff --git a/livebook/05_parquet.livemd b/livebook/05_parquet.livemd index eaeb153..6fac166 100644 --- a/livebook/05_parquet.livemd +++ b/livebook/05_parquet.livemd @@ -28,7 +28,7 @@ local? = File.exists?(Path.join(__DIR__, "../native/ex_arrow_native/Cargo.toml") } else { - {:ex_arrow, "~> 0.8.0"}, + {:ex_arrow, "~> 0.9.0"}, [], [] } diff --git a/livebook/06_datasets.livemd b/livebook/06_datasets.livemd index 0a440ab..c2cb1c4 100644 --- a/livebook/06_datasets.livemd +++ b/livebook/06_datasets.livemd @@ -27,7 +27,7 @@ local? = File.exists?(Path.join(__DIR__, "../native/ex_arrow_native/Cargo.toml") } else { - {:ex_arrow, "~> 0.8.0"}, + {:ex_arrow, "~> 0.9.0"}, [], [] } diff --git a/livebook/README.md b/livebook/README.md index 3907856..a8ebfd5 100644 --- a/livebook/README.md +++ b/livebook/README.md @@ -28,7 +28,7 @@ Together they demonstrate ExArrow functionality: IPC (stream + file), Flight (cl | Where you open the notebook | `ex_arrow` source | |----------------------------|-------------------| | From `livebook/` in a git clone | Local path + `EX_ARROW_BUILD=1` (compile NIF from Rust) | -| From Livebook autosave or elsewhere | Hex `~> 0.8.0` (precompiled NIF, no Rust) | +| From Livebook autosave or elsewhere | Hex `~> 0.9.0` (precompiled NIF, no Rust) | ### ADBC in Livebook diff --git a/mix.exs b/mix.exs index c49aaaa..5d16ae3 100644 --- a/mix.exs +++ b/mix.exs @@ -1,7 +1,7 @@ defmodule ExArrow.MixProject do use Mix.Project - @version "0.8.0" + @version "0.9.0" @source_url "https://github.com/thanos/ex_arrow" def project do @@ -37,7 +37,7 @@ defmodule ExArrow.MixProject do defp package do [ - description: "Apache Arrow support for the BEAM: IPC, Flight, ADBC bindings", + description: "Apache Arrow for the BEAM: IPC, Flight, ADBC, Parquet, Dataset/Scanner", licenses: ["MIT"], maintainers: ["Thanos Vassilakis"], links: %{ diff --git a/native/ex_arrow_native/Cargo.lock b/native/ex_arrow_native/Cargo.lock index ab448af..be74c60 100644 --- a/native/ex_arrow_native/Cargo.lock +++ b/native/ex_arrow_native/Cargo.lock @@ -474,7 +474,7 @@ checksum = "877a4ace8713b0bcf2a4e7eec82529c029f1d0619886d18145fea96c3ffe5c0f" [[package]] name = "ex_arrow_native" -version = "0.8.0" +version = "0.9.0" dependencies = [ "adbc_core", "adbc_driver_manager", diff --git a/native/ex_arrow_native/Cargo.toml b/native/ex_arrow_native/Cargo.toml index 0faa5df..562e054 100644 --- a/native/ex_arrow_native/Cargo.toml +++ b/native/ex_arrow_native/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "ex_arrow_native" -version = "0.8.0" +version = "0.9.0" edition = "2021" description = "Apache Arrow NIFs for ExArrow Elixir library" license = "Apache-2.0" diff --git a/native/ex_arrow_native/src/flight.rs b/native/ex_arrow_native/src/flight.rs index 1af07e7..afa4900 100644 --- a/native/ex_arrow_native/src/flight.rs +++ b/native/ex_arrow_native/src/flight.rs @@ -67,9 +67,9 @@ fn parse_tls_mode<'a>(term: Term<'a>) -> Result { if atom == system_certs() { return Ok(TlsMode::SystemCerts); } - return Err(format!( - "unknown tls_mode atom; expected :plaintext or :system_certs" - )); + return Err( + "unknown tls_mode atom; expected :plaintext or :system_certs".to_string(), + ); } // Tuple path: {:custom_ca, pem_binary} ─────────────────────────────────── @@ -194,7 +194,7 @@ fn schema_ipc_bytes(schema: &SchemaRef) -> Result { let msg: IpcMessage = SchemaAsIpc::new(schema, &IpcWriteOptions::default()) .try_into() .map_err(|e| Status::internal(format!("schema IPC encode: {e}")))?; - Ok(Bytes::from(msg.0)) + Ok(msg.0) } // ── Descriptor codec helpers ───────────────────────────────────────────────── @@ -482,7 +482,7 @@ impl FlightService for EchoFlightService { let ticket_key = first .flight_descriptor .as_ref() - .and_then(|d| descriptor_to_key(d)) + .and_then(descriptor_to_key) .unwrap_or_else(|| String::from_utf8_lossy(ECHO_TICKET).into_owned()); // Re-prepend the first message and decode all batches. @@ -649,7 +649,7 @@ pub fn flight_server_start<'a>( return; } }; - let local = listener.local_addr().unwrap_or_else(|_| addr); + let local = listener.local_addr().unwrap_or(addr); let actual_port = local.port(); let actual_host = local.ip().to_string(); let incoming = TcpListenerStream::new(listener); diff --git a/native/ex_arrow_native/src/ipc.rs b/native/ex_arrow_native/src/ipc.rs index d14c371..289983e 100644 --- a/native/ex_arrow_native/src/ipc.rs +++ b/native/ex_arrow_native/src/ipc.rs @@ -493,11 +493,11 @@ fn extract_primitive_buffer<'a>(env: Env<'a>, array: &ArrayRef) -> Term<'a> { }; let len = bool_arr.len(); let mut byte_buf = vec![0u8; len]; - for i in 0..len { - // is_null check ensures null slots emit 0 instead of the - // unspecified backing bit that value(i) returns. + // is_null check ensures null slots emit 0 instead of the + // unspecified backing bit that value(i) returns. + for (i, slot) in byte_buf.iter_mut().enumerate() { if !bool_arr.is_null(i) && bool_arr.value(i) { - byte_buf[i] = 1; + *slot = 1; } } let mut owned = match rustler::OwnedBinary::new(len) {