@@ -24,9 +24,10 @@ use arrow::array::{
2424use arrow:: datatypes:: { DataType , Field , Schema } ;
2525use datafusion:: datasource:: physical_plan:: ParquetSource ;
2626use datafusion:: physical_plan:: collect;
27- use datafusion:: prelude:: SessionContext ;
27+ use datafusion:: prelude:: { ParquetReadOptions , SessionConfig , SessionContext } ;
2828use datafusion:: test:: object_store:: local_unpartitioned_file;
2929use datafusion_common:: Result ;
30+ use datafusion_common:: ScalarValue ;
3031use datafusion_common:: test_util:: batches_to_sort_string;
3132use datafusion_execution:: object_store:: ObjectStoreUrl ;
3233
@@ -145,6 +146,154 @@ async fn multi_parquet_coercion_projection() {
145146 " ) ;
146147}
147148
149+ /// Writes `batch` to a temp parquet file where `rle_cols` use RLE_DICTIONARY encoding and all other columns use plain.
150+ fn store_mixed_encoding_parquet (
151+ batch : & RecordBatch ,
152+ rle_cols : & [ & str ] ,
153+ ) -> ( ObjectMeta , NamedTempFile ) {
154+ use parquet:: schema:: types:: ColumnPath ;
155+ let mut output = tempfile:: Builder :: new ( )
156+ . suffix ( ".parquet" )
157+ . tempfile ( )
158+ . expect ( "creating temp file" ) ;
159+ let mut builder = WriterProperties :: builder ( ) . set_dictionary_enabled ( false ) ;
160+ for col in rle_cols {
161+ builder = builder
162+ . set_column_dictionary_enabled ( ColumnPath :: new ( vec ! [ col. to_string( ) ] ) , true ) ;
163+ }
164+ let mut writer =
165+ ArrowWriter :: try_new ( & mut output, batch. schema ( ) , Some ( builder. build ( ) ) )
166+ . expect ( "creating writer" ) ;
167+ writer. write ( batch) . expect ( "Writing batch" ) ;
168+ writer. close ( ) . unwrap ( ) ;
169+ let meta = local_unpartitioned_file ( & output) ;
170+ ( meta, output)
171+ }
172+
173+ /// Promotion happens at logical planning time: after `register_parquet` with the
174+ /// flag on, `ctx.table()` already shows RLE columns as `Dictionary(Int32, Utf8)`
175+ /// before any query executes.
176+ #[ tokio:: test]
177+ async fn rle_dictionary_logical_schema_promotion ( ) {
178+ let schema = Arc :: new ( Schema :: new ( vec ! [
179+ Field :: new( "env" , DataType :: Utf8 , true ) ,
180+ Field :: new( "version" , DataType :: Utf8 , true ) ,
181+ ] ) ) ;
182+ let batch = RecordBatch :: try_new (
183+ Arc :: clone ( & schema) ,
184+ vec ! [
185+ Arc :: new( StringArray :: from( vec![ "prod" , "staging" ] ) ) as ArrayRef ,
186+ Arc :: new( StringArray :: from( vec![ "1.0" , "1.1" ] ) ) as ArrayRef ,
187+ ] ,
188+ )
189+ . unwrap ( ) ;
190+ let ( _meta, file) = store_mixed_encoding_parquet ( & batch, & [ "env" ] ) ;
191+
192+ let config = SessionConfig :: new ( ) . set (
193+ "datafusion.execution.parquet.enable_rle_to_dictionary" ,
194+ & ScalarValue :: Boolean ( Some ( true ) ) ,
195+ ) ;
196+ let ctx = SessionContext :: new_with_config ( config) ;
197+ ctx. register_parquet (
198+ "t" ,
199+ file. path ( ) . to_str ( ) . unwrap ( ) ,
200+ ParquetReadOptions :: default ( ) ,
201+ )
202+ . await
203+ . unwrap ( ) ;
204+
205+ let logical_schema = ctx. table ( "t" ) . await . unwrap ( ) . schema ( ) . clone ( ) ;
206+ let dict_type =
207+ DataType :: Dictionary ( Box :: new ( DataType :: Int32 ) , Box :: new ( DataType :: Utf8 ) ) ;
208+ assert_eq ! (
209+ logical_schema
210+ . field_with_unqualified_name( "env" )
211+ . unwrap( )
212+ . data_type( ) ,
213+ & dict_type,
214+ ) ;
215+ assert_eq ! (
216+ logical_schema
217+ . field_with_unqualified_name( "version" )
218+ . unwrap( )
219+ . data_type( ) ,
220+ & DataType :: Utf8View ,
221+ ) ;
222+ }
223+
224+ /// Five string columns, three RLE_DICTIONARY encoded and two plain.
225+ /// Flag on: only the three RLE columns become `Dictionary(Int32, Utf8)`; plain
226+ /// columns stay `Utf8View`. Flag off: all columns stay `Utf8View`.
227+ #[ tokio:: test]
228+ async fn rle_dictionary_selective_promotion ( ) {
229+ let schema = Arc :: new ( Schema :: new ( vec ! [
230+ Field :: new( "env" , DataType :: Utf8 , true ) ,
231+ Field :: new( "region" , DataType :: Utf8 , true ) ,
232+ Field :: new( "tier" , DataType :: Utf8 , true ) ,
233+ Field :: new( "version" , DataType :: Utf8 , true ) ,
234+ Field :: new( "build" , DataType :: Utf8 , true ) ,
235+ ] ) ) ;
236+ let make_col = |vals : Vec < & str > | -> ArrayRef { Arc :: new ( StringArray :: from ( vals) ) } ;
237+ let batch = RecordBatch :: try_new (
238+ Arc :: clone ( & schema) ,
239+ vec ! [
240+ make_col( vec![ "prod" , "staging" , "prod" , "canary" , "staging" , "prod" ] ) ,
241+ make_col( vec![
242+ "us-east" , "us-west" , "us-east" , "eu" , "us-west" , "us-east" ,
243+ ] ) ,
244+ make_col( vec![ "free" , "pro" , "free" , "enterprise" , "pro" , "free" ] ) ,
245+ make_col( vec![ "1.0" , "1.1" , "1.2" , "1.0" , "1.1" , "1.2" ] ) ,
246+ make_col( vec![
247+ "debug" , "release" , "debug" , "release" , "debug" , "release" ,
248+ ] ) ,
249+ ] ,
250+ )
251+ . unwrap ( ) ;
252+
253+ let ( _meta, file) = store_mixed_encoding_parquet ( & batch, & [ "env" , "region" , "tier" ] ) ;
254+ let path = file. path ( ) . to_str ( ) . unwrap ( ) ;
255+ let sql = "SELECT env, region, tier, version, build FROM t LIMIT 1" ;
256+ let dict_type =
257+ DataType :: Dictionary ( Box :: new ( DataType :: Int32 ) , Box :: new ( DataType :: Utf8 ) ) ;
258+
259+ // Flag off: all string columns stay Utf8View.
260+ let ctx = SessionContext :: new ( ) ;
261+ ctx. register_parquet ( "t" , path, ParquetReadOptions :: default ( ) )
262+ . await
263+ . unwrap ( ) ;
264+ let result = ctx. sql ( sql) . await . unwrap ( ) . collect ( ) . await . unwrap ( ) ;
265+ assert ! ( !result. is_empty( ) ) ;
266+ let s = result[ 0 ] . schema ( ) ;
267+ for col in [ "env" , "region" , "tier" , "version" , "build" ] {
268+ assert_eq ! (
269+ s. field_with_name( col) . unwrap( ) . data_type( ) ,
270+ & DataType :: Utf8View
271+ ) ;
272+ }
273+
274+ // Flag on: only RLE-encoded columns become Dict; plain columns stay Utf8View.
275+ let config = SessionConfig :: new ( ) . set (
276+ "datafusion.execution.parquet.enable_rle_to_dictionary" ,
277+ & ScalarValue :: Boolean ( Some ( true ) ) ,
278+ ) ;
279+ let ctx = SessionContext :: new_with_config ( config) ;
280+ ctx. register_parquet ( "t" , path, ParquetReadOptions :: default ( ) )
281+ . await
282+ . unwrap ( ) ;
283+ let result = ctx. sql ( sql) . await . unwrap ( ) . collect ( ) . await . unwrap ( ) ;
284+ assert ! ( !result. is_empty( ) ) ;
285+ let s = result[ 0 ] . schema ( ) ;
286+ for col in [ "env" , "region" , "tier" ] {
287+ assert_eq ! ( s. field_with_name( col) . unwrap( ) . data_type( ) , & dict_type) ;
288+ }
289+ for col in [ "version" , "build" ] {
290+ assert_eq ! (
291+ s. field_with_name( col) . unwrap( ) . data_type( ) ,
292+ & DataType :: Utf8View
293+ ) ;
294+ }
295+ }
296+
148297/// Writes `batches` to a temporary parquet file
149298pub fn store_parquet (
150299 batches : Vec < RecordBatch > ,
@@ -154,14 +303,10 @@ pub fn store_parquet(
154303 . into_iter ( )
155304 . map ( |batch| {
156305 let mut output = NamedTempFile :: new ( ) . expect ( "creating temp file" ) ;
157-
158- let builder = WriterProperties :: builder ( ) ;
159- let props = builder. build ( ) ;
160-
306+ let props = WriterProperties :: builder ( ) . build ( ) ;
161307 let mut writer =
162308 ArrowWriter :: try_new ( & mut output, batch. schema ( ) , Some ( props) )
163309 . expect ( "creating writer" ) ;
164-
165310 writer. write ( & batch) . expect ( "Writing batch" ) ;
166311 writer. close ( ) . unwrap ( ) ;
167312 output
0 commit comments