diff --git a/Cargo.lock b/Cargo.lock index 566cc1166813c..024e83a46d922 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -165,8 +165,7 @@ checksum = "7c02d123df017efcdfbd739ef81735b36c5ba83ec3c59c80a9d7ecc718f92e50" [[package]] name = "arrow" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b952ca5a8046ad741b60f142d6eca4aeebcad615694202bc64c5341f23e32c5b" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "arrow-arith", "arrow-array", @@ -188,8 +187,7 @@ dependencies = [ [[package]] name = "arrow-arith" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "64a13b8d3008c4e9063c597a08f46446fe3fd5789277127672d6c0bdbb43b1ff" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "arrow-array", "arrow-buffer", @@ -202,8 +200,7 @@ dependencies = [ [[package]] name = "arrow-array" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9486151b2f0785bafc6fa04fc5c99fcb4495455662e58787ea32eaaed33c4192" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "ahash", "arrow-buffer", @@ -221,8 +218,7 @@ dependencies = [ [[package]] name = "arrow-avro" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2e4f9b23a0d7b613acb59fa20bdbe0f80ffdae6411498378340b3915e45f5b84" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "arrow-array", "arrow-buffer", @@ -245,8 +241,7 @@ dependencies = [ [[package]] name = "arrow-buffer" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c4776577a87794bfdf0b4e90e2ea12454fa7738ea2823c4be5b9d1851da7b434" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "bytes", "half", @@ -257,8 +252,7 @@ dependencies = [ [[package]] name = "arrow-cast" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a9ad451ce4f98710828a455b96991b8f031deb2e67f5fcad6773f017e4a69c3a" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "arrow-array", "arrow-buffer", @@ -279,8 +273,7 @@ dependencies = [ [[package]] name = "arrow-csv" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8aa7bf96d6141a7bcca2eed57c7c9767d2a2175281857b8a7b68308992864784" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "arrow-array", "arrow-cast", @@ -294,8 +287,7 @@ dependencies = [ [[package]] name = "arrow-data" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b38fe43e2e8704360f1464e6e8cc4fc381ef02cc4fb0192afa8df1aaa0115c66" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "arrow-buffer", "arrow-schema", @@ -307,8 +299,7 @@ dependencies = [ [[package]] name = "arrow-flight" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "42115e09dbb694b5955da998912121451c6910b338228cb80a5701370dba43ff" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "arrow-arith", "arrow-array", @@ -335,8 +326,7 @@ dependencies = [ [[package]] name = "arrow-ipc" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "29dac499fcbc6ba74ee0324057821d381929a48526a3966bd9dffb44aa06d98c" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "arrow-array", "arrow-buffer", @@ -351,8 +341,7 @@ dependencies = [ [[package]] name = "arrow-json" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0fe05e916ddc50f4c7a363cd69c0ef5894fcee063517e9a0b8582f0c56746af6" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "arrow-array", "arrow-buffer", @@ -376,8 +365,7 @@ dependencies = [ [[package]] name = "arrow-ord" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0e13dbdc2a9c053c10c7baa6e30faee04a180aa7ce88e471835850ce37abd20b" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "arrow-array", "arrow-buffer", @@ -389,8 +377,7 @@ dependencies = [ [[package]] name = "arrow-row" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4d5a1f8c733d15260b305683472ee8ad89c62cbd706703ca873b90d051b41592" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "arrow-array", "arrow-buffer", @@ -402,8 +389,7 @@ dependencies = [ [[package]] name = "arrow-schema" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d9e4969dc350d571766247143ab36a5187d095d3d3690970408bc630d47c69e5" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "bitflags", "serde", @@ -414,8 +400,7 @@ dependencies = [ [[package]] name = "arrow-select" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "402770dba90865359d98d1ef92ef16e23d75c0cca9c2c880c8a05468b7743bf9" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "ahash", "arrow-array", @@ -428,8 +413,7 @@ dependencies = [ [[package]] name = "arrow-string" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a2b0afbb8b9016700938291123df30838b89decc3213dba00852021988b170d3" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "arrow-array", "arrow-buffer", @@ -4053,9 +4037,9 @@ checksum = "112b39cec0b298b6c1999fee3e31427f74f676e4cb9879ed1a121b43661a4154" [[package]] name = "lz4_flex" -version = "0.13.0" +version = "0.13.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "db9a0d582c2874f68138a16ce1867e0ffde6c0bb0a0df85e1f36d04146db488a" +checksum = "7ef0d4ed8669f8f8826eb00dc878084aa8f253506c4fd5e8f58f5bce72ddb97e" dependencies = [ "twox-hash", ] @@ -4464,8 +4448,7 @@ dependencies = [ [[package]] name = "parquet" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5302d4da74d6596a1f11f9928767995b53bca657cbeea1e4e8c5074f8a1157dd" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "ahash", "arrow-array", @@ -4851,7 +4834,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "03da047801ff44bb6a4d407d4860c05fd70bb81714e6b2f3812603d5b145b042" dependencies = [ "heck", - "itertools 0.14.0", + "itertools 0.13.0", "log", "multimap", "petgraph", @@ -4870,7 +4853,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b570b25f7617e43d59005d0990ccb79e950a423952cea19671b7a876da390adf" dependencies = [ "anyhow", - "itertools 0.14.0", + "itertools 0.13.0", "proc-macro2", "quote", "syn 2.0.119", diff --git a/Cargo.toml b/Cargo.toml index 6f4c10f8e7552..8bc1fcfdc899f 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -209,6 +209,31 @@ url = "2.5.7" uuid = "1.23" zstd = { version = "0.13", default-features = false } +# TEMPORARY patch: points at an unreleased arrow-rs branch that adds the +# dictionary-page decode API this branch's Parquet dictionary row-group +# pruning depends on (arrow-rs issue #9010 / PR #10420, which supersedes +# the earlier, now-closed PR #9011). Remove once arrow-rs releases a +# version with this API and the dependency above is bumped to it -- do +# not merge this patch. +[patch.crates-io] +arrow = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-arith = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-array = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-avro = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-buffer = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-cast = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-csv = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-data = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-flight = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-ipc = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-json = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-ord = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-row = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-schema = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-select = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-string = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +parquet = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } + [workspace.lints.clippy] # Detects large stack-allocated futures that may cause stack overflow crashes (see threshold in clippy.toml) large_futures = "warn" diff --git a/datafusion/common/src/config.rs b/datafusion/common/src/config.rs index 81f573fc2a23e..35898ee8ab087 100644 --- a/datafusion/common/src/config.rs +++ b/datafusion/common/src/config.rs @@ -1181,6 +1181,16 @@ config_namespace! { /// (reading) Use any available bloom filters when reading parquet files pub bloom_filter_on_read: bool, default = true + /// (reading) Use fully dictionary-encoded `BYTE_ARRAY` (Utf8/Binary) + /// column chunks as an exact row-group membership index when reading + /// parquet files. Unlike bloom filters, a fully dictionary-encoded + /// column chunk's dictionary is the exact, complete set of the row + /// group's distinct values, so this can prune both `IN`/`=` and + /// `NOT IN`/`!=` predicates. Only column chunks whose page encoding + /// statistics prove every data page came from the dictionary are + /// used; chunks that fell back to `PLAIN` encoding are ignored. + pub dictionary_filter_on_read: bool, default = false + /// (reading) The maximum predicate cache size, in bytes. When /// `pushdown_filters` is enabled, sets the maximum memory used to cache /// the results of predicate evaluation between filter evaluation and diff --git a/datafusion/common/src/file_options/parquet_writer.rs b/datafusion/common/src/file_options/parquet_writer.rs index 320bfcf33e488..f737a78c2efe0 100644 --- a/datafusion/common/src/file_options/parquet_writer.rs +++ b/datafusion/common/src/file_options/parquet_writer.rs @@ -242,6 +242,7 @@ impl ParquetOptions { maximum_parallel_row_group_writers: _, maximum_buffered_record_batches_per_stream: _, bloom_filter_on_read: _, // reads not used for writer props + dictionary_filter_on_read: _, // reads not used for writer props schema_force_view_types: _, binary_as_string: _, // not used for writer props coerce_int96: _, // not used for writer props @@ -500,6 +501,7 @@ mod tests { maximum_buffered_record_batches_per_stream: defaults .maximum_buffered_record_batches_per_stream, bloom_filter_on_read: defaults.bloom_filter_on_read, + dictionary_filter_on_read: defaults.dictionary_filter_on_read, schema_force_view_types: defaults.schema_force_view_types, binary_as_string: defaults.binary_as_string, skip_arrow_metadata: defaults.skip_arrow_metadata, @@ -620,6 +622,8 @@ mod tests { maximum_buffered_record_batches_per_stream: global_options_defaults .maximum_buffered_record_batches_per_stream, bloom_filter_on_read: global_options_defaults.bloom_filter_on_read, + dictionary_filter_on_read: global_options_defaults + .dictionary_filter_on_read, max_predicate_cache_size: global_options_defaults .max_predicate_cache_size, schema_force_view_types: global_options_defaults.schema_force_view_types, diff --git a/datafusion/datasource-parquet/Cargo.toml b/datafusion/datasource-parquet/Cargo.toml index 32424069c17a0..2590751aafe5d 100644 --- a/datafusion/datasource-parquet/Cargo.toml +++ b/datafusion/datasource-parquet/Cargo.toml @@ -91,3 +91,7 @@ harness = false [[bench]] name = "parquet_metadata_statistics" harness = false + +[[bench]] +name = "parquet_dictionary_pruning" +harness = false diff --git a/datafusion/datasource-parquet/benches/parquet_dictionary_pruning.rs b/datafusion/datasource-parquet/benches/parquet_dictionary_pruning.rs new file mode 100644 index 0000000000000..eedd2552eb907 --- /dev/null +++ b/datafusion/datasource-parquet/benches/parquet_dictionary_pruning.rs @@ -0,0 +1,348 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Benchmarks for row-group pruning by Parquet dictionaries. +//! +//! Compares three levels of row-group pruning for a query that looks for a +//! single, sparse value in a high-cardinality `Utf8` column: +//! +//! - `statistics_only`: min/max statistics alone (the baseline every reader +//! already gets, dictionary or bloom filter disabled). +//! - `bloom_filter`: adds Parquet Split Block Bloom Filters +//! (`bloom_filter_on_read`). +//! - `dictionary`: adds exact Parquet dictionary-page pruning +//! (`dictionary_filter_on_read`), this crate's new row-group index. +//! +//! The dataset has `TOTAL_ROW_GROUPS` row groups, each with +//! `DISTINCT_VALUES_PER_ROW_GROUP` distinct values unique to that row group +//! (so the column is high-cardinality overall, but any single row group's +//! dictionary is small and never falls back to `PLAIN`). The query looks for +//! a value that exists in exactly one row group, so a fully effective +//! pruning strategy skips reading data pages for all the others. +//! +//! Run with `cargo bench -p datafusion-datasource-parquet --bench parquet_dictionary_pruning`. + +use std::hint::black_box; +use std::path::{Path, PathBuf}; +use std::sync::{Arc, LazyLock}; + +use arrow::array::{ArrayRef, RecordBatch, StringArray}; +use arrow::datatypes::{DataType, Field, Schema, SchemaRef}; +use criterion::{Criterion, Throughput, criterion_group, criterion_main}; +use datafusion_datasource_parquet::{ + BloomFilterStatistics, DictionaryStatistics, ParquetAccessPlan, ParquetFileMetrics, + RowGroupAccessPlanFilter, is_fully_dictionary_encoded, +}; +use datafusion_expr::{col, lit}; +use datafusion_physical_expr::planner::logical2physical; +use datafusion_physical_plan::metrics::ExecutionPlanMetricsSet; +use datafusion_pruning::PruningPredicate; +use parquet::arrow::arrow_reader::ParquetRecordBatchReaderBuilder; +use parquet::arrow::{ArrowWriter, parquet_column}; +use parquet::file::metadata::ParquetMetaDataReader; +use parquet::file::properties::WriterProperties; +use parquet::file::reader::{FileReader, SerializedFileReader}; +use parquet::schema::types::SchemaDescriptor; +use tempfile::TempDir; + +const TOTAL_ROW_GROUPS: usize = 200; +const DISTINCT_VALUES_PER_ROW_GROUP: usize = 1_000; +const ROWS_PER_ROW_GROUP: usize = DISTINCT_VALUES_PER_ROW_GROUP; +const TOTAL_VALUES: usize = TOTAL_ROW_GROUPS * DISTINCT_VALUES_PER_ROW_GROUP; +const COLUMN_NAME: &str = "s"; + +/// Interleaves (round-robins) values across row groups: row group `rg` gets +/// the values at global positions `rg, rg + TOTAL_ROW_GROUPS, rg + 2 * +/// TOTAL_ROW_GROUPS, ...`. Every row group's `[min, max]` therefore spans +/// almost the entire value domain (min is close to 0, max close to +/// `TOTAL_VALUES`), so plain min/max statistics can't prune any of them -- +/// only the *set* of values actually present (from a bloom filter or exact +/// dictionary) can distinguish row groups. This mirrors data that arrives +/// already shuffled with respect to a low/no-correlation column, e.g. trace +/// IDs or session IDs sharded across row groups by arrival time. +fn value_at(rg: usize, k: usize) -> String { + format!("val-{:06}", rg + k * TOTAL_ROW_GROUPS) +} + +/// Present in exactly one row group. Chosen near the middle of the value +/// domain so every row group's `[min, max]` range contains it, but only one +/// row group's dictionary actually does. +fn needle() -> String { + let global_index = TOTAL_VALUES / 2; + value_at( + global_index % TOTAL_ROW_GROUPS, + global_index / TOTAL_ROW_GROUPS, + ) +} + +struct BenchmarkDataset { + _tempdir: TempDir, + file_path: PathBuf, +} + +impl BenchmarkDataset { + fn path(&self) -> &Path { + &self.file_path + } +} + +static DATASET: LazyLock = LazyLock::new(|| { + create_dataset().expect("failed to prepare parquet benchmark dataset") +}); + +fn schema() -> SchemaRef { + Arc::new(Schema::new(vec![Field::new( + COLUMN_NAME, + DataType::Utf8, + false, + )])) +} + +fn create_dataset() -> datafusion_common::Result { + let tempdir = TempDir::new()?; + let file_path = tempdir.path().join("dictionary_pruning.parquet"); + + let schema = schema(); + // Dictionary and bloom filter both enabled, so the same file drives all + // three benchmark scenarios below. + let writer_props = WriterProperties::builder() + .set_max_row_group_row_count(Some(ROWS_PER_ROW_GROUP)) + .set_dictionary_enabled(true) + .set_bloom_filter_enabled(true) + .build(); + + let mut writer = ArrowWriter::try_new( + std::fs::File::create(&file_path)?, + Arc::clone(&schema), + Some(writer_props), + )?; + + for rg in 0..TOTAL_ROW_GROUPS { + let values: Vec = (0..DISTINCT_VALUES_PER_ROW_GROUP) + .map(|k| value_at(rg, k)) + .collect(); + let array: ArrayRef = Arc::new(StringArray::from_iter_values( + values.iter().map(|v| v.as_str()), + )); + let batch = RecordBatch::try_new(Arc::clone(&schema), vec![array])?; + writer.write(&batch)?; + } + writer.close()?; + + let reader = + ParquetRecordBatchReaderBuilder::try_new(std::fs::File::open(&file_path)?)?; + assert_eq!(reader.metadata().row_groups().len(), TOTAL_ROW_GROUPS); + + Ok(BenchmarkDataset { + _tempdir: tempdir, + file_path, + }) +} + +/// `s = needle()` as a [`PruningPredicate`], plus the leaf column index it +/// resolves to in `parquet_schema`. +fn needle_predicate(parquet_schema: &SchemaDescriptor) -> (PruningPredicate, usize) { + let schema = schema(); + let expr = logical2physical(&col(COLUMN_NAME).eq(lit(needle())), &schema); + let predicate = PruningPredicate::try_new(expr, schema).expect("valid predicate"); + let (column_idx, _) = parquet_column(parquet_schema, predicate.schema(), COLUMN_NAME) + .expect("column present"); + (predicate, column_idx) +} + +/// Total compressed bytes of the queried column across `indexes`' row +/// groups -- i.e. the data-page bytes an actual scan would have to read for +/// the row groups that survive pruning. +fn column_bytes( + metadata: &parquet::file::metadata::ParquetMetaData, + column_idx: usize, + indexes: impl Iterator, +) -> i64 { + indexes + .map(|idx| metadata.row_group(idx).column(column_idx).compressed_size()) + .sum() +} + +fn prune_by_statistics_only(path: &Path) -> Vec { + let file = std::fs::File::open(path).expect("open file"); + let reader = SerializedFileReader::new(file).expect("open reader"); + let metadata = reader.metadata(); + let (predicate, _) = needle_predicate(metadata.file_metadata().schema_descr()); + + let mut access_plan = RowGroupAccessPlanFilter::new(ParquetAccessPlan::new_all( + metadata.num_row_groups(), + )); + let metrics_set = ExecutionPlanMetricsSet::new(); + let metrics = ParquetFileMetrics::new(0, &path.display().to_string(), &metrics_set); + access_plan.prune_by_statistics( + predicate.schema(), + metadata.file_metadata().schema_descr(), + metadata.row_groups(), + &predicate, + &metrics, + ); + access_plan.row_group_indexes().collect() +} + +fn prune_by_bloom_filter(path: &Path) -> Vec { + let file = std::fs::File::open(path).expect("open file"); + let builder = ParquetRecordBatchReaderBuilder::try_new(file).expect("open reader"); + let metadata = builder.metadata().clone(); + let (predicate, column_idx) = + needle_predicate(metadata.file_metadata().schema_descr()); + let physical_type = metadata + .file_metadata() + .schema_descr() + .column(column_idx) + .physical_type(); + let type_length = metadata + .file_metadata() + .schema_descr() + .column(column_idx) + .type_length(); + + let mut access_plan = RowGroupAccessPlanFilter::new(ParquetAccessPlan::new_all( + metadata.num_row_groups(), + )); + let metrics_set = ExecutionPlanMetricsSet::new(); + let metrics = ParquetFileMetrics::new(0, &path.display().to_string(), &metrics_set); + access_plan.prune_by_statistics( + predicate.schema(), + metadata.file_metadata().schema_descr(), + metadata.row_groups(), + &predicate, + &metrics, + ); + + let mut row_group_bloom_filters = + vec![BloomFilterStatistics::new(); metadata.num_row_groups()]; + for idx in access_plan.row_group_indexes() { + let mut stats = BloomFilterStatistics::with_capacity(1); + if let Ok(Some(bf)) = builder.get_row_group_column_bloom_filter(idx, column_idx) { + stats.insert(COLUMN_NAME, bf, physical_type, type_length); + } + row_group_bloom_filters[idx] = stats; + } + access_plan.prune_by_bloom_filters(&predicate, &metrics, &row_group_bloom_filters); + access_plan.row_group_indexes().collect() +} + +fn prune_by_dictionary(path: &Path) -> Vec { + let file = std::fs::File::open(path).expect("open file"); + let reader = SerializedFileReader::new(file).expect("open reader"); + let metadata = reader.metadata(); + let (predicate, column_idx) = + needle_predicate(metadata.file_metadata().schema_descr()); + + let mut access_plan = RowGroupAccessPlanFilter::new(ParquetAccessPlan::new_all( + metadata.num_row_groups(), + )); + let metrics_set = ExecutionPlanMetricsSet::new(); + let metrics = ParquetFileMetrics::new(0, &path.display().to_string(), &metrics_set); + access_plan.prune_by_statistics( + predicate.schema(), + metadata.file_metadata().schema_descr(), + metadata.row_groups(), + &predicate, + &metrics, + ); + + let file = std::fs::File::open(path).expect("open file"); + let mut row_group_dictionaries = + vec![DictionaryStatistics::new(); metadata.num_row_groups()]; + for idx in access_plan.row_group_indexes() { + let col_meta = metadata.row_group(idx).column(column_idx); + if !is_fully_dictionary_encoded(col_meta) { + continue; + } + let mut stats = DictionaryStatistics::with_capacity(1); + if let Ok(Some(dict)) = ParquetMetaDataReader::read_column_dictionary( + &file, metadata, idx, column_idx, + ) { + stats.insert(COLUMN_NAME, &dict).expect("decode dictionary"); + } + row_group_dictionaries[idx] = stats; + } + access_plan.prune_by_dictionary(&predicate, &metrics, &row_group_dictionaries); + access_plan.row_group_indexes().collect() +} + +fn parquet_dictionary_pruning(c: &mut Criterion) { + let dataset_path = DATASET.path().to_owned(); + let mut group = c.benchmark_group("parquet_dictionary_pruning"); + group.throughput(Throughput::Elements(TOTAL_ROW_GROUPS as u64)); + + // Sanity + a one-time report of what each strategy actually buys in data + // scanned, since that's the real payoff -- the pruning phase itself + // benchmarked below reads a bloom filter or a dictionary from every + // surviving row group, so it is not free, and a bigger, denser + // dictionary can cost more to decode than a compact bloom filter would. + // The win is in how much *data-page* reading gets skipped afterward. + { + let file = std::fs::File::open(&dataset_path).expect("open file"); + let metadata = SerializedFileReader::new(file) + .expect("open reader") + .metadata() + .clone(); + let (_, column_idx) = needle_predicate(metadata.file_metadata().schema_descr()); + + // Every row group's [min, max] range spans the needle (values are + // interleaved across row groups, see `value_at`), so statistics + // alone can't prune any of them -- this is the baseline the other + // two scenarios are compared against. + let statistics_only = prune_by_statistics_only(&dataset_path); + assert_eq!(statistics_only.len(), TOTAL_ROW_GROUPS); + // Bloom filters are probabilistic but should reliably prune this + // exact, absent-from-most-groups scenario down to (approximately) + // one row group. + let bloom_filter = prune_by_bloom_filter(&dataset_path); + assert!(bloom_filter.len() <= TOTAL_ROW_GROUPS); + // Dictionaries are exact: exactly the one row group containing the + // needle survives. + let dictionary = prune_by_dictionary(&dataset_path); + assert_eq!(dictionary.len(), 1); + + let statistics_only_bytes = + column_bytes(&metadata, column_idx, statistics_only.into_iter()); + let bloom_filter_bytes = + column_bytes(&metadata, column_idx, bloom_filter.into_iter()); + let dictionary_bytes = + column_bytes(&metadata, column_idx, dictionary.into_iter()); + eprintln!( + "parquet_dictionary_pruning: column data bytes an actual scan would \ + read -- statistics_only: {statistics_only_bytes}, bloom_filter: \ + {bloom_filter_bytes}, dictionary: {dictionary_bytes}" + ); + } + + group.bench_function("statistics_only", |b| { + b.iter(|| black_box(prune_by_statistics_only(&dataset_path))); + }); + + group.bench_function("bloom_filter", |b| { + b.iter(|| black_box(prune_by_bloom_filter(&dataset_path))); + }); + + group.bench_function("dictionary", |b| { + b.iter(|| black_box(prune_by_dictionary(&dataset_path))); + }); + + group.finish(); +} + +criterion_group!(benches, parquet_dictionary_pruning); +criterion_main!(benches); diff --git a/datafusion/datasource-parquet/src/dictionary_filter.rs b/datafusion/datasource-parquet/src/dictionary_filter.rs new file mode 100644 index 0000000000000..c5210554bb24e --- /dev/null +++ b/datafusion/datasource-parquet/src/dictionary_filter.rs @@ -0,0 +1,560 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Loaded Parquet dictionary-page data, with a [`PruningStatistics`] adapter +//! so the predicate-pruning machinery in [`datafusion_pruning`] can consume +//! it. +//! +//! Some writers (e.g. Grafana Tempo) use a low-cardinality string column's +//! Parquet dictionary as a makeshift row-group index: the dictionary page is +//! the exact, complete set of the row group's distinct non-null values. If a +//! required value is not in the dictionary, the whole row group can be +//! skipped without reading any data pages. This is a strictly stronger +//! (exact, non-probabilistic) signal than a bloom filter, letting us prune +//! both `IN`/`=` (value absent) and `NOT IN`/`!=` (value is the only one +//! present) directions -- see [`DictionaryStatistics::contained`]. + +use std::collections::{HashMap, HashSet}; + +use arrow::array::{ArrayRef, BinaryArray, BooleanArray, StringArray}; +use arrow::datatypes::DataType; +use datafusion_common::pruning::PruningStatistics; +use datafusion_common::{Column, DataFusionError, Result, ScalarValue, internal_err}; +use parquet::basic::{Encoding, Type}; +use parquet::file::metadata::ColumnChunkMetaData; + +/// In-memory decoded Parquet dictionary-page values, keyed by column name. +/// +/// This structure implements [`PruningStatistics`] and is used to prune +/// Parquet row groups based on the query predicate. Unlike bloom filters, +/// which are probabilistic, the values stored here are assumed to be the +/// *exact* set of distinct values present in the column for the row group -- +/// callers must only [`insert`](Self::insert) dictionaries for column chunks +/// that pass [`is_fully_dictionary_encoded`], or pruning will be unsound. +#[derive(Debug, Clone, Default)] +pub struct DictionaryStatistics { + /// Per-column exact value sets, keyed by predicate column name. Values + /// are stored as raw bytes so `Utf8`- and `Binary`-typed dictionaries + /// compare equally to whichever `ScalarValue` variant the predicate + /// literal happens to use (see [`normalize_literal`]). + column_values: HashMap>>, +} + +impl DictionaryStatistics { + /// Create an empty [`DictionaryStatistics`] + pub fn new() -> Self { + Default::default() + } + + /// Create an empty [`DictionaryStatistics`] with the specified capacity + pub fn with_capacity(capacity: usize) -> Self { + Self { + column_values: HashMap::with_capacity(capacity), + } + } + + /// Record the exact dictionary values for `column`, decoded as a + /// `Utf8` or `Binary` array (e.g. from + /// `ParquetRecordBatchStreamBuilder::get_row_group_column_dictionary`). + /// + /// # Panics / Errors + /// + /// This does not itself verify that the column chunk is fully + /// dictionary-encoded -- see [`is_fully_dictionary_encoded`]. Passing a + /// dictionary that is only a subset of the chunk's actual values (e.g. + /// because part of the chunk fell back to `PLAIN`) will make pruning + /// unsound. + pub fn insert( + &mut self, + column: impl Into, + dictionary: &ArrayRef, + ) -> Result<()> { + let values = match dictionary.data_type() { + DataType::Utf8 => dictionary + .as_any() + .downcast_ref::() + .ok_or_else(|| { + DataFusionError::Internal("Expected a StringArray".to_string()) + })? + .iter() + .flatten() + .map(|v| v.as_bytes().to_vec()) + .collect(), + DataType::Binary => dictionary + .as_any() + .downcast_ref::() + .ok_or_else(|| { + DataFusionError::Internal("Expected a BinaryArray".to_string()) + })? + .iter() + .flatten() + .map(|v| v.to_vec()) + .collect(), + other => { + return internal_err!( + "DictionaryStatistics only supports Utf8/Binary dictionaries, got {other}" + ); + } + }; + self.column_values.insert(column.into(), values); + Ok(()) + } +} + +/// Returns `value` normalized to its raw byte representation, if `value` is +/// a string- or binary-like scalar. Mirrors the literal normalization used +/// by [`crate::bloom_filter::BloomFilterStatistics`] so a predicate literal +/// (which may be `Utf8`, `Utf8View`, `LargeUtf8`, etc. depending on how the +/// query was planned) compares equal to a dictionary entry regardless of +/// which of those variants was used. +fn normalize_literal(value: &ScalarValue) -> Option<&[u8]> { + match value { + ScalarValue::Utf8(Some(v)) + | ScalarValue::Utf8View(Some(v)) + | ScalarValue::LargeUtf8(Some(v)) => Some(v.as_bytes()), + ScalarValue::Binary(Some(v)) + | ScalarValue::BinaryView(Some(v)) + | ScalarValue::LargeBinary(Some(v)) => Some(v.as_slice()), + ScalarValue::Dictionary(_, inner) => normalize_literal(inner), + _ => None, + } +} + +impl PruningStatistics for DictionaryStatistics { + fn min_values(&self, _column: &Column) -> Option { + None + } + + fn max_values(&self, _column: &Column) -> Option { + None + } + + fn num_containers(&self) -> usize { + 1 + } + + fn null_counts(&self, _column: &Column) -> Option { + None + } + + fn row_counts(&self) -> Option { + None + } + + /// Use the exact dictionary values to determine whether the column is + /// definitely disjoint from (`Some(false)`), or definitely a subset of + /// (`Some(true)`), `values`. + /// + /// Because the dictionary is the row group's *complete* set of distinct + /// values (guaranteed by the caller checking + /// [`is_fully_dictionary_encoded`] before inserting it), this can prove + /// both directions exactly, unlike a bloom filter's probabilistic + /// "definitely absent" only. + fn contained( + &self, + column: &Column, + values: &HashSet, + ) -> Option { + let dictionary_values = self.column_values.get(column.name.as_str())?; + + let mut literal_bytes: HashSet<&[u8]> = HashSet::with_capacity(values.len()); + for value in values { + // A literal we can't normalize (e.g. a non-string/binary type + // that somehow ended up compared against this column) makes the + // guarantee inconclusive rather than wrong. + literal_bytes.insert(normalize_literal(value)?); + } + + let none_present = literal_bytes + .iter() + .all(|literal| !dictionary_values.contains(*literal)); + if none_present { + return Some(BooleanArray::from(vec![Some(false)])); + } + + let all_present = dictionary_values + .iter() + .all(|value| literal_bytes.contains(value.as_slice())); + if all_present { + return Some(BooleanArray::from(vec![Some(true)])); + } + + Some(BooleanArray::from(vec![None])) + } +} + +/// Returns `true` if `col_meta`'s column chunk is entirely `BYTE_ARRAY` and +/// dictionary-encoded, i.e. its dictionary page is the exact, complete set of +/// the row group's distinct non-null values for that column. +/// +/// Dictionary encoding is best-effort: writers fall back to `PLAIN` once a +/// column's dictionary grows past a size limit, and once that happens the +/// data pages after the fallback contain values that never entered the +/// dictionary. `dictionary_page_offset().is_some()` alone does not rule this +/// out. Page-encoding statistics +/// () record every +/// encoding used across a column chunk's data pages, so requiring the mask +/// to contain *only* a dictionary encoding proves every data page decoded +/// from the dictionary. +/// +/// First cut: only `BYTE_ARRAY` (`Utf8`/`Binary`) columns are supported. +pub fn is_fully_dictionary_encoded(col_meta: &ColumnChunkMetaData) -> bool { + col_meta.column_descr().physical_type() == Type::BYTE_ARRAY + && col_meta.dictionary_page_offset().is_some() + && col_meta.page_encoding_stats_mask().is_some_and(|mask| { + mask.is_only(Encoding::PLAIN_DICTIONARY) + || mask.is_only(Encoding::RLE_DICTIONARY) + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + use std::sync::Arc; + + use crate::reader::ParquetFileReader; + use crate::test_util::ExpectedPruning; + use crate::{ParquetAccessPlan, ParquetFileMetrics, RowGroupAccessPlanFilter}; + + use arrow::datatypes::{Field, Schema}; + use bytes::{BufMut, BytesMut}; + use datafusion_expr::{Expr, col, lit}; + use datafusion_physical_expr::planner::logical2physical; + use datafusion_physical_plan::metrics::ExecutionPlanMetricsSet; + use datafusion_pruning::PruningPredicate; + use object_store::ObjectStoreExt; + use parquet::arrow::ArrowWriter; + use parquet::arrow::ParquetRecordBatchStreamBuilder; + use parquet::arrow::async_reader::ParquetObjectReader; + use parquet::arrow::parquet_column; + use parquet::basic::EncodingMask; + use parquet::file::metadata::ColumnChunkMetaData; + use parquet::file::properties::WriterProperties; + use parquet::schema::types::{SchemaDescriptor, Type as SchemaType}; + + #[test] + fn is_fully_dictionary_encoded_requires_only_dictionary_encoding() { + let schema_descr = Arc::new(SchemaDescriptor::new(Arc::new( + SchemaType::group_type_builder("schema") + .with_fields(vec![Arc::new( + SchemaType::primitive_type_builder("s", Type::BYTE_ARRAY) + .build() + .unwrap(), + )]) + .build() + .unwrap(), + ))); + + let fully_dictionary_encoded = + ColumnChunkMetaData::builder(schema_descr.column(0)) + .set_dictionary_page_offset(Some(0)) + .set_data_page_offset(100) + .set_page_encoding_stats_mask(EncodingMask::new_from_encodings( + [Encoding::RLE_DICTIONARY].iter(), + )) + .build() + .unwrap(); + assert!(is_fully_dictionary_encoded(&fully_dictionary_encoded)); + + // Plain fallback mid-chunk: a dictionary page exists, but at least + // one data page used PLAIN, so the dictionary is not exhaustive. + let plain_fallback = ColumnChunkMetaData::builder(schema_descr.column(0)) + .set_dictionary_page_offset(Some(0)) + .set_data_page_offset(100) + .set_page_encoding_stats_mask(EncodingMask::new_from_encodings( + [Encoding::RLE_DICTIONARY, Encoding::PLAIN].iter(), + )) + .build() + .unwrap(); + assert!(!is_fully_dictionary_encoded(&plain_fallback)); + + // No dictionary page at all. + let no_dictionary = ColumnChunkMetaData::builder(schema_descr.column(0)) + .set_data_page_offset(100) + .set_page_encoding_stats_mask(EncodingMask::new_from_encodings( + [Encoding::PLAIN].iter(), + )) + .build() + .unwrap(); + assert!(!is_fully_dictionary_encoded(&no_dictionary)); + + // No page encoding stats recorded at all: treat as not prunable. + let no_stats = ColumnChunkMetaData::builder(schema_descr.column(0)) + .set_dictionary_page_offset(Some(0)) + .set_data_page_offset(100) + .build() + .unwrap(); + assert!(!is_fully_dictionary_encoded(&no_stats)); + } + + struct DictionaryFilterTest { + schema: Schema, + post_pruning_row_groups: ExpectedPruning, + } + + impl DictionaryFilterTest { + /// A small dictionary-encoded string column with three distinct + /// values, one row group per value so pruning is easy to verify. + fn new_single_value_row_groups() -> (Self, bytes::Bytes) { + let schema = Schema::new(vec![Field::new("s", DataType::Utf8, false)]); + let arrow_schema = Arc::new(schema.clone()); + let values = ["alpha", "beta", "gamma"]; + let array: ArrayRef = Arc::new(StringArray::from_iter_values(values)); + let batch = + arrow::array::RecordBatch::try_new(arrow_schema.clone(), vec![array]) + .unwrap(); + + let props = WriterProperties::builder() + .set_dictionary_enabled(true) + .set_max_row_group_row_count(Some(1)) + .build(); + let mut out = BytesMut::new().writer(); + { + let mut writer = + ArrowWriter::try_new(&mut out, arrow_schema, Some(props)).unwrap(); + writer.write(&batch).unwrap(); + writer.finish().unwrap(); + } + + ( + Self { + schema, + post_pruning_row_groups: ExpectedPruning::None, + }, + out.into_inner().freeze(), + ) + } + + fn with_expect_all_pruned(mut self) -> Self { + self.post_pruning_row_groups = ExpectedPruning::All; + self + } + + fn with_expect_some_pruned(mut self, remaining: Vec) -> Self { + self.post_pruning_row_groups = ExpectedPruning::Some(remaining); + self + } + + async fn run(self, data: bytes::Bytes, expr: Expr) { + let Self { + schema, + post_pruning_row_groups, + } = self; + + let expr = logical2physical(&expr, &Arc::new(schema)); + let pruning_predicate = PruningPredicate::try_new( + expr, + Arc::new(Schema::new(vec![Field::new("s", DataType::Utf8, false)])), + ) + .unwrap(); + + let pruned_row_groups = + test_row_group_dictionary_pruning_predicate(data, &pruning_predicate) + .await + .unwrap(); + + post_pruning_row_groups.assert(&pruned_row_groups); + } + } + + /// Evaluates the pruning predicate on the specified row groups and + /// returns the row groups that are left, loading dictionaries exactly + /// like [`crate::opener`]'s dictionary-loading stage does. + async fn test_row_group_dictionary_pruning_predicate( + data: bytes::Bytes, + pruning_predicate: &PruningPredicate, + ) -> Result { + use datafusion_datasource::PartitionedFile; + use object_store::ObjectMeta; + + let object_meta = ObjectMeta { + location: object_store::path::Path::parse("test.parquet") + .expect("creating path"), + last_modified: chrono::DateTime::from(std::time::SystemTime::now()), + size: data.len() as u64, + e_tag: None, + version: None, + }; + let in_memory = object_store::memory::InMemory::new(); + in_memory + .put(&object_meta.location, data.into()) + .await + .expect("put parquet file into in memory object store"); + + let metrics = ExecutionPlanMetricsSet::new(); + let file_metrics = + ParquetFileMetrics::new(0, object_meta.location.as_ref(), &metrics); + let inner = + ParquetObjectReader::new(Arc::new(in_memory), object_meta.location.clone()) + .with_file_size(object_meta.size); + + let partitioned_file = PartitionedFile::new_from_meta(object_meta); + + let reader = ParquetFileReader { + inner, + file_metrics: file_metrics.clone(), + partitioned_file, + }; + let mut builder = ParquetRecordBatchStreamBuilder::new(reader).await.unwrap(); + + let access_plan = ParquetAccessPlan::new_all(builder.metadata().num_row_groups()); + let mut pruned_row_groups = RowGroupAccessPlanFilter::new(access_plan); + let literal_columns = pruning_predicate.literal_columns(); + let parquet_columns: Vec<_> = literal_columns + .into_iter() + .filter_map(|column_name| { + let (column_idx, _) = parquet_column( + builder.parquet_schema(), + pruning_predicate.schema(), + &column_name, + )?; + Some((column_name.to_string(), column_idx)) + }) + .collect::>(); + + let num_row_groups = builder.metadata().num_row_groups(); + let mut row_group_dictionaries = Vec::with_capacity(num_row_groups); + row_group_dictionaries.resize_with(num_row_groups, DictionaryStatistics::new); + + for idx in pruned_row_groups.row_group_indexes() { + let mut dict_stats = + DictionaryStatistics::with_capacity(parquet_columns.len()); + for (column_name, column_idx) in &parquet_columns { + let col_meta = builder.metadata().row_group(idx).column(*column_idx); + if !is_fully_dictionary_encoded(col_meta) { + continue; + } + let dict = match builder + .get_row_group_column_dictionary(idx, *column_idx) + .await + { + Ok(Some(dict)) => dict, + Ok(None) => continue, + Err(e) => { + log::debug!("Ignoring error reading dictionary: {e}"); + file_metrics.predicate_evaluation_errors.add(1); + continue; + } + }; + dict_stats.insert(column_name, &dict).unwrap(); + } + row_group_dictionaries[idx] = dict_stats; + } + pruned_row_groups.prune_by_dictionary( + pruning_predicate, + &file_metrics, + &row_group_dictionaries, + ); + + Ok(pruned_row_groups) + } + + #[tokio::test] + async fn test_row_group_dictionary_pruning_predicate_absent_value() { + let (test, data) = DictionaryFilterTest::new_single_value_row_groups(); + test.with_expect_all_pruned() + .run(data, col("s").eq(lit("delta"))) + .await + } + + #[tokio::test] + async fn test_row_group_dictionary_pruning_predicate_in_list_absent() { + let (test, data) = DictionaryFilterTest::new_single_value_row_groups(); + test.with_expect_all_pruned() + .run( + data, + col("s").in_list(vec![lit("delta"), lit("epsilon")], false), + ) + .await + } + + #[tokio::test] + async fn test_row_group_dictionary_pruning_predicate_present_value() { + // Each row group has exactly one distinct value, so `s = 'beta'` only + // survives in the row group whose sole value is "beta"; the other + // two row groups are provably disjoint from {"beta"}. + let (test, data) = DictionaryFilterTest::new_single_value_row_groups(); + test.with_expect_some_pruned(vec![1]) + .run(data, col("s").eq(lit("beta"))) + .await + } + + #[tokio::test] + async fn test_row_group_dictionary_pruning_predicate_not_eq_sole_value() { + // Each row group has exactly one distinct value, so `s != ` can never be true within that row group -- this exercises + // the `NOT IN`/`!=` direction that a bloom filter alone can't prove. + let (test, data) = DictionaryFilterTest::new_single_value_row_groups(); + test.with_expect_some_pruned(vec![1, 2]) + .run(data, col("s").not_eq(lit("alpha"))) + .await + } + + #[tokio::test] + async fn test_row_group_dictionary_pruning_predicate_not_in_list() { + let (test, data) = DictionaryFilterTest::new_single_value_row_groups(); + test.with_expect_some_pruned(vec![2]) + .run( + data, + col("s") + .not_eq(lit("alpha")) + .and(col("s").not_eq(lit("beta"))), + ) + .await + } + + #[tokio::test] + async fn test_row_group_dictionary_pruning_predicate_plain_fallback_not_pruned() { + // A large, high-cardinality dictionary forces PLAIN fallback, so the + // encoding-stats gate should treat the row group as not prunable by + // dictionary even though a query would otherwise be able to prune it + // via exact statistics. + let schema = Schema::new(vec![Field::new("s", DataType::Utf8, false)]); + let arrow_schema = Arc::new(schema.clone()); + let values: Vec = (0..5000).map(|i| format!("value_{i}")).collect(); + let array: ArrayRef = Arc::new(StringArray::from_iter_values( + values.iter().map(|v| v.as_str()), + )); + let batch = arrow::array::RecordBatch::try_new(arrow_schema.clone(), vec![array]) + .unwrap(); + + let props = WriterProperties::builder() + .set_dictionary_enabled(true) + .set_dictionary_page_size_limit(64) + .build(); + let mut out = BytesMut::new().writer(); + { + let mut writer = + ArrowWriter::try_new(&mut out, arrow_schema, Some(props)).unwrap(); + writer.write(&batch).unwrap(); + writer.finish().unwrap(); + } + let data = out.into_inner().freeze(); + + let test = DictionaryFilterTest { + schema, + post_pruning_row_groups: ExpectedPruning::None, + }; + // Query a value written late in the column: fallback happens almost + // immediately given the tiny size limit above, so this value is + // almost certainly PLAIN-encoded. A broken gate that used only the + // (small, abandoned) dictionary would incorrectly consider it absent + // and prune the row group, even though the value is actually present. + test.run(data, col("s").eq(lit("value_4999"))).await + } +} diff --git a/datafusion/datasource-parquet/src/metrics.rs b/datafusion/datasource-parquet/src/metrics.rs index cbdcb73196b17..b653a2d8f4977 100644 --- a/datafusion/datasource-parquet/src/metrics.rs +++ b/datafusion/datasource-parquet/src/metrics.rs @@ -49,6 +49,8 @@ pub struct ParquetFileMetrics { pub predicate_evaluation_errors: Count, /// Number of row groups pruned by bloom filters pub row_groups_pruned_bloom_filter: PruningMetrics, + /// Number of row groups pruned by exact Parquet dictionary-page values + pub row_groups_pruned_dictionary: PruningMetrics, /// Number of row groups pruned due to limit pruning. pub limit_pruned_row_groups: PruningMetrics, /// Number of row groups pruned by statistics @@ -123,6 +125,11 @@ impl ParquetFileMetrics { .with_type(MetricType::Summary) .pruning_metrics("row_groups_pruned_bloom_filter", partition); + let row_groups_pruned_dictionary = builder + .clone() + .with_type(MetricType::Summary) + .pruning_metrics("row_groups_pruned_dictionary", partition); + let limit_pruned_row_groups = builder .clone() .with_type(MetricType::Summary) @@ -215,6 +222,7 @@ impl ParquetFileMetrics { files_ranges_pruned_statistics, predicate_evaluation_errors, row_groups_pruned_bloom_filter, + row_groups_pruned_dictionary, row_groups_pruned_statistics, limit_pruned_row_groups, bytes_scanned, diff --git a/datafusion/datasource-parquet/src/mod.rs b/datafusion/datasource-parquet/src/mod.rs index 25b79a618830c..cfe8660d00cf8 100644 --- a/datafusion/datasource-parquet/src/mod.rs +++ b/datafusion/datasource-parquet/src/mod.rs @@ -27,6 +27,7 @@ pub mod access_plan; mod bloom_filter; mod decoder_projection; +mod dictionary_filter; pub mod file_format; pub mod metadata; mod metrics; @@ -49,6 +50,7 @@ mod writer; pub use access_plan::{ParquetAccessPlan, ParquetRowSelection, RowGroupAccess}; pub use bloom_filter::BloomFilterStatistics; +pub use dictionary_filter::{DictionaryStatistics, is_fully_dictionary_encoded}; pub use file_format::*; pub use metrics::ParquetFileMetrics; pub use page_filter::PagePruningAccessPlanFilter; diff --git a/datafusion/datasource-parquet/src/opener/mod.rs b/datafusion/datasource-parquet/src/opener/mod.rs index af97a192fa7ce..15622db02b154 100644 --- a/datafusion/datasource-parquet/src/opener/mod.rs +++ b/datafusion/datasource-parquet/src/opener/mod.rs @@ -25,6 +25,7 @@ use self::early_stop::EarlyStoppingStream; use self::encryption::EncryptionContext; use crate::access_plan::PreparedAccessPlan; use crate::decoder_projection::DecoderProjection; +use crate::dictionary_filter::is_fully_dictionary_encoded; use crate::page_filter::PagePruningAccessPlanFilter; use crate::push_decoder::{ DecoderBuilderConfig, PushDecoderStreamState, RgPlanEntry, RowGroupPruner, @@ -32,9 +33,9 @@ use crate::push_decoder::{ use crate::row_filter::RowFilterGenerator; use crate::row_group_filter::RowGroupAccessPlanFilter; use crate::{ - BloomFilterStatistics, Int96Coercer, ParquetAccessPlan, ParquetFileMetrics, - ParquetFileReaderFactory, ParquetRowSelection, ParquetVirtualColumn, - apply_file_schema_type_coercions, + BloomFilterStatistics, DictionaryStatistics, Int96Coercer, ParquetAccessPlan, + ParquetFileMetrics, ParquetFileReaderFactory, ParquetRowSelection, + ParquetVirtualColumn, apply_file_schema_type_coercions, }; use arrow::array::RecordBatch; use arrow::datatypes::DataType; @@ -269,6 +270,9 @@ pub(super) struct ParquetMorselizer { /// Should the bloom filter be read from parquet, if present, to skip row /// groups pub enable_bloom_filter: bool, + /// Should fully dictionary-encoded `BYTE_ARRAY` column chunks be used as + /// an exact row-group membership index, if present, to skip row groups + pub enable_dictionary_filter: bool, /// Should row group pruning be applied pub enable_row_group_stats_pruning: bool, /// Coerce INT96 timestamps to specific TimeUnit @@ -305,6 +309,7 @@ impl fmt::Debug for ParquetMorselizer { .field("preserve_order", &self.preserve_order) .field("enable_page_index", &self.enable_page_index) .field("enable_bloom_filter", &self.enable_bloom_filter) + .field("enable_dictionary_filter", &self.enable_dictionary_filter) .finish() } } @@ -350,6 +355,12 @@ impl Morselizer for ParquetMorselizer { /// PruneWithBloomFilters /// | /// v +/// LoadDictionaries +/// | +/// v +/// PruneWithDictionaries +/// | +/// v /// BuildStream /// | /// v @@ -384,6 +395,10 @@ enum ParquetOpenState { LoadBloomFilters(BoxFuture<'static, Result>), /// Pruning with preloaded Bloom Filters PruneWithBloomFilters(Box), + /// Loading Parquet dictionary pages required for row-group pruning + LoadDictionaries(BoxFuture<'static, Result>), + /// Pruning with preloaded dictionary pages + PruneWithDictionaries(Box), /// Builds the final reader stream /// /// TODO: split state as this currently does both I/O and CPU work. @@ -407,6 +422,8 @@ impl fmt::Debug for ParquetOpenState { ParquetOpenState::PruneWithStatistics(_) => "PruneWithStatistics", ParquetOpenState::LoadBloomFilters(_) => "LoadBloomFilters", ParquetOpenState::PruneWithBloomFilters(_) => "PruneWithBloomFilters", + ParquetOpenState::LoadDictionaries(_) => "LoadDictionaries", + ParquetOpenState::PruneWithDictionaries(_) => "PruneWithDictionaries", ParquetOpenState::BuildStream(_) => "BuildStream", ParquetOpenState::Ready(_) => "Ready", ParquetOpenState::Done => "Done", @@ -444,6 +461,7 @@ struct PreparedParquetOpen { force_filter_selections: bool, enable_page_index: bool, enable_bloom_filter: bool, + enable_dictionary_filter: bool, enable_row_group_stats_pruning: bool, limit: Option, coerce_int96: Option, @@ -496,6 +514,18 @@ struct BloomFiltersLoadedParquetOpen { row_group_bloom_filters: Vec, } +/// State of [`ParquetOpenState`] +/// +/// Result of loading dictionary pages needed for row-group pruning. +struct DictionariesLoadedParquetOpen { + prepared: RowGroupsPrunedParquetOpen, + /// Dictionary values loaded for each row group that remains under + /// consideration. + /// + /// indexed by parquet row-group index + row_group_dictionaries: Vec, +} + impl ParquetOpenState { /// Applies one CPU-only state transition. /// @@ -583,8 +613,17 @@ impl ParquetOpenState { ParquetOpenState::LoadBloomFilters(future) => { Ok(ParquetOpenState::LoadBloomFilters(future)) } - ParquetOpenState::PruneWithBloomFilters(loaded) => Ok( - ParquetOpenState::BuildStream(Box::new(loaded.prune_bloom_filters())), + ParquetOpenState::PruneWithBloomFilters(loaded) => { + let prepared_row_groups = loaded.prune_bloom_filters(); + Ok(ParquetOpenState::LoadDictionaries( + prepared_row_groups.load_dictionaries().boxed(), + )) + } + ParquetOpenState::LoadDictionaries(future) => { + Ok(ParquetOpenState::LoadDictionaries(future)) + } + ParquetOpenState::PruneWithDictionaries(loaded) => Ok( + ParquetOpenState::BuildStream(Box::new(loaded.prune_dictionaries())), ), ParquetOpenState::BuildStream(prepared) => { Ok(ParquetOpenState::Ready(prepared.build_stream()?)) @@ -705,6 +744,13 @@ impl MorselPlanner for ParquetMorselPlanner { ))) }))) } + ParquetOpenState::LoadDictionaries(future) => { + Ok(Some(Self::schedule_io(async move { + Ok(ParquetOpenState::PruneWithDictionaries(Box::new( + future.await?, + ))) + }))) + } ParquetOpenState::Ready(stream) => { let morsels: Vec> = vec![Box::new(ParquetStreamMorsel::new(stream))]; @@ -843,6 +889,7 @@ impl ParquetMorselizer { force_filter_selections: self.force_filter_selections, enable_page_index: self.enable_page_index, enable_bloom_filter: self.enable_bloom_filter, + enable_dictionary_filter: self.enable_dictionary_filter, enable_row_group_stats_pruning: self.enable_row_group_stats_pruning, limit: self.limit, coerce_int96: self.coerce_int96, @@ -1124,6 +1171,15 @@ impl FiltersPreparedParquetOpen { .row_groups_pruned_bloom_filter .add_matched(row_groups.remaining_row_group_count()); } + + if !prepared.enable_dictionary_filter || row_groups.is_empty() { + // Update metrics: dictionary filter unavailable, so all row + // groups are matched (not pruned) + prepared + .file_metrics + .row_groups_pruned_dictionary + .add_matched(row_groups.remaining_row_group_count()); + } } else { // Update metrics: no predicate, so all row groups are matched (not pruned) let remaining = row_groups.remaining_row_group_count(); @@ -1135,6 +1191,10 @@ impl FiltersPreparedParquetOpen { .file_metrics .row_groups_pruned_bloom_filter .add_matched(remaining); + prepared + .file_metrics + .row_groups_pruned_dictionary + .add_matched(remaining); } Ok(RowGroupsPrunedParquetOpen { @@ -1248,6 +1308,126 @@ impl RowGroupsPrunedParquetOpen { row_group_bloom_filters, }) } + + /// Load dictionary pages needed for pruning when enabled and a pruning + /// predicate exists. + /// + /// Only column chunks that are fully `BYTE_ARRAY` dictionary-encoded + /// (see [`is_fully_dictionary_encoded`]) are read: their dictionary is + /// the exact set of the row group's distinct values, so partially + /// dictionary-encoded chunks (writers fall back to `PLAIN` past a size + /// limit) are skipped rather than risk unsound pruning. + async fn load_dictionaries(mut self) -> Result { + let num_row_groups = self + .prepared + .loaded + .reader_metadata + .metadata() + .num_row_groups(); + let mut row_group_dictionaries = + vec![DictionaryStatistics::new(); num_row_groups]; + + if let Some(predicate) = + self.prepared.pruning_predicate.as_ref().map(|p| p.as_ref()) + && self.prepared.loaded.prepared.enable_dictionary_filter + && !self.row_groups.is_empty() + { + // Use the existing reader for dictionary I/O; + // replace with a fresh reader for decoding below. + let reader_metadata = self.prepared.loaded.reader_metadata.clone(); + let replacement_reader = { + let prepared = &self.prepared.loaded.prepared; + prepared.parquet_file_reader_factory.create_reader( + prepared.partition_index, + prepared.partitioned_file.clone(), + prepared.metadata_size_hint, + &prepared.metrics, + )? + }; + + let prepared = &mut self.prepared.loaded.prepared; + let mut builder = ParquetRecordBatchStreamBuilder::new_with_metadata( + mem::replace(&mut prepared.async_file_reader, replacement_reader), + reader_metadata, + ); + let parquet_columns: Vec<(String, usize)> = predicate + .literal_columns() + .into_iter() + .filter_map(|column_name| { + let parquet_schema = builder.parquet_schema(); + let (column_idx, _) = parquet_column( + parquet_schema, + &prepared.physical_file_schema, + &column_name, + )?; + Some((column_name, column_idx)) + }) + .collect(); + + let file_metadata = Arc::clone(builder.metadata()); + for idx in self.row_groups.row_group_indexes() { + let mut row_group_dictionary = + DictionaryStatistics::with_capacity(parquet_columns.len()); + for (column_name, column_idx) in &parquet_columns { + let col_meta = file_metadata.row_group(idx).column(*column_idx); + if !is_fully_dictionary_encoded(col_meta) { + continue; + } + let dictionary = match builder + .get_row_group_column_dictionary(idx, *column_idx) + .await + { + Ok(Some(dictionary)) => dictionary, + Ok(None) => continue, + Err(e) => { + debug!("Ignoring error reading dictionary page: {e}"); + prepared.file_metrics.predicate_evaluation_errors.add(1); + continue; + } + }; + if let Err(e) = row_group_dictionary.insert(column_name, &dictionary) + { + debug!("Ignoring error decoding dictionary page: {e}"); + prepared.file_metrics.predicate_evaluation_errors.add(1); + } + } + row_group_dictionaries[idx] = row_group_dictionary; + } + } + + Ok(DictionariesLoadedParquetOpen { + prepared: self, + row_group_dictionaries, + }) + } +} + +impl DictionariesLoadedParquetOpen { + /// Apply dictionary-based pruning using already loaded dictionary values. + fn prune_dictionaries(mut self) -> RowGroupsPrunedParquetOpen { + if let Some(predicate) = self + .prepared + .prepared + .pruning_predicate + .as_ref() + .map(|p| p.as_ref()) + && self + .prepared + .prepared + .loaded + .prepared + .enable_dictionary_filter + && !self.prepared.row_groups.is_empty() + { + self.prepared.row_groups.prune_by_dictionary( + predicate, + &self.prepared.prepared.loaded.prepared.file_metrics, + &self.row_group_dictionaries, + ); + } + + self.prepared + } } impl BloomFiltersLoadedParquetOpen { @@ -1749,6 +1929,7 @@ mod test { force_filter_selections: bool, enable_page_index: bool, enable_bloom_filter: bool, + enable_dictionary_filter: bool, enable_row_group_stats_pruning: bool, coerce_int96: Option, max_predicate_cache_size: Option, @@ -1857,6 +2038,7 @@ mod test { force_filter_selections: false, enable_page_index: false, enable_bloom_filter: false, + enable_dictionary_filter: false, enable_row_group_stats_pruning: false, coerce_int96: None, max_predicate_cache_size: None, @@ -2025,6 +2207,7 @@ mod test { force_filter_selections: self.force_filter_selections, enable_page_index: self.enable_page_index, enable_bloom_filter: self.enable_bloom_filter, + enable_dictionary_filter: self.enable_dictionary_filter, enable_row_group_stats_pruning: self.enable_row_group_stats_pruning, coerce_int96: self.coerce_int96, // End-to-end coercion behavior (including timezone) is diff --git a/datafusion/datasource-parquet/src/row_group_filter.rs b/datafusion/datasource-parquet/src/row_group_filter.rs index 2a2544b99b06c..2c02d9c5d48cd 100644 --- a/datafusion/datasource-parquet/src/row_group_filter.rs +++ b/datafusion/datasource-parquet/src/row_group_filter.rs @@ -20,6 +20,7 @@ use std::sync::Arc; use super::{ParquetAccessPlan, ParquetFileMetrics, RowGroupAccess}; use crate::bloom_filter::BloomFilterStatistics; +use crate::dictionary_filter::DictionaryStatistics; use arrow::array::{ArrayRef, BooleanArray, UInt64Array}; use arrow::datatypes::Schema; use datafusion_common::pruning::PruningStatistics; @@ -453,6 +454,48 @@ impl RowGroupAccessPlanFilter { } } } + + /// Prune remaining row groups using loaded Parquet dictionary pages and + /// the [`PruningPredicate`]. + /// + /// Updates this set with row groups that should not be scanned. + /// `row_group_dictionaries[idx]` contains the exact dictionary values for + /// the parquet row group at index `idx`. + /// + /// # Panics + /// if `row_group_dictionaries` does not have the same number of row groups as this set + pub fn prune_by_dictionary( + &mut self, + predicate: &PruningPredicate, + metrics: &ParquetFileMetrics, + row_group_dictionaries: &[DictionaryStatistics], + ) { + assert_eq!(row_group_dictionaries.len(), self.access_plan.len()); + for (idx, stats) in row_group_dictionaries.iter().enumerate() { + if !self.access_plan.should_scan(idx) { + continue; + } + + // Can this group be pruned? + let prune_group = match predicate.prune(stats) { + Ok(values) => !values[0], + Err(e) => { + log::debug!( + "Error evaluating row group predicate on dictionary: {e}" + ); + metrics.predicate_evaluation_errors.add(1); + false + } + }; + + if prune_group { + metrics.row_groups_pruned_dictionary.add_pruned(1); + self.access_plan.skip(idx) + } else { + metrics.row_groups_pruned_dictionary.add_matched(1); + } + } + } } /// Wraps a slice of [`RowGroupMetaData`] in a way that implements [`PruningStatistics`]. diff --git a/datafusion/datasource-parquet/src/source.rs b/datafusion/datasource-parquet/src/source.rs index 3443b08475e0d..71565b681e5e3 100644 --- a/datafusion/datasource-parquet/src/source.rs +++ b/datafusion/datasource-parquet/src/source.rs @@ -479,6 +479,23 @@ impl ParquetSource { self.table_parquet_options.global.bloom_filter_on_read } + /// If enabled, the reader will use fully dictionary-encoded `BYTE_ARRAY` + /// column chunks as an exact row-group membership index. Defaults to + /// false. + pub fn with_dictionary_filter_on_read( + mut self, + dictionary_filter_on_read: bool, + ) -> Self { + self.table_parquet_options.global.dictionary_filter_on_read = + dictionary_filter_on_read; + self + } + + /// Return the value described in [`Self::with_dictionary_filter_on_read`] + fn dictionary_filter_on_read(&self) -> bool { + self.table_parquet_options.global.dictionary_filter_on_read + } + /// Return the maximum predicate cache size, in bytes, used when /// `pushdown_filters` pub fn max_predicate_cache_size(&self) -> Option { @@ -638,6 +655,7 @@ impl FileSource for ParquetSource { force_filter_selections: self.force_filter_selections(), enable_page_index: self.enable_page_index(), enable_bloom_filter: self.bloom_filter_on_read(), + enable_dictionary_filter: self.dictionary_filter_on_read(), enable_row_group_stats_pruning: self.table_parquet_options.global.pruning, coerce_int96, coerce_int96_tz, diff --git a/datafusion/physical-expr-common/src/metrics/value.rs b/datafusion/physical-expr-common/src/metrics/value.rs index 232fefcc5f47e..cf6d0ff566681 100644 --- a/datafusion/physical-expr-common/src/metrics/value.rs +++ b/datafusion/physical-expr-common/src/metrics/value.rs @@ -1041,29 +1041,30 @@ impl MetricValue { "files_ranges_pruned_statistics" => 4, "row_groups_pruned_statistics" => 5, "row_groups_pruned_bloom_filter" => 6, - "page_index_pages_pruned" => 7, - "page_index_rows_pruned" => 8, - _ => 9, + "row_groups_pruned_dictionary" => 7, + "page_index_pages_pruned" => 8, + "page_index_rows_pruned" => 9, + _ => 10, }, - Self::SpillCount(_) => 10, - Self::SpilledBytes(_) => 11, - Self::SpilledRows(_) => 12, - Self::CurrentMemoryUsage(_) => 13, + Self::SpillCount(_) => 11, + Self::SpilledBytes(_) => 12, + Self::SpilledRows(_) => 13, + Self::CurrentMemoryUsage(_) => 14, Self::Count { name, .. } => match name.as_ref() { // This Parquet page-index metric is a plain Count because it // records pages that skipped page-index evaluation, not a // pruned/matched pair. Keep it grouped with the other // page-index pruning metrics in EXPLAIN output. - "page_index_pages_skipped_by_fully_matched" => 8, - _ => 14, + "page_index_pages_skipped_by_fully_matched" => 9, + _ => 15, }, - Self::PeakMemoryUsage { .. } => 13, - Self::Gauge { .. } => 15, - Self::Time { .. } => 16, - Self::Ratio { .. } => 17, - Self::StartTimestamp(_) => 18, // show timestamps last - Self::EndTimestamp(_) => 19, - Self::Custom { .. } => 20, + Self::PeakMemoryUsage { .. } => 14, + Self::Gauge { .. } => 16, + Self::Time { .. } => 17, + Self::Ratio { .. } => 18, + Self::StartTimestamp(_) => 19, // show timestamps last + Self::EndTimestamp(_) => 20, + Self::Custom { .. } => 21, } } diff --git a/datafusion/proto-common/proto/datafusion_common.proto b/datafusion/proto-common/proto/datafusion_common.proto index 7fff5b6b715ff..9e955bb7919b3 100644 --- a/datafusion/proto-common/proto/datafusion_common.proto +++ b/datafusion/proto-common/proto/datafusion_common.proto @@ -572,6 +572,7 @@ message ParquetOptions { bool schema_force_view_types = 28; // default = false bool binary_as_string = 29; // default = false bool skip_arrow_metadata = 30; // default = false + bool dictionary_filter_on_read = 38; // default = false oneof metadata_size_hint_opt { uint64 metadata_size_hint = 4; diff --git a/datafusion/proto-common/src/from_proto/mod.rs b/datafusion/proto-common/src/from_proto/mod.rs index 97cc9af230105..5e925a91f28b8 100644 --- a/datafusion/proto-common/src/from_proto/mod.rs +++ b/datafusion/proto-common/src/from_proto/mod.rs @@ -1102,6 +1102,7 @@ impl TryFrom<&protobuf::ParquetOptions> for ParquetOptions { }) .unwrap_or(None), bloom_filter_on_read: value.bloom_filter_on_read, + dictionary_filter_on_read: value.dictionary_filter_on_read, bloom_filter_on_write: value.bloom_filter_on_write, bloom_filter_fpp: value.clone() .bloom_filter_fpp_opt diff --git a/datafusion/proto-common/src/generated/pbjson.rs b/datafusion/proto-common/src/generated/pbjson.rs index 963faa5a3e9cb..c5ad6a7f705af 100644 --- a/datafusion/proto-common/src/generated/pbjson.rs +++ b/datafusion/proto-common/src/generated/pbjson.rs @@ -6400,6 +6400,9 @@ impl serde::Serialize for ParquetOptions { if self.skip_arrow_metadata { len += 1; } + if self.dictionary_filter_on_read { + len += 1; + } if self.dictionary_page_size_limit != 0 { len += 1; } @@ -6514,6 +6517,9 @@ impl serde::Serialize for ParquetOptions { if self.skip_arrow_metadata { struct_ser.serialize_field("skipArrowMetadata", &self.skip_arrow_metadata)?; } + if self.dictionary_filter_on_read { + struct_ser.serialize_field("dictionaryFilterOnRead", &self.dictionary_filter_on_read)?; + } if self.dictionary_page_size_limit != 0 { #[allow(clippy::needless_borrow)] #[allow(clippy::needless_borrows_for_generic_args)] @@ -6681,6 +6687,8 @@ impl<'de> serde::Deserialize<'de> for ParquetOptions { "binaryAsString", "skip_arrow_metadata", "skipArrowMetadata", + "dictionary_filter_on_read", + "dictionaryFilterOnRead", "dictionary_page_size_limit", "dictionaryPageSizeLimit", "data_page_row_count_limit", @@ -6736,6 +6744,7 @@ impl<'de> serde::Deserialize<'de> for ParquetOptions { SchemaForceViewTypes, BinaryAsString, SkipArrowMetadata, + DictionaryFilterOnRead, DictionaryPageSizeLimit, DataPageRowCountLimit, MaxRowGroupSize, @@ -6792,6 +6801,7 @@ impl<'de> serde::Deserialize<'de> for ParquetOptions { "schemaForceViewTypes" | "schema_force_view_types" => Ok(GeneratedField::SchemaForceViewTypes), "binaryAsString" | "binary_as_string" => Ok(GeneratedField::BinaryAsString), "skipArrowMetadata" | "skip_arrow_metadata" => Ok(GeneratedField::SkipArrowMetadata), + "dictionaryFilterOnRead" | "dictionary_filter_on_read" => Ok(GeneratedField::DictionaryFilterOnRead), "dictionaryPageSizeLimit" | "dictionary_page_size_limit" => Ok(GeneratedField::DictionaryPageSizeLimit), "dataPageRowCountLimit" | "data_page_row_count_limit" => Ok(GeneratedField::DataPageRowCountLimit), "maxRowGroupSize" | "max_row_group_size" => Ok(GeneratedField::MaxRowGroupSize), @@ -6846,6 +6856,7 @@ impl<'de> serde::Deserialize<'de> for ParquetOptions { let mut schema_force_view_types__ = None; let mut binary_as_string__ = None; let mut skip_arrow_metadata__ = None; + let mut dictionary_filter_on_read__ = None; let mut dictionary_page_size_limit__ = None; let mut data_page_row_count_limit__ = None; let mut max_row_group_size__ = None; @@ -6976,6 +6987,12 @@ impl<'de> serde::Deserialize<'de> for ParquetOptions { } skip_arrow_metadata__ = Some(map_.next_value()?); } + GeneratedField::DictionaryFilterOnRead => { + if dictionary_filter_on_read__.is_some() { + return Err(serde::de::Error::duplicate_field("dictionaryFilterOnRead")); + } + dictionary_filter_on_read__ = Some(map_.next_value()?); + } GeneratedField::DictionaryPageSizeLimit => { if dictionary_page_size_limit__.is_some() { return Err(serde::de::Error::duplicate_field("dictionaryPageSizeLimit")); @@ -7110,6 +7127,7 @@ impl<'de> serde::Deserialize<'de> for ParquetOptions { schema_force_view_types: schema_force_view_types__.unwrap_or_default(), binary_as_string: binary_as_string__.unwrap_or_default(), skip_arrow_metadata: skip_arrow_metadata__.unwrap_or_default(), + dictionary_filter_on_read: dictionary_filter_on_read__.unwrap_or_default(), dictionary_page_size_limit: dictionary_page_size_limit__.unwrap_or_default(), data_page_row_count_limit: data_page_row_count_limit__.unwrap_or_default(), max_row_group_size: max_row_group_size__.unwrap_or_default(), diff --git a/datafusion/proto-common/src/generated/prost.rs b/datafusion/proto-common/src/generated/prost.rs index 93b97c4f1376c..c4db891e8b7ac 100644 --- a/datafusion/proto-common/src/generated/prost.rs +++ b/datafusion/proto-common/src/generated/prost.rs @@ -856,6 +856,9 @@ pub struct ParquetOptions { /// default = false #[prost(bool, tag = "30")] pub skip_arrow_metadata: bool, + /// default = false + #[prost(bool, tag = "38")] + pub dictionary_filter_on_read: bool, #[prost(uint64, tag = "12")] pub dictionary_page_size_limit: u64, #[prost(uint64, tag = "18")] diff --git a/datafusion/proto-common/src/to_proto/mod.rs b/datafusion/proto-common/src/to_proto/mod.rs index d2e1ca50c812d..4b1fa5f4d4f11 100644 --- a/datafusion/proto-common/src/to_proto/mod.rs +++ b/datafusion/proto-common/src/to_proto/mod.rs @@ -926,6 +926,7 @@ impl TryFrom<&ParquetOptions> for protobuf::ParquetOptions { data_page_row_count_limit: value.data_page_row_count_limit as u64, encoding_opt: value.encoding.clone().map(protobuf::parquet_options::EncodingOpt::Encoding), bloom_filter_on_read: value.bloom_filter_on_read, + dictionary_filter_on_read: value.dictionary_filter_on_read, bloom_filter_on_write: value.bloom_filter_on_write, bloom_filter_fpp_opt: value.bloom_filter_fpp.map(protobuf::parquet_options::BloomFilterFppOpt::BloomFilterFpp), bloom_filter_ndv_opt: value.bloom_filter_ndv.map(protobuf::parquet_options::BloomFilterNdvOpt::BloomFilterNdv), diff --git a/datafusion/proto-models/src/generated/datafusion_proto_common.rs b/datafusion/proto-models/src/generated/datafusion_proto_common.rs index 93b97c4f1376c..c4db891e8b7ac 100644 --- a/datafusion/proto-models/src/generated/datafusion_proto_common.rs +++ b/datafusion/proto-models/src/generated/datafusion_proto_common.rs @@ -856,6 +856,9 @@ pub struct ParquetOptions { /// default = false #[prost(bool, tag = "30")] pub skip_arrow_metadata: bool, + /// default = false + #[prost(bool, tag = "38")] + pub dictionary_filter_on_read: bool, #[prost(uint64, tag = "12")] pub dictionary_page_size_limit: u64, #[prost(uint64, tag = "18")] diff --git a/datafusion/proto/src/logical_plan/file_formats.rs b/datafusion/proto/src/logical_plan/file_formats.rs index 8940b16bf83f5..48f71837e9dee 100644 --- a/datafusion/proto/src/logical_plan/file_formats.rs +++ b/datafusion/proto/src/logical_plan/file_formats.rs @@ -436,6 +436,7 @@ mod parquet { parquet_options::EncodingOpt::Encoding(encoding) }), bloom_filter_on_read: global_options.global.bloom_filter_on_read, + dictionary_filter_on_read: global_options.global.dictionary_filter_on_read, bloom_filter_on_write: global_options.global.bloom_filter_on_write, bloom_filter_fpp_opt: global_options.global.bloom_filter_fpp.map(|fpp| { parquet_options::BloomFilterFppOpt::BloomFilterFpp(fpp) @@ -590,6 +591,7 @@ mod parquet { } }), bloom_filter_on_read: proto.bloom_filter_on_read, + dictionary_filter_on_read: proto.dictionary_filter_on_read, bloom_filter_on_write: proto.bloom_filter_on_write, bloom_filter_fpp: proto .bloom_filter_fpp_opt diff --git a/datafusion/proto/tests/cases/roundtrip_logical_plan.rs b/datafusion/proto/tests/cases/roundtrip_logical_plan.rs index 74f7253386764..9f39749595f48 100644 --- a/datafusion/proto/tests/cases/roundtrip_logical_plan.rs +++ b/datafusion/proto/tests/cases/roundtrip_logical_plan.rs @@ -867,6 +867,7 @@ async fn roundtrip_logical_plan_copy_to_writer_options() -> Result<()> { let mut parquet_format = table_options.parquet; parquet_format.global.bloom_filter_on_read = true; + parquet_format.global.dictionary_filter_on_read = true; parquet_format.global.created_by = "DataFusion Test".to_string(); parquet_format.global.writer_version = DFParquetWriterVersion::V2_0; parquet_format.global.write_batch_size = 111; @@ -1256,6 +1257,7 @@ async fn roundtrip_default_codec_parquet() -> Result<()> { TableOptions::default_from_session_config(ctx.state().config_options()); let mut parquet_format = table_options.parquet; parquet_format.global.bloom_filter_on_read = true; + parquet_format.global.dictionary_filter_on_read = true; parquet_format.global.created_by = "DefaultCodecTest".to_string(); let file_type = format_as_file_type(Arc::new( @@ -1290,6 +1292,7 @@ async fn roundtrip_default_codec_parquet() -> Result<()> { .unwrap(); let decoded = pq.options.as_ref().unwrap(); assert!(decoded.global.bloom_filter_on_read); + assert!(decoded.global.dictionary_filter_on_read); assert_eq!("DefaultCodecTest", decoded.global.created_by); } _ => panic!("Expected CopyTo plan"), diff --git a/datafusion/sqllogictest/test_files/dynamic_filter_pushdown_config.slt b/datafusion/sqllogictest/test_files/dynamic_filter_pushdown_config.slt index c58047c4abe10..18e13d031dfb4 100644 --- a/datafusion/sqllogictest/test_files/dynamic_filter_pushdown_config.slt +++ b/datafusion/sqllogictest/test_files/dynamic_filter_pushdown_config.slt @@ -104,7 +104,7 @@ Plan with Metrics 03)----ProjectionExec: expr=[id@0 as id, value@1 as v, value@1 + id@0 as name], metrics=[output_rows=10, ] 04)------FilterExec: value@1 > 3, metrics=[output_rows=10, , selectivity=100% (10/10)] 05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, metrics=[output_rows=10, ] -06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/dynamic_filter_pushdown_config/test_data.parquet]]}, projection=[id, value], file_type=parquet, predicate=value@1 > 3 AND DynamicFilter [ value@1 IS NULL OR value@1 > 800 ], dynamic_rg_pruning=eligible, pruning_predicate=value_null_count@1 != row_count@2 AND value_max@0 > 3 AND (value_null_count@1 > 0 OR value_null_count@1 != row_count@2 AND value_max@0 > 800), required_guarantees=[], metrics=[output_rows=10, elapsed_compute=, output_bytes=80.0 B, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched -> 1 fully matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=210, page_index_load_skipped=1, row_groups_pruned_dynamic_filter=0, metadata_load_time=, scan_efficiency_ratio=18.31% (210/1.15 K)] +06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/dynamic_filter_pushdown_config/test_data.parquet]]}, projection=[id, value], file_type=parquet, predicate=value@1 > 3 AND DynamicFilter [ value@1 IS NULL OR value@1 > 800 ], dynamic_rg_pruning=eligible, pruning_predicate=value_null_count@1 != row_count@2 AND value_max@0 > 3 AND (value_null_count@1 > 0 OR value_null_count@1 != row_count@2 AND value_max@0 > 800), required_guarantees=[], metrics=[output_rows=10, elapsed_compute=, output_bytes=80.0 B, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched -> 1 fully matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=210, page_index_load_skipped=1, row_groups_pruned_dynamic_filter=0, metadata_load_time=, scan_efficiency_ratio=18.31% (210/1.15 K)] statement ok set datafusion.explain.analyze_level = dev; diff --git a/datafusion/sqllogictest/test_files/dynamic_row_group_pruning.slt b/datafusion/sqllogictest/test_files/dynamic_row_group_pruning.slt index 2149cacfc0a55..06221c7915502 100644 --- a/datafusion/sqllogictest/test_files/dynamic_row_group_pruning.slt +++ b/datafusion/sqllogictest/test_files/dynamic_row_group_pruning.slt @@ -98,7 +98,7 @@ explain analyze select v from t order by v desc limit 3; ---- Plan with Metrics 01)SortExec: TopK(fetch=3), expr=[v@0 DESC], preserve_partitioning=[false], filter=[v@0 IS NULL OR v@0 > 12], metrics=[output_rows=3, elapsed_compute=, output_bytes=] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/dynamic_row_group_pruning/data.parquet]]}, projection=[v], file_type=parquet, predicate=DynamicFilter [ v@0 IS NULL OR v@0 > 12 ], sort_order_for_reorder=[v@0 DESC], reverse_row_groups=true, dynamic_rg_pruning=eligible, pruning_predicate=v_null_count@0 > 0 OR v_null_count@0 != row_count@2 AND v_max@1 > 12, required_guarantees=[], metrics=[output_rows=3, elapsed_compute=, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=5 total → 5 matched, row_groups_pruned_bloom_filter=5 total → 5 matched, page_index_pages_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, row_groups_pruned_dynamic_filter=4, metadata_load_time=, scan_efficiency_ratio=] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/dynamic_row_group_pruning/data.parquet]]}, projection=[v], file_type=parquet, predicate=DynamicFilter [ v@0 IS NULL OR v@0 > 12 ], sort_order_for_reorder=[v@0 DESC], reverse_row_groups=true, dynamic_rg_pruning=eligible, pruning_predicate=v_null_count@0 > 0 OR v_null_count@0 != row_count@2 AND v_max@1 > 12, required_guarantees=[], metrics=[output_rows=3, elapsed_compute=, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=5 total → 5 matched, row_groups_pruned_bloom_filter=5 total → 5 matched, row_groups_pruned_dictionary=5 total → 5 matched, page_index_pages_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, row_groups_pruned_dynamic_filter=4, metadata_load_time=, scan_efficiency_ratio=] statement ok drop table t; diff --git a/datafusion/sqllogictest/test_files/explain_analyze.slt b/datafusion/sqllogictest/test_files/explain_analyze.slt index d64efe80ccae5..db5db835bb3fb 100644 --- a/datafusion/sqllogictest/test_files/explain_analyze.slt +++ b/datafusion/sqllogictest/test_files/explain_analyze.slt @@ -247,7 +247,7 @@ explain analyze select * from cat_tracking where species > 'M' AND s >= 50 order ---- Plan with Metrics 01)SortExec: TopK(fetch=3), expr=[species@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[species@0 < Nlpine Sheep], metrics=[output_rows=3] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, scan_efficiency_ratio=21.75% (485/2.23 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, scan_efficiency_ratio=21.75% (485/2.23 K)] statement ok reset datafusion.explain.analyze_categories; @@ -262,7 +262,7 @@ explain analyze select * from cat_tracking where species > 'M' AND s >= 50 order ---- Plan with Metrics 01)SortExec: TopK(fetch=3), expr=[species@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[species@0 < Nlpine Sheep], metrics=[output_rows=3, output_bytes=] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, scan_efficiency_ratio=] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, scan_efficiency_ratio=] statement ok reset datafusion.explain.analyze_categories; @@ -277,7 +277,7 @@ explain analyze select * from cat_tracking where species > 'M' AND s >= 50 order ---- Plan with Metrics 01)SortExec: TopK(fetch=3), expr=[species@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[species@0 < Nlpine Sheep], metrics=[output_rows=3, output_bytes=] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, scan_efficiency_ratio=] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, scan_efficiency_ratio=] statement ok reset datafusion.explain.analyze_categories; @@ -559,7 +559,7 @@ EXPLAIN (ANALYZE, METRICS 'rows', LEVEL summary) select * from cat_tracking wher ---- Plan with Metrics 01)SortExec: TopK(fetch=3), expr=[species@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[species@0 < Nlpine Sheep], metrics=[output_rows=3] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, scan_efficiency_ratio=21.75% (485/2.23 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, scan_efficiency_ratio=21.75% (485/2.23 K)] # ---- Quoted-string METRICS with multiple categories ---- @@ -568,7 +568,7 @@ EXPLAIN (ANALYZE, METRICS 'rows,bytes', LEVEL summary) select * from cat_trackin ---- Plan with Metrics 01)SortExec: TopK(fetch=3), expr=[species@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[species@0 < Nlpine Sheep], metrics=[output_rows=3, output_bytes=] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, scan_efficiency_ratio=] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, scan_efficiency_ratio=] # ---- (METRICS 'timing', LEVEL summary) — timing metrics only ---- @@ -588,7 +588,7 @@ EXPLAIN (ANALYZE, METRICS 'rows,bytes', TIMING off, LEVEL summary) select * from ---- Plan with Metrics 01)SortExec: TopK(fetch=3), expr=[species@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[species@0 < Nlpine Sheep], metrics=[output_rows=3, output_bytes=] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, scan_efficiency_ratio=] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, scan_efficiency_ratio=] # ---- TIMING sugar: `METRICS 'rows', TIMING on` ↔ rows + timing ---- @@ -597,7 +597,7 @@ EXPLAIN (ANALYZE, METRICS 'rows', TIMING on, LEVEL summary) select * from cat_tr ---- Plan with Metrics 01)SortExec: TopK(fetch=3), expr=[species@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[species@0 < Nlpine Sheep], metrics=[output_rows=3, elapsed_compute=] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, elapsed_compute=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, metadata_load_time=, scan_efficiency_ratio=21.75% (485/2.23 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, elapsed_compute=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, metadata_load_time=, scan_efficiency_ratio=21.75% (485/2.23 K)] # ---- SUMMARY sugar: `SUMMARY on` ↔ `LEVEL summary` ---- # Equivalent to METRICS 'rows', LEVEL summary above. @@ -607,7 +607,7 @@ EXPLAIN (ANALYZE, METRICS 'rows', SUMMARY on) select * from cat_tracking where s ---- Plan with Metrics 01)SortExec: TopK(fetch=3), expr=[species@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[species@0 < Nlpine Sheep], metrics=[output_rows=3] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, scan_efficiency_ratio=21.75% (485/2.23 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, scan_efficiency_ratio=21.75% (485/2.23 K)] # ---- Statement option overrides session config ---- # Session says 'timing' but statement-level `METRICS 'rows'` wins. @@ -620,7 +620,7 @@ EXPLAIN (ANALYZE, METRICS 'rows', LEVEL summary) select * from cat_tracking wher ---- Plan with Metrics 01)SortExec: TopK(fetch=3), expr=[species@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[species@0 < Nlpine Sheep], metrics=[output_rows=3] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, scan_efficiency_ratio=21.75% (485/2.23 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, scan_efficiency_ratio=21.75% (485/2.23 K)] # ---- pgjson format: structural golden with no metrics ---- @@ -682,7 +682,7 @@ EXPLAIN (ANALYZE, METRICS rows, LEVEL summary) select * from cat_tracking where ---- Plan with Metrics 01)SortExec: TopK(fetch=3), expr=[species@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[species@0 < Nlpine Sheep], metrics=[output_rows=3] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, scan_efficiency_ratio=21.75% (485/2.23 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, scan_efficiency_ratio=21.75% (485/2.23 K)] statement ok reset datafusion.sql_parser.dialect; diff --git a/datafusion/sqllogictest/test_files/information_schema.slt b/datafusion/sqllogictest/test_files/information_schema.slt index 77acaa4747f9d..102efa51cbec5 100644 --- a/datafusion/sqllogictest/test_files/information_schema.slt +++ b/datafusion/sqllogictest/test_files/information_schema.slt @@ -249,6 +249,7 @@ datafusion.execution.parquet.created_by datafusion datafusion.execution.parquet.data_page_row_count_limit 20000 datafusion.execution.parquet.data_pagesize_limit 1048576 datafusion.execution.parquet.dictionary_enabled true +datafusion.execution.parquet.dictionary_filter_on_read false datafusion.execution.parquet.dictionary_page_size_limit 1048576 datafusion.execution.parquet.enable_page_index true datafusion.execution.parquet.encoding NULL @@ -408,6 +409,7 @@ datafusion.execution.parquet.created_by datafusion (writing) Sets "created by" p datafusion.execution.parquet.data_page_row_count_limit 20000 (writing) Sets best effort maximum number of rows in data page datafusion.execution.parquet.data_pagesize_limit 1048576 (writing) Sets best effort maximum size of data page in bytes datafusion.execution.parquet.dictionary_enabled true (writing) Sets if dictionary encoding is enabled. If NULL, uses default parquet writer setting +datafusion.execution.parquet.dictionary_filter_on_read false (reading) Use fully dictionary-encoded `BYTE_ARRAY` (Utf8/Binary) column chunks as an exact row-group membership index when reading parquet files. Unlike bloom filters, a fully dictionary-encoded column chunk's dictionary is the exact, complete set of the row group's distinct values, so this can prune both `IN`/`=` and `NOT IN`/`!=` predicates. Only column chunks whose page encoding statistics prove every data page came from the dictionary are used; chunks that fell back to `PLAIN` encoding are ignored. datafusion.execution.parquet.dictionary_page_size_limit 1048576 (writing) Sets best effort maximum dictionary page size, in bytes datafusion.execution.parquet.enable_page_index true (reading) If true, reads the Parquet data page level metadata (the Page Index), if present, to reduce the I/O and number of rows decoded. datafusion.execution.parquet.encoding NULL (writing) Sets default encoding for any column. Valid values are: plain, plain_dictionary, rle, bit_packed, delta_binary_packed, delta_length_byte_array, delta_byte_array, rle_dictionary, and byte_stream_split. These values are not case sensitive. If NULL, uses default parquet writer setting diff --git a/datafusion/sqllogictest/test_files/limit_pruning.slt b/datafusion/sqllogictest/test_files/limit_pruning.slt index 4ef0b5c74f3e7..ae1d64f88aae0 100644 --- a/datafusion/sqllogictest/test_files/limit_pruning.slt +++ b/datafusion/sqllogictest/test_files/limit_pruning.slt @@ -63,7 +63,7 @@ set datafusion.explain.analyze_level = summary; query TT explain analyze select * from tracking_data where species > 'M' AND s >= 50 limit 3; ---- -Plan with Metrics DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/limit_pruning/data.parquet]]}, projection=[species, s], limit=3, file_type=parquet, predicate=species@0 > M AND s@1 >= 50, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50, required_guarantees=[], metrics=[output_rows=3, elapsed_compute=, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, page_index_pages_pruned=0 total → 0 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=2 total → 0 matched, bytes_scanned=, metadata_load_time=, scan_efficiency_ratio= (159/2.23 K)] +Plan with Metrics DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/limit_pruning/data.parquet]]}, projection=[species, s], limit=3, file_type=parquet, predicate=species@0 > M AND s@1 >= 50, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50, required_guarantees=[], metrics=[output_rows=3, elapsed_compute=, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 3 matched, page_index_pages_pruned=0 total → 0 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=2 total → 0 matched, bytes_scanned=, metadata_load_time=, scan_efficiency_ratio= (159/2.23 K)] statement ok CREATE TABLE fully_matched_limit_source AS VALUES @@ -120,7 +120,7 @@ explain analyze select * from tracking_data where species > 'M' AND s >= 50 orde ---- Plan with Metrics 01)SortExec: TopK(fetch=3), expr=[species@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[species@0 < Nlpine Sheep], metrics=[output_rows=3, elapsed_compute=, output_bytes=] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/limit_pruning/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, elapsed_compute=, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, row_groups_pruned_dynamic_filter=0, metadata_load_time=, scan_efficiency_ratio= (/)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/limit_pruning/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, elapsed_compute=, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, row_groups_pruned_dynamic_filter=0, metadata_load_time=, scan_efficiency_ratio= (/)] statement ok drop table tracking_data; diff --git a/datafusion/sqllogictest/test_files/parquet_dictionary_pruning.slt b/datafusion/sqllogictest/test_files/parquet_dictionary_pruning.slt new file mode 100644 index 0000000000000..504954677c388 --- /dev/null +++ b/datafusion/sqllogictest/test_files/parquet_dictionary_pruning.slt @@ -0,0 +1,201 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at + +# http://www.apache.org/licenses/LICENSE-2.0 + +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +# End-to-end SLT for **row-group pruning by Parquet dictionaries** +# (`datafusion.execution.parquet.dictionary_filter_on_read`). +# +# Some stores use a low-cardinality string column's Parquet dictionary as a +# makeshift row-group index: when the whole column chunk is dictionary +# encoded, the dictionary page is the exact, complete set of that row +# group's distinct values, so a row group can be skipped without reading any +# data pages if a required value isn't in its dictionary. Unlike bloom +# filters, this is exact in both directions, so it can also prune `!=`/`NOT +# IN` when a row group's dictionary is a subset of the excluded values. +# +# Builds a 3-row-group parquet file where each row group has exactly one +# distinct value of `s`: +# RG 0: "alpha", RG 1: "beta", RG 2: "gamma" + +statement ok +set datafusion.explain.analyze_level = summary; + +statement ok +CREATE TABLE source_data AS VALUES + ('alpha'), ('alpha'), ('alpha'), + ('beta'), ('beta'), ('beta'), + ('gamma'), ('gamma'), ('gamma'); + +statement ok +COPY (SELECT column1 as s FROM source_data) +TO 'test_files/scratch/parquet_dictionary_pruning/data.parquet' +STORED AS PARQUET +OPTIONS ( + 'format.max_row_group_size' '3' +); + +statement ok +drop table source_data; + +statement ok +CREATE EXTERNAL TABLE t +STORED AS PARQUET +LOCATION 'test_files/scratch/parquet_dictionary_pruning/data.parquet'; + +# Sanity: query returns the right rows regardless of the config below. +query T rowsort +SELECT s FROM t; +---- +alpha +alpha +alpha +beta +beta +beta +gamma +gamma +gamma + +######## +# Off by default: no dictionary pruning happens without opting in. +######## + +query error DataFusion error: Error during planning: SHOW \[VARIABLE\] is not supported unless information_schema is enabled +SHOW datafusion.execution.parquet.dictionary_filter_on_read + +query TT +explain analyze select s from t where s = 'delta'; +---- +Plan with Metrics +01)FilterExec: s@0 = delta, metrics=[output_rows=0, elapsed_compute=, selectivity=N/A (0/0)] +02)--RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, metrics=[output_rows=0, elapsed_compute=] +03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_dictionary_pruning/data.parquet]]}, projection=[s], file_type=parquet, predicate=s@0 = delta, pruning_predicate=s_null_count@2 != row_count@3 AND s_min@0 <= delta AND delta <= s_max@1, required_guarantees=[s in (delta)], metrics=[output_rows=0, elapsed_compute=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=3 total → 0 matched, row_groups_pruned_bloom_filter=0 total → 0 matched, row_groups_pruned_dictionary=0 total → 0 matched, page_index_pages_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, page_index_load_skipped=1, row_groups_pruned_dynamic_filter=0, metadata_load_time=] + +statement ok +set datafusion.execution.parquet.dictionary_filter_on_read = true; + +######## +# `IN`/`=` direction: a value absent from every row group's dictionary +# prunes all of them without reading any data pages. +######## + +query TT +explain analyze select s from t where s = 'delta'; +---- +Plan with Metrics +01)FilterExec: s@0 = delta, metrics=[output_rows=0, elapsed_compute=, selectivity=N/A (0/0)] +02)--RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, metrics=[output_rows=0, elapsed_compute=] +03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_dictionary_pruning/data.parquet]]}, projection=[s], file_type=parquet, predicate=s@0 = delta, pruning_predicate=s_null_count@2 != row_count@3 AND s_min@0 <= delta AND delta <= s_max@1, required_guarantees=[s in (delta)], metrics=[output_rows=0, elapsed_compute=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=3 total → 0 matched, row_groups_pruned_bloom_filter=0 total → 0 matched, row_groups_pruned_dictionary=0 total → 0 matched, page_index_pages_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, page_index_load_skipped=1, row_groups_pruned_dynamic_filter=0, metadata_load_time=] + +query TT +explain analyze select s from t where s IN ('delta', 'epsilon'); +---- +Plan with Metrics +01)FilterExec: s@0 = delta OR s@0 = epsilon, metrics=[output_rows=0, elapsed_compute=, selectivity=N/A (0/0)] +02)--RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, metrics=[output_rows=0, elapsed_compute=] +03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_dictionary_pruning/data.parquet]]}, projection=[s], file_type=parquet, predicate=s@0 = delta OR s@0 = epsilon, pruning_predicate=s_null_count@2 != row_count@3 AND s_min@0 <= delta AND delta <= s_max@1 OR s_null_count@2 != row_count@3 AND s_min@0 <= epsilon AND epsilon <= s_max@1, required_guarantees=[s in (delta, epsilon)], metrics=[output_rows=0, elapsed_compute=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=3 total → 0 matched, row_groups_pruned_bloom_filter=0 total → 0 matched, row_groups_pruned_dictionary=0 total → 0 matched, page_index_pages_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, page_index_load_skipped=1, row_groups_pruned_dynamic_filter=0, metadata_load_time=] + +# A value that IS present in one row group's dictionary is not pruned. +query T +select s from t where s = 'beta'; +---- +beta +beta +beta + +######## +# `!=`/`NOT IN` direction: exact dictionaries can prune this too, unlike +# bloom filters. Each row group has exactly one distinct value, so `s != +# 'alpha'` can never be true for rows in the row group whose only value is +# "alpha". +######## + +query T rowsort +select s from t where s != 'alpha'; +---- +beta +beta +beta +gamma +gamma +gamma + +query TT +explain analyze select s from t where s != 'alpha'; +---- +Plan with Metrics +01)FilterExec: s@0 != alpha, metrics=[output_rows=6, elapsed_compute=, selectivity=100% (6/6)] +02)--RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, metrics=[output_rows=6, elapsed_compute=] +03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_dictionary_pruning/data.parquet]]}, projection=[s], file_type=parquet, predicate=s@0 != alpha, pruning_predicate=s_null_count@2 != row_count@3 AND (s_min@0 != alpha OR alpha != s_max@1), required_guarantees=[s not in (alpha)], metrics=[output_rows=6, elapsed_compute=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=3 total → 2 matched -> 2 fully matched, row_groups_pruned_bloom_filter=2 total → 2 matched, row_groups_pruned_dictionary=2 total → 2 matched, page_index_pages_pruned=0 total → 0 matched, page_index_pages_skipped_by_fully_matched=2, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, page_index_load_skipped=1, row_groups_pruned_dynamic_filter=0, metadata_load_time=] + +######## +# Encoding-stats gate: a column with a large, high-cardinality dictionary +# falls back to `PLAIN` for some data pages. Since not every value is +# guaranteed to be drawn from the dictionary, it must not be used for +# pruning even though a value may look absent. +######## + +statement ok +CREATE TABLE wide_source_data AS +SELECT 'value_' || i AS s FROM generate_series(0, 4999) t(i); + +statement ok +COPY wide_source_data +TO 'test_files/scratch/parquet_dictionary_pruning/wide.parquet' +STORED AS PARQUET +OPTIONS ( + 'format.dictionary_page_size_limit' '64' +); + +statement ok +drop table wide_source_data; + +statement ok +CREATE EXTERNAL TABLE wide +STORED AS PARQUET +LOCATION 'test_files/scratch/parquet_dictionary_pruning/wide.parquet'; + +# `value_4999` is written late enough that the tiny dictionary size limit +# above has already forced a `PLAIN` fallback for it; it must still be +# found (not incorrectly pruned). +query T +select s from wide where s = 'value_4999'; +---- +value_4999 + +query TT +explain analyze select s from wide where s = 'value_4999'; +---- +Plan with Metrics +01)FilterExec: s@0 = value_4999, metrics=[output_rows=1, elapsed_compute=, selectivity=0.02% (1/5.00 K)] +02)--RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, metrics=[output_rows=5.00 K, elapsed_compute=] +03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_dictionary_pruning/wide.parquet]]}, projection=[s], file_type=parquet, predicate=s@0 = value_4999, pruning_predicate=s_null_count@2 != row_count@3 AND s_min@0 <= value_4999 AND value_4999 <= s_max@1, required_guarantees=[s in (value_4999)], metrics=[output_rows=5.00 K, elapsed_compute=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=2 total → 2 matched, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, row_groups_pruned_dynamic_filter=0, metadata_load_time=] + +statement ok +drop table wide; + +######## +# Clean up after the test +######## + +statement ok +drop table t; + +statement ok +RESET datafusion.execution.parquet.dictionary_filter_on_read; + +statement ok +RESET datafusion.explain.analyze_level; diff --git a/datafusion/sqllogictest/test_files/push_down_filter_parquet.slt b/datafusion/sqllogictest/test_files/push_down_filter_parquet.slt index e879947e324bb..3072ac21aafa7 100644 --- a/datafusion/sqllogictest/test_files/push_down_filter_parquet.slt +++ b/datafusion/sqllogictest/test_files/push_down_filter_parquet.slt @@ -206,7 +206,7 @@ EXPLAIN ANALYZE SELECT t FROM topk_pushdown ORDER BY t * t LIMIT 10; ---- Plan with Metrics 01)SortExec: TopK(fetch=10), expr=[t@0 * t@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[t@0 * t@0 < 1884329474306198481], metrics=[output_rows=10, output_batches=1, row_replacements=10] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_pushdown.parquet]]}, projection=[t], output_ordering=[t@0 ASC NULLS LAST], file_type=parquet, predicate=DynamicFilter [ t@0 * t@0 < 1884329474306198481 ], dynamic_rg_pruning=eligible, metrics=[output_rows=128, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=782 total → 782 matched, row_groups_pruned_bloom_filter=782 total → 782 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=128, pushdown_rows_pruned=99.87 K, predicate_cache_inner_records=128, predicate_cache_records=128, scan_efficiency_ratio=64.87% (258.7 K/398.8 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_pushdown.parquet]]}, projection=[t], output_ordering=[t@0 ASC NULLS LAST], file_type=parquet, predicate=DynamicFilter [ t@0 * t@0 < 1884329474306198481 ], dynamic_rg_pruning=eligible, metrics=[output_rows=128, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=782 total → 782 matched, row_groups_pruned_bloom_filter=782 total → 782 matched, row_groups_pruned_dictionary=782 total → 782 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=128, pushdown_rows_pruned=99.87 K, predicate_cache_inner_records=128, predicate_cache_records=128, scan_efficiency_ratio=64.87% (258.7 K/398.8 K)] statement ok reset datafusion.explain.analyze_categories; @@ -268,7 +268,7 @@ EXPLAIN ANALYZE SELECT * FROM topk_single_col ORDER BY b DESC LIMIT 1; ---- Plan with Metrics 01)SortExec: TopK(fetch=1), expr=[b@1 DESC], preserve_partitioning=[false], filter=[b@1 IS NULL OR b@1 > bd], metrics=[output_rows=1, output_batches=1, row_replacements=1] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_single_col.parquet]]}, projection=[a, b, c], file_type=parquet, predicate=DynamicFilter [ b@1 IS NULL OR b@1 > bd ], sort_order_for_reorder=[b@1 DESC], reverse_row_groups=true, dynamic_rg_pruning=eligible, pruning_predicate=b_null_count@0 > 0 OR b_null_count@0 != row_count@2 AND b_max@1 > bd, required_guarantees=[], metrics=[output_rows=4, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=4, pushdown_rows_pruned=0, predicate_cache_inner_records=4, predicate_cache_records=4, scan_efficiency_ratio=21.62% (222/1.03 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_single_col.parquet]]}, projection=[a, b, c], file_type=parquet, predicate=DynamicFilter [ b@1 IS NULL OR b@1 > bd ], sort_order_for_reorder=[b@1 DESC], reverse_row_groups=true, dynamic_rg_pruning=eligible, pruning_predicate=b_null_count@0 > 0 OR b_null_count@0 != row_count@2 AND b_max@1 > bd, required_guarantees=[], metrics=[output_rows=4, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=4, pushdown_rows_pruned=0, predicate_cache_inner_records=4, predicate_cache_records=4, scan_efficiency_ratio=21.62% (222/1.03 K)] statement ok reset datafusion.explain.analyze_categories; @@ -319,7 +319,7 @@ EXPLAIN ANALYZE SELECT * FROM topk_multi_col ORDER BY b ASC NULLS LAST, a DESC L ---- Plan with Metrics 01)SortExec: TopK(fetch=2), expr=[b@1 ASC NULLS LAST, a@0 DESC], preserve_partitioning=[false], filter=[b@1 < bb OR b@1 = bb AND (a@0 IS NULL OR a@0 > ac)], metrics=[output_rows=2, output_batches=1, row_replacements=2] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_multi_col.parquet]]}, projection=[a, b, c], file_type=parquet, predicate=DynamicFilter [ b@1 < bb OR b@1 = bb AND (a@0 IS NULL OR a@0 > ac) ], sort_order_for_reorder=[b@1 ASC NULLS LAST, a@0 DESC], dynamic_rg_pruning=eligible, pruning_predicate=b_null_count@1 != row_count@2 AND b_min@0 < bb OR b_null_count@1 != row_count@2 AND b_min@0 <= bb AND bb <= b_max@3 AND (a_null_count@4 > 0 OR a_null_count@4 != row_count@2 AND a_max@5 > ac), required_guarantees=[], metrics=[output_rows=4, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=4, pushdown_rows_pruned=0, predicate_cache_inner_records=8, predicate_cache_records=8, scan_efficiency_ratio=21.62% (222/1.03 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_multi_col.parquet]]}, projection=[a, b, c], file_type=parquet, predicate=DynamicFilter [ b@1 < bb OR b@1 = bb AND (a@0 IS NULL OR a@0 > ac) ], sort_order_for_reorder=[b@1 ASC NULLS LAST, a@0 DESC], dynamic_rg_pruning=eligible, pruning_predicate=b_null_count@1 != row_count@2 AND b_min@0 < bb OR b_null_count@1 != row_count@2 AND b_min@0 <= bb AND bb <= b_max@3 AND (a_null_count@4 > 0 OR a_null_count@4 != row_count@2 AND a_max@5 > ac), required_guarantees=[], metrics=[output_rows=4, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=4, pushdown_rows_pruned=0, predicate_cache_inner_records=8, predicate_cache_records=8, scan_efficiency_ratio=21.62% (222/1.03 K)] statement ok reset datafusion.explain.analyze_categories; @@ -388,8 +388,8 @@ FROM join_probe p INNER JOIN join_build AS build ---- Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(a@0, a@0), (b@1, b@1)], projection=[a@3, b@4, c@2, e@5], metrics=[output_rows=2, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=1, input_rows=2, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/join_build.parquet]]}, projection=[a, b, c], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=19.58% (196/1.00 K)] -03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/join_probe.parquet]]}, projection=[a, b, e], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= ba AND b@1 <= bb AND struct(a@0, b@1) IN (SET) ([{c0:aa,c1:ba}, {c0:ab,c1:bb}]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= ba AND b_null_count@5 != row_count@2 AND b_min@6 <= bb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=22.05% (228/1.03 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/join_build.parquet]]}, projection=[a, b, c], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=19.58% (196/1.00 K)] +03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/join_probe.parquet]]}, projection=[a, b, e], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= ba AND b@1 <= bb AND struct(a@0, b@1) IN (SET) ([{c0:aa,c1:ba}, {c0:ab,c1:bb}]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= ba AND b_null_count@5 != row_count@2 AND b_min@6 <= bb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=22.05% (228/1.03 K)] statement ok reset datafusion.explain.analyze_categories; @@ -474,9 +474,9 @@ INNER JOIN nested_t3 ON nested_t2.c = nested_t3.d; Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(c@3, d@0)], metrics=[output_rows=2, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=1, input_rows=2, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] 02)--HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(a@0, b@0)], metrics=[output_rows=2, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=1, input_rows=2, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nested_t1.parquet]]}, projection=[a, x], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=17.37% (132/760)] -04)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nested_t2.parquet]]}, projection=[b, c, y], file_type=parquet, predicate=DynamicFilter [ b@0 >= aa AND b@0 <= ab AND b@0 IN (SET) ([aa, ab]) ], dynamic_rg_pruning=eligible, pruning_predicate=b_null_count@1 != row_count@2 AND b_max@0 >= aa AND b_null_count@1 != row_count@2 AND b_min@3 <= ab AND (b_null_count@1 != row_count@2 AND b_min@3 <= aa AND aa <= b_max@0 OR b_null_count@1 != row_count@2 AND b_min@3 <= ab AND ab <= b_max@0), required_guarantees=[b in (aa, ab)], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=5 total → 5 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=3, predicate_cache_inner_records=5, predicate_cache_records=2, scan_efficiency_ratio=22.46% (234/1.04 K)] -05)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nested_t3.parquet]]}, projection=[d, z], file_type=parquet, predicate=DynamicFilter [ d@0 >= ca AND d@0 <= cb AND hash_lookup ], dynamic_rg_pruning=eligible, pruning_predicate=d_null_count@1 != row_count@2 AND d_max@0 >= ca AND d_null_count@1 != row_count@2 AND d_min@3 <= cb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=8 total → 8 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=6, predicate_cache_inner_records=8, predicate_cache_records=2, scan_efficiency_ratio=21.45% (172/802)] +03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nested_t1.parquet]]}, projection=[a, x], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=17.37% (132/760)] +04)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nested_t2.parquet]]}, projection=[b, c, y], file_type=parquet, predicate=DynamicFilter [ b@0 >= aa AND b@0 <= ab AND b@0 IN (SET) ([aa, ab]) ], dynamic_rg_pruning=eligible, pruning_predicate=b_null_count@1 != row_count@2 AND b_max@0 >= aa AND b_null_count@1 != row_count@2 AND b_min@3 <= ab AND (b_null_count@1 != row_count@2 AND b_min@3 <= aa AND aa <= b_max@0 OR b_null_count@1 != row_count@2 AND b_min@3 <= ab AND ab <= b_max@0), required_guarantees=[b in (aa, ab)], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=5 total → 5 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=3, predicate_cache_inner_records=5, predicate_cache_records=2, scan_efficiency_ratio=22.46% (234/1.04 K)] +05)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nested_t3.parquet]]}, projection=[d, z], file_type=parquet, predicate=DynamicFilter [ d@0 >= ca AND d@0 <= cb AND hash_lookup ], dynamic_rg_pruning=eligible, pruning_predicate=d_null_count@1 != row_count@2 AND d_max@0 >= ca AND d_null_count@1 != row_count@2 AND d_min@3 <= cb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=8 total → 8 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=6, predicate_cache_inner_records=8, predicate_cache_records=2, scan_efficiency_ratio=21.45% (172/802)] statement ok reset datafusion.explain.analyze_categories; @@ -605,8 +605,8 @@ LIMIT 2; Plan with Metrics 01)SortExec: TopK(fetch=2), expr=[e@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[e@0 < bb], metrics=[output_rows=2, output_batches=1, row_replacements=2] 02)--HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(a@0, d@0)], projection=[e@2], metrics=[output_rows=2, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=1, input_rows=2, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_join_build.parquet]]}, projection=[a], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=6.39% (64/1.00 K)] -04)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_join_probe.parquet]]}, projection=[d, e], file_type=parquet, predicate=DynamicFilter [ d@0 >= aa AND d@0 <= ab AND d@0 IN (SET) ([aa, ab]) ] AND DynamicFilter [ e@1 < bb ], dynamic_rg_pruning=eligible, pruning_predicate=d_null_count@1 != row_count@2 AND d_max@0 >= aa AND d_null_count@1 != row_count@2 AND d_min@3 <= ab AND (d_null_count@1 != row_count@2 AND d_min@3 <= aa AND aa <= d_max@0 OR d_null_count@1 != row_count@2 AND d_min@3 <= ab AND ab <= d_max@0) AND e_null_count@5 != row_count@2 AND e_min@4 < bb, required_guarantees=[d in (aa, ab)], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=4 total → 4 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=14.89% (154/1.03 K)] +03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_join_build.parquet]]}, projection=[a], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=6.39% (64/1.00 K)] +04)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_join_probe.parquet]]}, projection=[d, e], file_type=parquet, predicate=DynamicFilter [ d@0 >= aa AND d@0 <= ab AND d@0 IN (SET) ([aa, ab]) ] AND DynamicFilter [ e@1 < bb ], dynamic_rg_pruning=eligible, pruning_predicate=d_null_count@1 != row_count@2 AND d_max@0 >= aa AND d_null_count@1 != row_count@2 AND d_min@3 <= ab AND (d_null_count@1 != row_count@2 AND d_min@3 <= aa AND aa <= d_max@0 OR d_null_count@1 != row_count@2 AND d_min@3 <= ab AND ab <= d_max@0) AND e_null_count@5 != row_count@2 AND e_min@4 < bb, required_guarantees=[d in (aa, ab)], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=4 total → 4 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=14.89% (154/1.03 K)] statement ok reset datafusion.explain.analyze_categories; @@ -656,7 +656,7 @@ EXPLAIN ANALYZE SELECT b, a FROM topk_proj ORDER BY a LIMIT 2; Plan with Metrics 01)ProjectionExec: expr=[b@1 as b, a@0 as a], metrics=[output_rows=2, output_batches=1] 02)--SortExec: TopK(fetch=2), expr=[a@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[a@0 < 2], metrics=[output_rows=2, output_batches=1, row_replacements=2] -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_proj.parquet]]}, projection=[a, b], file_type=parquet, predicate=DynamicFilter [ a@0 < 2 ], sort_order_for_reorder=[a@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_min@0 < 2, required_guarantees=[], metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=3, pushdown_rows_pruned=0, predicate_cache_inner_records=3, predicate_cache_records=3, scan_efficiency_ratio=13.21% (141/1.07 K)] +03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_proj.parquet]]}, projection=[a, b], file_type=parquet, predicate=DynamicFilter [ a@0 < 2 ], sort_order_for_reorder=[a@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_min@0 < 2, required_guarantees=[], metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=3, pushdown_rows_pruned=0, predicate_cache_inner_records=3, predicate_cache_records=3, scan_efficiency_ratio=13.21% (141/1.07 K)] # Case 2: prune — `SELECT a` — filter stays as `a < 2` on the scan. query TT @@ -664,7 +664,7 @@ EXPLAIN ANALYZE SELECT a FROM topk_proj ORDER BY a LIMIT 2; ---- Plan with Metrics 01)SortExec: TopK(fetch=2), expr=[a@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[a@0 < 2], metrics=[output_rows=2, output_batches=1, row_replacements=2] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_proj.parquet]]}, projection=[a], file_type=parquet, predicate=DynamicFilter [ a@0 < 2 ], sort_order_for_reorder=[a@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_min@0 < 2, required_guarantees=[], metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=3, pushdown_rows_pruned=0, predicate_cache_inner_records=3, predicate_cache_records=3, scan_efficiency_ratio=6.84% (73/1.07 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_proj.parquet]]}, projection=[a], file_type=parquet, predicate=DynamicFilter [ a@0 < 2 ], sort_order_for_reorder=[a@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_min@0 < 2, required_guarantees=[], metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=3, pushdown_rows_pruned=0, predicate_cache_inner_records=3, predicate_cache_records=3, scan_efficiency_ratio=6.84% (73/1.07 K)] # Case 3: expression — `SELECT a+1 AS a_plus_1` — the TopK filter is on # `a_plus_1`, the scan predicate must read `a@0 + 1`. @@ -673,7 +673,7 @@ EXPLAIN ANALYZE SELECT a + 1 AS a_plus_1, b FROM topk_proj ORDER BY a_plus_1 LIM ---- Plan with Metrics 01)SortExec: TopK(fetch=2), expr=[a_plus_1@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[a_plus_1@0 < 3], metrics=[output_rows=2, output_batches=1, row_replacements=2] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_proj.parquet]]}, projection=[CAST(a@0 AS Int64) + 1 as a_plus_1, b], file_type=parquet, predicate=DynamicFilter [ CAST(a@0 AS Int64) + 1 < 3 ], dynamic_rg_pruning=eligible, metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=3, pushdown_rows_pruned=0, predicate_cache_inner_records=3, predicate_cache_records=3, scan_efficiency_ratio=13.21% (141/1.07 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_proj.parquet]]}, projection=[CAST(a@0 AS Int64) + 1 as a_plus_1, b], file_type=parquet, predicate=DynamicFilter [ CAST(a@0 AS Int64) + 1 < 3 ], dynamic_rg_pruning=eligible, metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=3, pushdown_rows_pruned=0, predicate_cache_inner_records=3, predicate_cache_records=3, scan_efficiency_ratio=13.21% (141/1.07 K)] # Case 4: alias shadowing — `SELECT a+1 AS a` — the projection renames # `a+1` to `a`, so the TopK's `a < 3` must still be rewritten to @@ -683,7 +683,7 @@ EXPLAIN ANALYZE SELECT a + 1 AS a, b FROM topk_proj ORDER BY a LIMIT 2; ---- Plan with Metrics 01)SortExec: TopK(fetch=2), expr=[a@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[a@0 < 3], metrics=[output_rows=2, output_batches=1, row_replacements=2] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_proj.parquet]]}, projection=[CAST(a@0 AS Int64) + 1 as a, b], file_type=parquet, predicate=DynamicFilter [ CAST(a@0 AS Int64) + 1 < 3 ], sort_order_for_reorder=[a@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=3, pushdown_rows_pruned=0, predicate_cache_inner_records=3, predicate_cache_records=3, scan_efficiency_ratio=13.21% (141/1.07 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_proj.parquet]]}, projection=[CAST(a@0 AS Int64) + 1 as a, b], file_type=parquet, predicate=DynamicFilter [ CAST(a@0 AS Int64) + 1 < 3 ], sort_order_for_reorder=[a@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=3, pushdown_rows_pruned=0, predicate_cache_inner_records=3, predicate_cache_records=3, scan_efficiency_ratio=13.21% (141/1.07 K)] statement ok reset datafusion.explain.analyze_categories; @@ -740,12 +740,12 @@ INNER JOIN ( ---- Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(a@0, a@0)], projection=[a@0, min_value@2], metrics=[output_rows=2, output_batches=2, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=2, input_rows=2, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/join_agg_build.parquet]]}, projection=[a], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=14.45% (64/443)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/join_agg_build.parquet]]}, projection=[a], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=14.45% (64/443)] 03)--ProjectionExec: expr=[a@0 as a, min(join_agg_probe.value)@1 as min_value], metrics=[output_rows=2, output_batches=2] 04)----AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[min(join_agg_probe.value)], metrics=[output_rows=2, output_batches=2, spill_count=0, spilled_rows=0] 05)------RepartitionExec: partitioning=Hash([a@0], 4), input_partitions=1, metrics=[output_rows=2, output_batches=2, spill_count=0, spilled_rows=0] 06)--------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[min(join_agg_probe.value)], metrics=[output_rows=2, output_batches=1, spill_count=0, spilled_rows=0, skipped_aggregation_rows=0, reduction_factor=100% (2/2)] -07)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/join_agg_probe.parquet]]}, projection=[a, value], file_type=parquet, predicate=DynamicFilter [ a@0 >= h1 AND a@0 <= h2 AND a@0 IN (SET) ([h1, h2]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= h1 AND a_null_count@1 != row_count@2 AND a_min@3 <= h2 AND (a_null_count@1 != row_count@2 AND a_min@3 <= h1 AND h1 <= a_max@0 OR a_null_count@1 != row_count@2 AND a_min@3 <= h2 AND h2 <= a_max@0), required_guarantees=[a in (h1, h2)], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=4 total → 4 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=4, predicate_cache_records=2, scan_efficiency_ratio=19.07% (151/792)] +07)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/join_agg_probe.parquet]]}, projection=[a, value], file_type=parquet, predicate=DynamicFilter [ a@0 >= h1 AND a@0 <= h2 AND a@0 IN (SET) ([h1, h2]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= h1 AND a_null_count@1 != row_count@2 AND a_min@3 <= h2 AND (a_null_count@1 != row_count@2 AND a_min@3 <= h1 AND h1 <= a_max@0 OR a_null_count@1 != row_count@2 AND a_min@3 <= h2 AND h2 <= a_max@0), required_guarantees=[a in (h1, h2)], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=4 total → 4 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=4, predicate_cache_records=2, scan_efficiency_ratio=19.07% (151/792)] statement ok reset datafusion.explain.analyze_categories; @@ -807,8 +807,8 @@ ON nulls_build.a = nulls_probe.a AND nulls_build.b = nulls_probe.b; ---- Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(a@0, a@0), (b@1, b@1)], metrics=[output_rows=1, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=3, input_batches=1, input_rows=1, avg_fanout=100% (1/1), probe_hit_rate=100% (1/1)] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nulls_build.parquet]]}, projection=[a, b], file_type=parquet, metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=18.6% (144/774)] -03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nulls_probe.parquet]]}, projection=[a, b, c], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= 1 AND b@1 <= 2 AND struct(a@0, b@1) IN (SET) ([{c0:aa,c1:1}, {c0:,c1:2}, {c0:ab,c1:}]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= 1 AND b_null_count@5 != row_count@2 AND b_min@6 <= 2, required_guarantees=[], metrics=[output_rows=1, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=1, pushdown_rows_pruned=3, predicate_cache_inner_records=8, predicate_cache_records=2, scan_efficiency_ratio=20.18% (225/1.11 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nulls_build.parquet]]}, projection=[a, b], file_type=parquet, metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=18.6% (144/774)] +03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nulls_probe.parquet]]}, projection=[a, b, c], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= 1 AND b@1 <= 2 AND struct(a@0, b@1) IN (SET) ([{c0:aa,c1:1}, {c0:,c1:2}, {c0:ab,c1:}]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= 1 AND b_null_count@5 != row_count@2 AND b_min@6 <= 2, required_guarantees=[], metrics=[output_rows=1, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=1, pushdown_rows_pruned=3, predicate_cache_inner_records=8, predicate_cache_records=2, scan_efficiency_ratio=20.18% (225/1.11 K)] statement ok reset datafusion.explain.analyze_categories; @@ -873,8 +873,8 @@ ON lj_build.a = lj_probe.a AND lj_build.b = lj_probe.b; ---- Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=Left, on=[(a@0, a@0), (b@1, b@1)], metrics=[output_rows=2, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=2, input_rows=2, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/lj_build.parquet]]}, projection=[a, b, c], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=19.58% (196/1.00 K)] -03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/lj_probe.parquet]]}, projection=[a, b, e], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= ba AND b@1 <= bb AND struct(a@0, b@1) IN (SET) ([{c0:aa,c1:ba}, {c0:ab,c1:bb}]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= ba AND b_null_count@5 != row_count@2 AND b_min@6 <= bb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=22.05% (228/1.03 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/lj_build.parquet]]}, projection=[a, b, c], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=19.58% (196/1.00 K)] +03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/lj_probe.parquet]]}, projection=[a, b, e], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= ba AND b@1 <= bb AND struct(a@0, b@1) IN (SET) ([{c0:aa,c1:ba}, {c0:ab,c1:bb}]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= ba AND b_null_count@5 != row_count@2 AND b_min@6 <= bb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=22.05% (228/1.03 K)] # LEFT SEMI JOIN: only matching build rows are returned; probe scan still # receives the dynamic filter. @@ -889,8 +889,8 @@ WHERE EXISTS ( ---- Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=LeftSemi, on=[(a@0, a@0), (b@1, b@1)], metrics=[output_rows=2, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=2, input_rows=4, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/lj_build.parquet]]}, projection=[a, b, c], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=19.58% (196/1.00 K)] -03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/lj_probe.parquet]]}, projection=[a, b], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= ba AND b@1 <= bb AND struct(a@0, b@1) IN (SET) ([{c0:aa,c1:ba}, {c0:ab,c1:bb}]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= ba AND b_null_count@5 != row_count@2 AND b_min@6 <= bb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=14.89% (154/1.03 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/lj_build.parquet]]}, projection=[a, b, c], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=19.58% (196/1.00 K)] +03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/lj_probe.parquet]]}, projection=[a, b], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= ba AND b@1 <= bb AND struct(a@0, b@1) IN (SET) ([{c0:aa,c1:ba}, {c0:ab,c1:bb}]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= ba AND b_null_count@5 != row_count@2 AND b_min@6 <= bb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=14.89% (154/1.03 K)] statement ok reset datafusion.explain.analyze_categories; @@ -959,8 +959,8 @@ FROM hl_probe p INNER JOIN hl_build AS build ---- Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(a@0, a@0), (b@1, b@1)], projection=[a@3, b@4, c@2, e@5], metrics=[output_rows=2, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=1, input_rows=2, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/hl_build.parquet]]}, projection=[a, b, c], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=19.58% (196/1.00 K)] -03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/hl_probe.parquet]]}, projection=[a, b, e], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= ba AND b@1 <= bb AND hash_lookup ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= ba AND b_null_count@5 != row_count@2 AND b_min@6 <= bb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=22.05% (228/1.03 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/hl_build.parquet]]}, projection=[a, b, c], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=19.58% (196/1.00 K)] +03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/hl_probe.parquet]]}, projection=[a, b, e], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= ba AND b@1 <= bb AND hash_lookup ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= ba AND b_null_count@5 != row_count@2 AND b_min@6 <= bb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=22.05% (228/1.03 K)] statement ok drop table hl_build; @@ -1008,8 +1008,8 @@ FROM int_build b INNER JOIN int_probe p ---- Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(id1@0, id1@0), (id2@1, id2@1)], projection=[id1@0, id2@1, value@2, data@5], metrics=[output_rows=2, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=1, input_rows=2, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/int_build.parquet]]}, projection=[id1, id2, value], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=18.23% (204/1.12 K)] -03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/int_probe.parquet]]}, projection=[id1, id2, data], file_type=parquet, predicate=DynamicFilter [ id1@0 >= 1 AND id1@0 <= 2 AND id2@1 >= 10 AND id2@1 <= 20 AND hash_lookup ], dynamic_rg_pruning=eligible, pruning_predicate=id1_null_count@1 != row_count@2 AND id1_max@0 >= 1 AND id1_null_count@1 != row_count@2 AND id1_min@3 <= 2 AND id2_null_count@5 != row_count@2 AND id2_max@4 >= 10 AND id2_null_count@5 != row_count@2 AND id2_min@6 <= 20, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=20.67% (221/1.07 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/int_build.parquet]]}, projection=[id1, id2, value], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=18.23% (204/1.12 K)] +03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/int_probe.parquet]]}, projection=[id1, id2, data], file_type=parquet, predicate=DynamicFilter [ id1@0 >= 1 AND id1@0 <= 2 AND id2@1 >= 10 AND id2@1 <= 20 AND hash_lookup ], dynamic_rg_pruning=eligible, pruning_predicate=id1_null_count@1 != row_count@2 AND id1_max@0 >= 1 AND id1_null_count@1 != row_count@2 AND id1_min@3 <= 2 AND id2_null_count@5 != row_count@2 AND id2_max@4 >= 10 AND id2_null_count@5 != row_count@2 AND id2_min@6 <= 20, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=20.67% (221/1.07 K)] statement ok reset datafusion.explain.analyze_categories; diff --git a/docs/source/user-guide/configs.md b/docs/source/user-guide/configs.md index e01af3476b94c..ad97f6e84f669 100644 --- a/docs/source/user-guide/configs.md +++ b/docs/source/user-guide/configs.md @@ -92,6 +92,7 @@ The following configuration settings are available: | datafusion.execution.parquet.coerce_int96 | NULL | (reading) If true, parquet reader will read columns of physical type int96 as originating from a different resolution than nanosecond. This is useful for reading data from systems like Spark which stores microsecond resolution timestamps in an int96 allowing it to write values with a larger date range than 64-bit timestamps with nanosecond resolution. | | datafusion.execution.parquet.coerce_int96_tz | NULL | (reading) Optional timezone applied to INT96 columns when `coerce_int96` is set. When `Some`, INT96 columns coerce to `Timestamp(, Some())` instead of the default `Timestamp(, None)`. Spark and other systems write INT96 values as UTC-adjusted instants, so callers that need the resulting Arrow type to be timezone-aware (e.g. for Spark `TimestampType` semantics) should set this to `"UTC"`. No effect when `coerce_int96` is `None`. | | datafusion.execution.parquet.bloom_filter_on_read | true | (reading) Use any available bloom filters when reading parquet files | +| datafusion.execution.parquet.dictionary_filter_on_read | false | (reading) Use fully dictionary-encoded `BYTE_ARRAY` (Utf8/Binary) column chunks as an exact row-group membership index when reading parquet files. Unlike bloom filters, a fully dictionary-encoded column chunk's dictionary is the exact, complete set of the row group's distinct values, so this can prune both `IN`/`=` and `NOT IN`/`!=` predicates. Only column chunks whose page encoding statistics prove every data page came from the dictionary are used; chunks that fell back to `PLAIN` encoding are ignored. | | datafusion.execution.parquet.max_predicate_cache_size | NULL | (reading) The maximum predicate cache size, in bytes. When `pushdown_filters` is enabled, sets the maximum memory used to cache the results of predicate evaluation between filter evaluation and output generation. Decreasing this value will reduce memory usage, but may increase IO and CPU usage. None means use the default parquet reader setting. 0 means no caching. | | datafusion.execution.parquet.data_pagesize_limit | 1048576 | (writing) Sets best effort maximum size of data page in bytes | | datafusion.execution.parquet.write_batch_size | 1024 | (writing) Sets write_batch_size in rows |