diff --git a/Cargo.lock b/Cargo.lock index 566cc1166813c..14c91aa6ab198 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -165,8 +165,7 @@ checksum = "7c02d123df017efcdfbd739ef81735b36c5ba83ec3c59c80a9d7ecc718f92e50" [[package]] name = "arrow" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b952ca5a8046ad741b60f142d6eca4aeebcad615694202bc64c5341f23e32c5b" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "arrow-arith", "arrow-array", @@ -188,8 +187,7 @@ dependencies = [ [[package]] name = "arrow-arith" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "64a13b8d3008c4e9063c597a08f46446fe3fd5789277127672d6c0bdbb43b1ff" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "arrow-array", "arrow-buffer", @@ -202,8 +200,7 @@ dependencies = [ [[package]] name = "arrow-array" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9486151b2f0785bafc6fa04fc5c99fcb4495455662e58787ea32eaaed33c4192" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "ahash", "arrow-buffer", @@ -221,8 +218,7 @@ dependencies = [ [[package]] name = "arrow-avro" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2e4f9b23a0d7b613acb59fa20bdbe0f80ffdae6411498378340b3915e45f5b84" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "arrow-array", "arrow-buffer", @@ -245,8 +241,7 @@ dependencies = [ [[package]] name = "arrow-buffer" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c4776577a87794bfdf0b4e90e2ea12454fa7738ea2823c4be5b9d1851da7b434" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "bytes", "half", @@ -257,8 +252,7 @@ dependencies = [ [[package]] name = "arrow-cast" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a9ad451ce4f98710828a455b96991b8f031deb2e67f5fcad6773f017e4a69c3a" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "arrow-array", "arrow-buffer", @@ -279,8 +273,7 @@ dependencies = [ [[package]] name = "arrow-csv" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8aa7bf96d6141a7bcca2eed57c7c9767d2a2175281857b8a7b68308992864784" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "arrow-array", "arrow-cast", @@ -294,8 +287,7 @@ dependencies = [ [[package]] name = "arrow-data" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b38fe43e2e8704360f1464e6e8cc4fc381ef02cc4fb0192afa8df1aaa0115c66" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "arrow-buffer", "arrow-schema", @@ -307,8 +299,7 @@ dependencies = [ [[package]] name = "arrow-flight" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "42115e09dbb694b5955da998912121451c6910b338228cb80a5701370dba43ff" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "arrow-arith", "arrow-array", @@ -335,8 +326,7 @@ dependencies = [ [[package]] name = "arrow-ipc" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "29dac499fcbc6ba74ee0324057821d381929a48526a3966bd9dffb44aa06d98c" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "arrow-array", "arrow-buffer", @@ -351,8 +341,7 @@ dependencies = [ [[package]] name = "arrow-json" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0fe05e916ddc50f4c7a363cd69c0ef5894fcee063517e9a0b8582f0c56746af6" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "arrow-array", "arrow-buffer", @@ -376,8 +365,7 @@ dependencies = [ [[package]] name = "arrow-ord" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0e13dbdc2a9c053c10c7baa6e30faee04a180aa7ce88e471835850ce37abd20b" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "arrow-array", "arrow-buffer", @@ -389,8 +377,7 @@ dependencies = [ [[package]] name = "arrow-row" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4d5a1f8c733d15260b305683472ee8ad89c62cbd706703ca873b90d051b41592" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "arrow-array", "arrow-buffer", @@ -402,8 +389,7 @@ dependencies = [ [[package]] name = "arrow-schema" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d9e4969dc350d571766247143ab36a5187d095d3d3690970408bc630d47c69e5" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "bitflags", "serde", @@ -414,8 +400,7 @@ dependencies = [ [[package]] name = "arrow-select" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "402770dba90865359d98d1ef92ef16e23d75c0cca9c2c880c8a05468b7743bf9" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "ahash", "arrow-array", @@ -428,8 +413,7 @@ dependencies = [ [[package]] name = "arrow-string" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a2b0afbb8b9016700938291123df30838b89decc3213dba00852021988b170d3" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "arrow-array", "arrow-buffer", @@ -2067,6 +2051,7 @@ dependencies = [ "object_store", "parking_lot", "parquet", + "rand 0.9.4", "tempfile", "tokio", ] @@ -4053,9 +4038,9 @@ checksum = "112b39cec0b298b6c1999fee3e31427f74f676e4cb9879ed1a121b43661a4154" [[package]] name = "lz4_flex" -version = "0.13.0" +version = "0.13.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "db9a0d582c2874f68138a16ce1867e0ffde6c0bb0a0df85e1f36d04146db488a" +checksum = "7ef0d4ed8669f8f8826eb00dc878084aa8f253506c4fd5e8f58f5bce72ddb97e" dependencies = [ "twox-hash", ] @@ -4464,8 +4449,7 @@ dependencies = [ [[package]] name = "parquet" version = "59.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5302d4da74d6596a1f11f9928767995b53bca657cbeea1e4e8c5074f8a1157dd" +source = "git+https://github.com/DarkWanderer/arrow-rs?branch=get-dictionary#8145bb8ef9dcdddcbe22f90285c9218a5bec9b25" dependencies = [ "ahash", "arrow-array", @@ -4851,7 +4835,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "03da047801ff44bb6a4d407d4860c05fd70bb81714e6b2f3812603d5b145b042" dependencies = [ "heck", - "itertools 0.14.0", + "itertools 0.13.0", "log", "multimap", "petgraph", @@ -4870,7 +4854,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b570b25f7617e43d59005d0990ccb79e950a423952cea19671b7a876da390adf" dependencies = [ "anyhow", - "itertools 0.14.0", + "itertools 0.13.0", "proc-macro2", "quote", "syn 2.0.119", diff --git a/Cargo.toml b/Cargo.toml index 6f4c10f8e7552..8bc1fcfdc899f 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -209,6 +209,31 @@ url = "2.5.7" uuid = "1.23" zstd = { version = "0.13", default-features = false } +# TEMPORARY patch: points at an unreleased arrow-rs branch that adds the +# dictionary-page decode API this branch's Parquet dictionary row-group +# pruning depends on (arrow-rs issue #9010 / PR #10420, which supersedes +# the earlier, now-closed PR #9011). Remove once arrow-rs releases a +# version with this API and the dependency above is bumped to it -- do +# not merge this patch. +[patch.crates-io] +arrow = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-arith = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-array = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-avro = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-buffer = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-cast = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-csv = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-data = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-flight = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-ipc = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-json = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-ord = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-row = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-schema = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-select = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +arrow-string = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } +parquet = { git = "https://github.com/DarkWanderer/arrow-rs", branch = "get-dictionary" } + [workspace.lints.clippy] # Detects large stack-allocated futures that may cause stack overflow crashes (see threshold in clippy.toml) large_futures = "warn" diff --git a/datafusion/common/src/config.rs b/datafusion/common/src/config.rs index 81f573fc2a23e..35898ee8ab087 100644 --- a/datafusion/common/src/config.rs +++ b/datafusion/common/src/config.rs @@ -1181,6 +1181,16 @@ config_namespace! { /// (reading) Use any available bloom filters when reading parquet files pub bloom_filter_on_read: bool, default = true + /// (reading) Use fully dictionary-encoded `BYTE_ARRAY` (Utf8/Binary) + /// column chunks as an exact row-group membership index when reading + /// parquet files. Unlike bloom filters, a fully dictionary-encoded + /// column chunk's dictionary is the exact, complete set of the row + /// group's distinct values, so this can prune both `IN`/`=` and + /// `NOT IN`/`!=` predicates. Only column chunks whose page encoding + /// statistics prove every data page came from the dictionary are + /// used; chunks that fell back to `PLAIN` encoding are ignored. + pub dictionary_filter_on_read: bool, default = false + /// (reading) The maximum predicate cache size, in bytes. When /// `pushdown_filters` is enabled, sets the maximum memory used to cache /// the results of predicate evaluation between filter evaluation and diff --git a/datafusion/common/src/file_options/parquet_writer.rs b/datafusion/common/src/file_options/parquet_writer.rs index 320bfcf33e488..f737a78c2efe0 100644 --- a/datafusion/common/src/file_options/parquet_writer.rs +++ b/datafusion/common/src/file_options/parquet_writer.rs @@ -242,6 +242,7 @@ impl ParquetOptions { maximum_parallel_row_group_writers: _, maximum_buffered_record_batches_per_stream: _, bloom_filter_on_read: _, // reads not used for writer props + dictionary_filter_on_read: _, // reads not used for writer props schema_force_view_types: _, binary_as_string: _, // not used for writer props coerce_int96: _, // not used for writer props @@ -500,6 +501,7 @@ mod tests { maximum_buffered_record_batches_per_stream: defaults .maximum_buffered_record_batches_per_stream, bloom_filter_on_read: defaults.bloom_filter_on_read, + dictionary_filter_on_read: defaults.dictionary_filter_on_read, schema_force_view_types: defaults.schema_force_view_types, binary_as_string: defaults.binary_as_string, skip_arrow_metadata: defaults.skip_arrow_metadata, @@ -620,6 +622,8 @@ mod tests { maximum_buffered_record_batches_per_stream: global_options_defaults .maximum_buffered_record_batches_per_stream, bloom_filter_on_read: global_options_defaults.bloom_filter_on_read, + dictionary_filter_on_read: global_options_defaults + .dictionary_filter_on_read, max_predicate_cache_size: global_options_defaults .max_predicate_cache_size, schema_force_view_types: global_options_defaults.schema_force_view_types, diff --git a/datafusion/core/Cargo.toml b/datafusion/core/Cargo.toml index 8679dad9f9a32..2a799f345448c 100644 --- a/datafusion/core/Cargo.toml +++ b/datafusion/core/Cargo.toml @@ -304,6 +304,11 @@ harness = false name = "preserve_file_partitioning" required-features = ["parquet"] +[[bench]] +harness = false +name = "parquet_dictionary_pruning_query" +required-features = ["parquet", "sql"] + [[bench]] harness = false name = "reset_plan_states" diff --git a/datafusion/core/benches/parquet_dictionary_pruning_query.rs b/datafusion/core/benches/parquet_dictionary_pruning_query.rs new file mode 100644 index 0000000000000..5cadbf9a0c5ee --- /dev/null +++ b/datafusion/core/benches/parquet_dictionary_pruning_query.rs @@ -0,0 +1,1098 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! End-to-end wall-clock benchmark for row-group pruning by Parquet +//! dictionaries (`datafusion.execution.parquet.dictionary_filter_on_read`). +//! +//! `datafusion/datasource-parquet/benches/parquet_dictionary_pruning.rs` +//! only measures the *cost* of the pruning decision itself (calling +//! `prune_by_statistics` / `prune_by_bloom_filters` / `prune_by_dictionary` +//! directly, no `SessionContext`, no scan). This benchmark instead runs +//! full SQL queries end to end (`SessionContext::sql` -> `DataFrame::collect`) +//! against real Parquet files, over a dataset shaped like observability +//! span/log data -- the case this optimization is unambiguously the right +//! tool for: +//! +//! - Rows land in arrival (timestamp) order. Spans belonging to one trace +//! are emitted close together, so a given `trace_id` is *spatially +//! constrained* to a single row group, but many traces are interleaved +//! within it (see `generate_row_group`'s shuffle). Each trace has +//! `SPANS_PER_TRACE` spans (default 32, a representative distributed-trace +//! fan-out), which keeps the `trace_id` dictionary a small fraction of the +//! row group -- see `ATTRIBUTE_PAD_LEN` below for why that fraction +//! matters. +//! - `trace_id` is random hex, so every row group's `[min, max]` spans +//! essentially the whole domain -- min/max statistics prune *nothing*, +//! ever, for a `trace_id` predicate. +//! - Each row group's distinct trace set is small and bounded, so the +//! column chunk stays fully dictionary encoded -- its dictionary is the +//! *exact* membership index for that row group. +//! - Each span carries an `attributes` payload padded to `ATTRIBUTE_PAD_LEN` +//! hex characters (default 160, ~200 bytes/row after the surrounding +//! pseudo-JSON), so it dominates file bytes the way real span attribute +//! payloads do. The bigger this payload relative to a dictionary page, +//! the more a row group skipped by dictionary pruning actually saves. +//! +//! Bloom filters can answer the same lookups, but only probabilistically +//! (5% false-positive rate by default), must be enabled at write time (off +//! by default), and cost extra file bytes -- dictionary pages are already +//! there. Three query shapes exercise this: +//! +//! - `trace_lookup`: `trace_id = `, a single-value point lookup. +//! - `trace_lookup_in`: `trace_id IN ()`, default +//! 32 literals all drawn from one row group. A bloom filter's +//! false-positive rate compounds per literal: a non-matching row group +//! survives with probability `1 - (1 - fpp)^N`, so at `N = 32` bloom +//! filters retain most row groups even though none of them contain a +//! needle. Dictionaries stay exact at any `N`. +//! - `tenant_not_in`: `tenant NOT IN (...)`, the one direction bloom filters +//! structurally cannot serve at all -- they can prove "this value might be +//! present" but never "this row group contains nothing but excluded +//! values" (see the two-directional `contained` logic in +//! `datafusion/datasource-parquet/src/dictionary_filter.rs`). 3 of every 4 +//! row groups here are "noisy" (synthetic-monitoring/health-check +//! traffic, drawn only from the excluded tenants), so dictionary pruning +//! removes 75% of the scan; the query also sums `octet_length(attributes)` +//! so the skipped row groups are the ones carrying the dominant payload +//! column, not just a `count(*)` a reader could serve from metadata alone. +//! +//! `print_evidence_table` also reports measured bytes read per (query, +//! variant) pair, from `ParquetFileMetrics::bytes_scanned` -- incremented by +//! every `get_bytes` / `get_byte_ranges` call the async reader makes, so it +//! covers data-page reads, bloom filter reads, and dictionary-page reads, +//! but **not** the footer or page-index metadata (`get_metadata` never +//! touches it). It must not be read as total file I/O. +//! +//! For `stats_only`, `dictionary`, and `bloom_filter` (each reads at most +//! one kind of index), the report further splits `bytes_scanned` into index +//! bytes and data-page bytes. The split is derived from file metadata, not +//! separately measured: this dataset's statistics never prune a row group +//! (asserted below), so `load_bloom_filters` / `load_dictionaries` always +//! visit every row group for the predicate column, and their per-row-group +//! byte cost is exactly `bloom_filter_length()` (bloom) or +//! `data_page_offset() - dictionary_page_offset()` (dictionary) -- so +//! `data_page_bytes = bytes_scanned - index_bytes` follows exactly. +//! `bloom_and_dictionary` has no such closed form (dictionaries are only +//! read for whichever row groups survive the probabilistic bloom filter +//! stage), so its report prints the total with `n/a` for the split. +//! +//! Run with: +//! ```text +//! cargo bench -p datafusion --bench parquet_dictionary_pruning_query +//! ``` +//! +//! Dataset size knobs (env-overridable, see `BenchConfig::from_env`): +//! `ROW_GROUPS` (default 64), `ROWS_PER_ROW_GROUP` (default 65536), +//! `SPANS_PER_TRACE` (default 32), `ATTRIBUTE_PAD_LEN` (default 160), +//! `TRACE_IN_LIST_LEN` (default 32). + +use std::collections::HashMap; +use std::hint::black_box; +use std::path::{Path, PathBuf}; +use std::sync::{Arc, LazyLock}; +use std::time::Duration; + +use arrow::array::{ + ArrayRef, Int64Array, RecordBatch, StringArray, TimestampMillisecondArray, +}; +use arrow::datatypes::{DataType, Field, Schema, SchemaRef, TimeUnit}; +use criterion::{Criterion, criterion_group, criterion_main}; +use datafusion::physical_plan::display::DisplayableExecutionPlan; +use datafusion::physical_plan::metrics::{MetricValue, MetricsSet}; +use datafusion::physical_plan::{ExecutionPlan, collect}; +use datafusion::prelude::{ParquetReadOptions, SessionConfig, SessionContext}; +use datafusion_common::Result; +use datafusion_common::format::MetricType; +use datafusion_datasource_parquet::is_fully_dictionary_encoded; +use parquet::arrow::ArrowWriter; +use parquet::arrow::arrow_reader::ParquetRecordBatchReaderBuilder; +use parquet::basic::Compression; +use parquet::file::properties::WriterProperties; +use rand::rngs::SmallRng; +use rand::seq::SliceRandom; +use rand::{Rng, SeedableRng}; +use tempfile::TempDir; +use tokio::runtime::Runtime; + +/// Fixed seed so the generated dataset (and the needle trace ids computed +/// against it) is identical from run to run. +const SEED: u64 = 0x5EED_7A3C_5DA7_A5E7; +const SPAN_ID_SALT: u64 = 0x5BA1_1D00_5BA1_1D00; + +const NORMAL_TENANT_COUNT: u32 = 20; +const SERVICE_COUNT: u32 = 24; +const NAME_COUNT: u32 = 120; +/// Predicate column names, shared between dataset generation, the query +/// builders, and the byte-accounting report so they can't drift apart. +const TRACE_ID_COLUMN: &str = "trace_id"; +const TENANT_COLUMN: &str = "tenant"; +// These values sort between `tenant-010` and `tenant-011`, keeping them +// inside normal row groups' min/max range so statistics cannot prove either +// that a row group matches or that it does not match `tenant_not_in`. +const TENANT_OPS: &str = "tenant-010-ops"; +const TENANT_SYNTHETIC: &str = "tenant-010-synthetic"; +/// Every `NORMAL_ROW_GROUP_STRIDE`-th row group draws `tenant` from the full +/// normal domain; the rest are "noisy" row groups whose `tenant` column is +/// drawn only from the two excluded tenants above, so their dictionary is a +/// subset of `{tenant-010-ops, tenant-010-synthetic}` while `min != max` +/// (statistics can't prune them, only the dictionary can -- see +/// `tenant_not_in`). Synthetic-monitoring and health-check traffic +/// genuinely dominates span volume and arrives in scheduled bursts, so it +/// clusters into its own row groups -- the shape `NOT IN` pruning exists +/// for: 3 of every 4 row groups here are noisy. +const NORMAL_ROW_GROUP_STRIDE: usize = 4; + +#[derive(Debug, Clone, Copy)] +struct BenchConfig { + row_groups: usize, + rows_per_row_group: usize, + spans_per_trace: usize, + attribute_pad_len: usize, + trace_in_list_len: usize, +} + +impl BenchConfig { + fn from_env() -> Self { + let config = Self { + row_groups: env_usize("ROW_GROUPS", 64), + rows_per_row_group: env_usize("ROWS_PER_ROW_GROUP", 65_536), + spans_per_trace: env_usize("SPANS_PER_TRACE", 32), + attribute_pad_len: env_usize("ATTRIBUTE_PAD_LEN", 160), + trace_in_list_len: env_usize("TRACE_IN_LIST_LEN", 32), + }; + assert_eq!( + config.rows_per_row_group % config.spans_per_trace, + 0, + "ROWS_PER_ROW_GROUP ({}) must be a multiple of SPANS_PER_TRACE ({})", + config.rows_per_row_group, + config.spans_per_trace + ); + assert!( + config.rows_per_row_group >= 2, + "ROWS_PER_ROW_GROUP ({}) must be at least 2", + config.rows_per_row_group + ); + assert!( + config.trace_in_list_len <= config.traces_per_row_group(), + "TRACE_IN_LIST_LEN ({}) must not exceed traces per row group ({}) -- \ + otherwise needle_trace_ids_in would spill into a neighboring row group", + config.trace_in_list_len, + config.traces_per_row_group() + ); + config + } + + fn traces_per_row_group(&self) -> usize { + self.rows_per_row_group / self.spans_per_trace + } + + fn total_rows(&self) -> usize { + self.row_groups * self.rows_per_row_group + } + + fn is_noisy_row_group(&self, rg: usize) -> bool { + !rg.is_multiple_of(NORMAL_ROW_GROUP_STRIDE) + } + + fn noisy_row_group_count(&self) -> usize { + (0..self.row_groups) + .filter(|rg| self.is_noisy_row_group(*rg)) + .count() + } +} + +fn env_usize(name: &str, default: usize) -> usize { + std::env::var(name) + .ok() + .and_then(|v| v.parse().ok()) + .unwrap_or(default) +} + +/// Cheap, deterministic 64-bit mix (SplitMix64). Used to turn a plain +/// integer index into an evenly-distributed, hex-looking id without +/// needing to carry a stateful RNG across row-group boundaries (dataset +/// generation happens one row group at a time). +fn splitmix64(mut x: u64) -> u64 { + x = x.wrapping_add(0x9E37_79B9_7F4A_7C15); + let mut z = x; + z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9); + z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB); + z ^ (z >> 31) +} + +/// A 32-hex-char trace id, globally unique across distinct +/// `global_trace_index` values -- `global_trace_index` is `rg * +/// traces_per_row_group + t`, so a trace never spans row groups by +/// construction. +fn trace_id_for_index(global_trace_index: u64) -> String { + let hi = splitmix64(global_trace_index ^ SEED); + let lo = splitmix64(hi); + format!("{hi:016x}{lo:016x}") +} + +/// A 16-hex-char span id. High cardinality (one per row): the dictionary +/// for this column is expected to overflow into a `PLAIN` fallback -- see +/// the "Out of scope" note in the module doc about a future negative +/// benchmark variant using this column. +fn span_id_for_index(global_row_index: u64) -> String { + let v = splitmix64(global_row_index ^ SPAN_ID_SALT); + format!("{v:016x}") +} + +fn schema() -> SchemaRef { + Arc::new(Schema::new(vec![ + Field::new( + "ts", + DataType::Timestamp(TimeUnit::Millisecond, None), + false, + ), + Field::new("trace_id", DataType::Utf8, false), + Field::new("span_id", DataType::Utf8, false), + Field::new("tenant", DataType::Utf8, false), + Field::new("service", DataType::Utf8, false), + Field::new("name", DataType::Utf8, false), + Field::new("duration_ns", DataType::Int64, false), + Field::new("attributes", DataType::Utf8, false), + ])) +} + +/// ~200-byte pseudo-JSON attributes payload (`ATTRIBUTE_PAD_LEN` +/// hex-digit pad). Dominates file bytes, so row groups skipped by pruning +/// translate into real skipped I/O. +fn pseudo_json_attributes(rng: &mut SmallRng, pad_len: usize) -> String { + let pad: String = (0..pad_len) + .map(|_| { + let nibble = rng.random_range(0..16u32); + char::from_digit(nibble, 16).expect("valid hex digit") + }) + .collect(); + let retry = rng.random_range(0..5); + format!( + r#"{{"http.method":"GET","http.status_code":200,"retry":{retry},"pad":"{pad}"}}"# + ) +} + +/// Builds one row group's data. Trace ids for the row group are generated +/// then shuffled, so spans belonging to one trace are interleaved with +/// spans from other traces -- matching arrival-order data rather than a +/// sorted-by-trace layout. +fn generate_row_group(rg: usize, config: &BenchConfig) -> RecordBatch { + let rows = config.rows_per_row_group; + let traces_per_rg = config.traces_per_row_group(); + let mut rng = SmallRng::seed_from_u64(SEED.wrapping_add(rg as u64)); + + let mut trace_ids: Vec = Vec::with_capacity(rows); + for t in 0..traces_per_rg { + let global_trace_index = (rg * traces_per_rg + t) as u64; + let trace_id = trace_id_for_index(global_trace_index); + for _ in 0..config.spans_per_trace { + trace_ids.push(trace_id.clone()); + } + } + trace_ids.shuffle(&mut rng); + + let noisy = config.is_noisy_row_group(rg); + let base_ts_ms: i64 = 1_700_000_000_000 + (rg * config.rows_per_row_group) as i64; + + let mut ts = Vec::with_capacity(rows); + let mut span_id = Vec::with_capacity(rows); + let mut tenant = Vec::with_capacity(rows); + let mut service = Vec::with_capacity(rows); + let mut name = Vec::with_capacity(rows); + let mut duration_ns = Vec::with_capacity(rows); + let mut attributes = Vec::with_capacity(rows); + + for i in 0..rows { + let global_row = (rg * config.rows_per_row_group + i) as u64; + ts.push(base_ts_ms + i as i64); + span_id.push(span_id_for_index(global_row)); + tenant.push(if noisy { + if i % 2 == 0 { + TENANT_OPS.to_string() + } else { + TENANT_SYNTHETIC.to_string() + } + } else if i == 0 { + // Pin both extrema so every normal row group contains the + // excluded values within, rather than outside, its statistics. + "tenant-001".to_string() + } else if i == 1 { + format!("tenant-{NORMAL_TENANT_COUNT:03}") + } else { + format!("tenant-{:03}", rng.random_range(1..=NORMAL_TENANT_COUNT)) + }); + service.push(format!("service-{:02}", rng.random_range(0..SERVICE_COUNT))); + name.push(format!("op-{:03}", rng.random_range(0..NAME_COUNT))); + duration_ns.push(rng.random_range(1_000i64..50_000_000i64)); + attributes.push(pseudo_json_attributes(&mut rng, config.attribute_pad_len)); + } + + RecordBatch::try_new( + schema(), + vec![ + Arc::new(TimestampMillisecondArray::from(ts)) as ArrayRef, + Arc::new(StringArray::from(trace_ids)), + Arc::new(StringArray::from(span_id)), + Arc::new(StringArray::from(tenant)), + Arc::new(StringArray::from(service)), + Arc::new(StringArray::from(name)), + Arc::new(Int64Array::from(duration_ns)), + Arc::new(StringArray::from(attributes)), + ], + ) + .expect("valid record batch") +} + +/// The row group used to pick needle trace ids for `trace_lookup` / +/// `trace_lookup_in` -- any row group works, this just fixes one. +fn needle_row_group(config: &BenchConfig) -> usize { + config.row_groups / 2 +} + +/// A trace id present in exactly one row group. +fn needle_trace_id(config: &BenchConfig) -> String { + let rg = needle_row_group(config); + let traces_per_rg = config.traces_per_row_group(); + trace_id_for_index((rg * traces_per_rg) as u64) +} + +/// `TRACE_IN_LIST_LEN` trace ids, all drawn from the same single row group +/// as [`needle_trace_id`]. +fn needle_trace_ids_in(config: &BenchConfig) -> Vec { + let rg = needle_row_group(config); + let traces_per_rg = config.traces_per_row_group(); + (0..config.trace_in_list_len) + .map(|t| trace_id_for_index((rg * traces_per_rg + t) as u64)) + .collect() +} + +fn trace_lookup_query(needle: &str) -> String { + format!( + "SELECT ts, service, name, duration_ns, attributes FROM spans WHERE trace_id = '{needle}'" + ) +} + +fn trace_lookup_in_query(ids: &[String]) -> String { + let list = ids + .iter() + .map(|id| format!("'{id}'")) + .collect::>() + .join(", "); + format!( + "SELECT ts, service, name, duration_ns, attributes FROM spans WHERE trace_id IN ({list})" + ) +} + +fn tenant_not_in_query() -> String { + format!( + "SELECT count(*), sum(octet_length(attributes)) FROM spans \ + WHERE tenant NOT IN ('{TENANT_OPS}', '{TENANT_SYNTHETIC}')" + ) +} + +/// Which on-disk index (if any) a [`BenchVariant`] reads for its predicate +/// column, and therefore whether `bytes_scanned` can be split into index +/// bytes vs. data-page bytes. `Mixed` (`bloom_and_dictionary`) keeps that +/// split unrepresentable rather than printing a guessed number: dictionaries +/// are only read for whichever row groups survive the probabilistic bloom +/// filter stage, so which row groups those are isn't recoverable from +/// metadata alone. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +enum IndexKind { + None, + Bloom, + Dictionary, + Mixed, +} + +/// Sums the on-disk footprint of `kind`'s index for `column_idx`, across +/// every row group in `metadata`. Not just statistics survivors: this +/// dataset asserts statistics prune nothing (checked in `build_dataset`), so +/// `load_bloom_filters` / `load_dictionaries` (`opener/mod.rs`) always visit +/// every row group for the predicate column. +/// +/// Returns `None` when the total can't be derived from metadata alone: +/// `None`/`Mixed` kinds have no closed form (see [`IndexKind`]), and `Bloom` +/// returns `None` if any row group is missing `bloom_filter_length()` -- +/// arrow-rs then probes with `SBBF_HEADER_SIZE_ESTIMATE` bytes and the true +/// size isn't knowable without an extra read. +fn index_bytes( + metadata: &parquet::file::metadata::ParquetMetaData, + column_idx: usize, + kind: IndexKind, +) -> Option { + match kind { + IndexKind::None | IndexKind::Mixed => None, + IndexKind::Bloom => (0..metadata.num_row_groups()) + .map(|rg| { + metadata + .row_group(rg) + .column(column_idx) + .bloom_filter_length() + .map(|len| len as u64) + }) + .sum(), + IndexKind::Dictionary => Some( + (0..metadata.num_row_groups()) + .map(|rg| { + let col_meta = metadata.row_group(rg).column(column_idx); + // Chunks that fell back to PLAIN are skipped entirely by + // `load_dictionaries` (opener/mod.rs), so they cost 0 + // dictionary-read bytes -- not an unknown quantity. + if !is_fully_dictionary_encoded(col_meta) { + return 0; + } + let dictionary_page_offset = col_meta + .dictionary_page_offset() + .expect("is_fully_dictionary_encoded implies a dictionary page"); + (col_meta.data_page_offset() - dictionary_page_offset) as u64 + }) + .sum(), + ), + } +} + +#[derive(Clone, Copy)] +enum DatasetFile { + Spans, + SpansBloom, +} + +struct Dataset { + _tempdir: TempDir, + spans_path: PathBuf, + spans_bloom_path: PathBuf, + config: BenchConfig, + /// Precomputed by [`index_bytes`] in `build_dataset`, keyed by predicate + /// column and index kind. Absent entries mean "not derivable" -- see + /// [`IndexKind`] and [`index_bytes`]. + index_bytes: HashMap<(&'static str, IndexKind), u64>, +} + +impl Dataset { + fn path_for(&self, file: DatasetFile) -> &Path { + match file { + DatasetFile::Spans => &self.spans_path, + DatasetFile::SpansBloom => &self.spans_bloom_path, + } + } +} + +static DATASET: LazyLock = + LazyLock::new(|| build_dataset().expect("failed to build benchmark dataset")); + +fn build_dataset() -> Result { + let config = BenchConfig::from_env(); + let tempdir = TempDir::new()?; + let spans_path = tempdir.path().join("spans.parquet"); + let spans_bloom_path = tempdir.path().join("spans_bloom.parquet"); + + let schema = schema(); + let compression = Compression::ZSTD(Default::default()); + let spans_props = WriterProperties::builder() + .set_max_row_group_row_count(Some(config.rows_per_row_group)) + .set_dictionary_enabled(true) + .set_compression(compression) + .build(); + // Bloom filters off by default (matches DataFusion's own default), on + // for `spans_bloom.parquet` so the `bloom_filter` / `bloom_and_dictionary` + // variants have something to read. + let spans_bloom_props = WriterProperties::builder() + .set_max_row_group_row_count(Some(config.rows_per_row_group)) + .set_dictionary_enabled(true) + .set_compression(compression) + .set_bloom_filter_enabled(true) + .build(); + + let mut spans_writer = ArrowWriter::try_new( + std::fs::File::create(&spans_path)?, + Arc::clone(&schema), + Some(spans_props), + )?; + let mut spans_bloom_writer = ArrowWriter::try_new( + std::fs::File::create(&spans_bloom_path)?, + Arc::clone(&schema), + Some(spans_bloom_props), + )?; + + for rg in 0..config.row_groups { + let batch = generate_row_group(rg, &config); + spans_writer.write(&batch)?; + spans_bloom_writer.write(&batch)?; + } + spans_writer.close()?; + spans_bloom_writer.close()?; + + let spans_bytes = std::fs::metadata(&spans_path)?.len(); + let spans_bloom_bytes = std::fs::metadata(&spans_bloom_path)?.len(); + + // Load-bearing property: neither predicate column's dictionary may have + // overflowed into a PLAIN fallback. If one had, `is_fully_dictionary_encoded` + // would (correctly) refuse the chunk and the corresponding `dictionary*` + // variant would silently degrade to "no pruning" -- which would look + // like a regression in the numbers rather than a broken dataset. This is + // also the precondition `index_bytes` relies on below. + let reader = + ParquetRecordBatchReaderBuilder::try_new(std::fs::File::open(&spans_path)?)?; + let metadata = reader.metadata(); + assert_eq!( + metadata.num_row_groups(), + config.row_groups, + "unexpected row group count -- the writer flushed at a different \ + boundary than ROWS_PER_ROW_GROUP" + ); + for column_name in [TRACE_ID_COLUMN, TENANT_COLUMN] { + let column_idx = schema.index_of(column_name).expect("known column"); + for rg in 0..config.row_groups { + let col_meta = metadata.row_group(rg).column(column_idx); + assert!( + is_fully_dictionary_encoded(col_meta), + "{column_name} dictionary overflowed to PLAIN in row group {rg} -- \ + shrink ROWS_PER_ROW_GROUP or SPANS_PER_TRACE" + ); + } + } + + // The `bloom_filter` variant's index bytes come from spans_bloom.parquet + // (the only file bloom filters are written to); the `dictionary` + // variant's come from spans.parquet, read above. Both predicate columns + // are precomputed so the report needs no further metadata I/O. + let spans_bloom_reader = ParquetRecordBatchReaderBuilder::try_new( + std::fs::File::open(&spans_bloom_path)?, + )?; + let spans_bloom_metadata = spans_bloom_reader.metadata(); + let mut index_bytes_by_column = HashMap::new(); + for column_name in [TRACE_ID_COLUMN, TENANT_COLUMN] { + let column_idx = schema.index_of(column_name).expect("known column"); + if let Some(bytes) = index_bytes(metadata, column_idx, IndexKind::Dictionary) { + index_bytes_by_column.insert((column_name, IndexKind::Dictionary), bytes); + } + if let Some(bytes) = + index_bytes(spans_bloom_metadata, column_idx, IndexKind::Bloom) + { + index_bytes_by_column.insert((column_name, IndexKind::Bloom), bytes); + } + } + + println!( + "parquet_dictionary_pruning_query: {} row groups x {} rows = {} total rows, \ + {} traces/row group", + config.row_groups, + config.rows_per_row_group, + config.total_rows(), + config.traces_per_row_group(), + ); + println!( + "parquet_dictionary_pruning_query: spans.parquet = {spans_bytes} bytes, \ + spans_bloom.parquet = {spans_bloom_bytes} bytes (bloom filter overhead = {} bytes)", + spans_bloom_bytes as i64 - spans_bytes as i64, + ); + + Ok(Dataset { + _tempdir: tempdir, + spans_path, + spans_bloom_path, + config, + index_bytes: index_bytes_by_column, + }) +} + +#[derive(Clone, Copy)] +struct BenchVariant { + name: &'static str, + file: DatasetFile, + dictionary_filter_on_read: bool, + bloom_filter_on_read: bool, + /// Which index (if any) this variant reads for the predicate column -- + /// drives the byte report's index/data split. See [`IndexKind`]. + predicate_index_kind: IndexKind, +} + +const BENCH_VARIANTS: [BenchVariant; 4] = [ + BenchVariant { + name: "stats_only", + file: DatasetFile::Spans, + dictionary_filter_on_read: false, + bloom_filter_on_read: false, + predicate_index_kind: IndexKind::None, + }, + BenchVariant { + name: "dictionary", + file: DatasetFile::Spans, + dictionary_filter_on_read: true, + bloom_filter_on_read: false, + predicate_index_kind: IndexKind::Dictionary, + }, + BenchVariant { + name: "bloom_filter", + file: DatasetFile::SpansBloom, + dictionary_filter_on_read: false, + bloom_filter_on_read: true, + predicate_index_kind: IndexKind::Bloom, + }, + BenchVariant { + name: "bloom_and_dictionary", + file: DatasetFile::SpansBloom, + dictionary_filter_on_read: true, + bloom_filter_on_read: true, + predicate_index_kind: IndexKind::Mixed, + }, +]; + +fn session_config(variant: &BenchVariant) -> SessionConfig { + let mut cfg = SessionConfig::new(); + let opts = cfg.options_mut(); + opts.execution.parquet.dictionary_filter_on_read = variant.dictionary_filter_on_read; + opts.execution.parquet.bloom_filter_on_read = variant.bloom_filter_on_read; + cfg +} + +async fn run_query(variant: &BenchVariant, path: &Path, query: &str) -> Vec { + let ctx = SessionContext::new_with_config(session_config(variant)); + ctx.register_parquet( + "spans", + path.to_str().expect("utf8 path"), + ParquetReadOptions::default(), + ) + .await + .expect("register spans table"); + let df = ctx.sql(query).await.expect("plan query"); + df.collect().await.expect("collect query") +} + +struct ScanEvidence { + datasource_line: String, + metrics: MetricsSet, +} + +/// Recursively collects the typed metrics from every node in `plan`. +fn gather_metrics(plan: &Arc, out: &mut MetricsSet) { + if let Some(metrics) = plan.metrics() { + for metric in metrics.iter() { + out.push(Arc::clone(metric)); + } + } + for child in plan.children() { + gather_metrics(child, out); + } +} + +/// Runs `query` and returns both the typed execution metrics used by the +/// assertions and the formatted `DataSourceExec` line used in the report. +async fn scan_evidence(variant: &BenchVariant, path: &Path, query: &str) -> ScanEvidence { + let mut cfg = session_config(variant); + cfg.options_mut().explain.analyze_level = MetricType::Summary; + let ctx = SessionContext::new_with_config(cfg); + ctx.register_parquet( + "spans", + path.to_str().expect("utf8 path"), + ParquetReadOptions::default(), + ) + .await + .expect("register spans table"); + + let dataframe = ctx + .sql(query) + .await + .expect("plan query for pruning evidence"); + let plan = dataframe + .create_physical_plan() + .await + .expect("create physical plan for pruning evidence"); + collect(Arc::clone(&plan), ctx.task_ctx()) + .await + .expect("execute query for pruning evidence"); + + let formatted = DisplayableExecutionPlan::with_metrics(plan.as_ref()) + .set_metric_types(vec![MetricType::Summary]) + .indent(false) + .to_string(); + let datasource_line = formatted + .lines() + .find(|line| line.contains("DataSourceExec")) + .unwrap_or_else(|| { + panic!("no DataSourceExec line in plan for {query:?}:\n{formatted}") + }) + .trim() + .to_string(); + + let mut metrics = MetricsSet::new(); + gather_metrics(&plan, &mut metrics); + ScanEvidence { + datasource_line, + metrics, + } +} + +/// Returns the raw value of `key=...` from a metrics line, up to (but not +/// including) the next top-level comma. +fn metric_value<'a>(line: &'a str, key: &str) -> &'a str { + let marker = format!("{key}="); + let start = line + .find(&marker) + .unwrap_or_else(|| panic!("metric {key} not found in line: {line}")) + + marker.len(); + let rest = &line[start..]; + let end = rest.find(',').unwrap_or(rest.len()); + rest[..end].trim_end_matches(']').trim() +} + +/// Returns exact pruning counts from the typed metrics API. The formatted +/// display rounds large counts (for example, 1,001 becomes `1.00 K`), so it +/// must not be parsed for correctness assertions. +fn pruning_counts(metrics: &MetricsSet, name: &str) -> (usize, usize, usize) { + match metrics.sum_by_name(name) { + Some(MetricValue::PruningMetrics { + pruning_metrics, .. + }) => ( + pruning_metrics.pruned() + pruning_metrics.matched(), + pruning_metrics.matched(), + pruning_metrics.fully_matched(), + ), + Some(other) => panic!("metric {name} is not a pruning metric: {other:?}"), + None => panic!("pruning metric {name} not found"), + } +} + +/// Total bytes read for the scan (`bytes_scanned`, summed across +/// partitions), from the typed metrics API -- see +/// `ParquetFileMetrics::bytes_scanned`. +fn scan_bytes(metrics: &MetricsSet) -> usize { + metrics + .aggregate_by_name() + .sum_by_name("bytes_scanned") + .map(|v| v.as_usize()) + .expect("parquet scan should report a bytes_scanned metric") +} + +/// `(query name, SQL, predicate column)` for each query shape -- the +/// predicate column drives which precomputed [`IndexKind`] bytes the byte +/// report attributes to a given variant. +fn queries(config: &BenchConfig) -> Vec<(&'static str, String, &'static str)> { + vec![ + ( + "trace_lookup", + trace_lookup_query(&needle_trace_id(config)), + TRACE_ID_COLUMN, + ), + ( + "trace_lookup_in", + trace_lookup_in_query(&needle_trace_ids_in(config)), + TRACE_ID_COLUMN, + ), + ("tenant_not_in", tenant_not_in_query(), TENANT_COLUMN), + ] +} + +/// Formats a byte count as `" ( MiB)"` for the evidence report. +fn format_bytes(bytes: u64) -> String { + format!("{bytes} ({:.2} MiB)", bytes as f64 / (1024.0 * 1024.0)) +} + +/// Runs every (query, variant) pair once, prints a metrics report directly +/// pasteable into the PR reply, and asserts the pruning counts a reviewer +/// would expect so a regression fails the bench loudly rather than silently +/// producing flat numbers. +async fn print_evidence_table(dataset: &Dataset) { + let config = &dataset.config; + println!("\n=== parquet_dictionary_pruning_query: pruning evidence ==="); + println!( + "(bytes_scanned = data pages + bloom filter reads + dictionary page reads; \ + excludes footer/page-index metadata)" + ); + for (query_name, sql, predicate_column) in queries(config) { + let is_trace_lookup = + query_name == "trace_lookup" || query_name == "trace_lookup_in"; + let mut total_bytes: HashMap<&'static str, u64> = HashMap::new(); + let mut data_page_bytes: HashMap<&'static str, u64> = HashMap::new(); + for variant in &BENCH_VARIANTS { + let path = dataset.path_for(variant.file); + let evidence = scan_evidence(variant, path, &sql).await; + let line = &evidence.datasource_line; + let stats = metric_value(line, "row_groups_pruned_statistics"); + let bloom = metric_value(line, "row_groups_pruned_bloom_filter"); + let dict = metric_value(line, "row_groups_pruned_dictionary"); + println!("--- {query_name} / {} ---", variant.name); + println!(" row_groups_pruned_statistics: {stats}"); + println!(" row_groups_pruned_bloom_filter: {bloom}"); + println!(" row_groups_pruned_dictionary: {dict}"); + println!(" raw: {line}"); + + let bytes_scanned = scan_bytes(&evidence.metrics) as u64; + println!( + " bytes_scanned: {}", + format_bytes(bytes_scanned) + ); + match variant.predicate_index_kind { + IndexKind::None => { + // No index reads at all for this variant, so the whole + // scan is data pages -- a self-check for the attribution + // below: it must reproduce this same total by + // subtraction for `dictionary` and `bloom_filter`. + println!(" index bytes: 0"); + println!( + " data pages: {}", + format_bytes(bytes_scanned) + ); + data_page_bytes.insert(variant.name, bytes_scanned); + } + IndexKind::Mixed => { + println!( + " index/data split: n/a (bloom_and_dictionary \ + mixes bloom survivors with dictionary reads -- not derivable \ + from metadata)" + ); + } + IndexKind::Bloom | IndexKind::Dictionary => { + let kind = variant.predicate_index_kind; + let index = *dataset + .index_bytes + .get(&(predicate_column, kind)) + .unwrap_or_else(|| { + panic!( + "no precomputed index bytes for {predicate_column}/{kind:?} \ + -- build_dataset should have populated every (column, kind) \ + pair BENCH_VARIANTS can request" + ) + }); + let data = bytes_scanned.checked_sub(index).unwrap_or_else(|| { + panic!( + "index bytes ({index}) exceed bytes_scanned ({bytes_scanned}) \ + for {query_name} / {} -- the derived index/data split is wrong", + variant.name + ) + }); + let kind_label = if kind == IndexKind::Bloom { + "bloom filter" + } else { + "dictionary pages" + }; + println!( + " {kind_label} ({predicate_column}): {}", + format_bytes(index) + ); + println!(" data pages: {}", format_bytes(data)); + data_page_bytes.insert(variant.name, data); + } + } + total_bytes.insert(variant.name, bytes_scanned); + + // Random hex trace ids and (per-row-group) near-full-domain + // tenant sets mean statistics must never prune here -- if they + // started to, this dataset would stop isolating the + // dictionary's contribution. Only checked once, off the + // variant that has both bloom and dictionary pruning disabled, + // since the statistics stage itself doesn't depend on either + // flag. + if variant.name == "stats_only" { + let (total, matched, fully_matched) = + pruning_counts(&evidence.metrics, "row_groups_pruned_statistics"); + assert_eq!(total, config.row_groups); + assert_eq!( + matched, config.row_groups, + "statistics unexpectedly pruned a row group for {query_name} -- \ + the dataset no longer isolates the dictionary's contribution" + ); + if query_name == "tenant_not_in" { + assert_eq!( + fully_matched, 0, + "statistics unexpectedly proved a row group fully matched for \ + tenant_not_in -- the excluded values must remain within every \ + normal row group's min/max range" + ); + } + } + + if variant.bloom_filter_on_read { + let (total, matched, _) = + pruning_counts(&evidence.metrics, "row_groups_pruned_bloom_filter"); + assert_eq!(total, config.row_groups); + if query_name == "trace_lookup" { + // Bloom filters can retain false positives, so do not + // require the exact one-row-group result dictionaries + // provide. They must retain the needle's row group and + // prune at least one row group to be a useful baseline. + assert!( + matched > 0 && matched < total, + "expected bloom filters to retain the needle and prune at least \ + one row group for {query_name} ({})", + variant.name + ); + } else if query_name == "trace_lookup_in" { + // At N = TRACE_IN_LIST_LEN literals, a non-matching row + // group's false-positive probability compounds to + // 1 - (1 - fpp)^N, so bloom filters may retain most or + // even all row groups here. Only require that the + // needle's row group survives (bloom filters never + // produce false negatives) -- the retained count above + // is a printed observation, not an assertion. + assert!( + matched > 0, + "expected bloom filters to retain the needle's row group for \ + {query_name} ({})", + variant.name + ); + } else if query_name == "tenant_not_in" { + assert_eq!( + matched, total, + "bloom filters must not prune row groups for tenant_not_in ({})", + variant.name + ); + } + } + + if variant.dictionary_filter_on_read && is_trace_lookup { + let (_, matched, _) = + pruning_counts(&evidence.metrics, "row_groups_pruned_dictionary"); + assert_eq!( + matched, 1, + "expected dictionary pruning to narrow {query_name} down to \ + exactly the one row group containing the needle trace id \ + ({})", + variant.name + ); + } + + if variant.dictionary_filter_on_read && query_name == "tenant_not_in" { + let (_, matched, _) = + pruning_counts(&evidence.metrics, "row_groups_pruned_dictionary"); + let expected = config.row_groups - config.noisy_row_group_count(); + assert_eq!( + matched, expected, + "expected dictionary pruning to remove exactly the \ + tenant-noisy-only row groups for tenant_not_in ({})", + variant.name + ); + } + } + + // Cross-variant byte assertions -- deterministic given the dataset + // invariants asserted above, so a regression in the *bytes read* + // rather than the *row groups pruned* still fails loudly. + let stats_only_total = *total_bytes + .get("stats_only") + .expect("stats_only variant ran above"); + let dictionary_total = *total_bytes + .get("dictionary") + .expect("dictionary variant ran above"); + let bloom_total = *total_bytes + .get("bloom_filter") + .expect("bloom_filter variant ran above"); + assert!( + dictionary_total < stats_only_total, + "expected dictionary pruning to read fewer total bytes than stats_only for \ + {query_name} ({dictionary_total} >= {stats_only_total}) -- it keeps \ + strictly fewer row groups at no extra index cost" + ); + + let dictionary_data = *data_page_bytes + .get("dictionary") + .expect("dictionary data-page bytes computed above"); + let bloom_data = *data_page_bytes + .get("bloom_filter") + .expect("bloom_filter data-page bytes computed above"); + match query_name { + "trace_lookup" => { + // Not a strict `<`: at one literal, P(bloom filters also + // retain exactly the needle's row group and nothing else) + // = 0.95^63 ~= 4%, so a strict assertion would flake. `<=` + // still catches a real regression (dictionary reading more + // data than bloom). + assert!( + dictionary_data <= bloom_data, + "expected dictionary pruning to read no more data-page bytes than \ + bloom filters for {query_name} ({dictionary_data} > {bloom_data})" + ); + } + "trace_lookup_in" => { + // At TRACE_IN_LIST_LEN=32 literals, a non-matching row + // group's false-positive probability compounds to + // 1 - 0.95^32 ~= 81%, so bloom filters retain most row + // groups here while the dictionary stays exact at 1 -- safe + // to assert strictly. + assert!( + dictionary_data < bloom_data, + "expected dictionary pruning to read fewer data-page bytes than \ + bloom filters for {query_name} ({dictionary_data} >= {bloom_data})" + ); + } + "tenant_not_in" => { + assert!( + dictionary_data < bloom_data, + "expected dictionary pruning to read fewer data-page bytes than \ + bloom filters for {query_name} ({dictionary_data} >= {bloom_data}) \ + -- bloom filters cannot prune a NOT IN at all here" + ); + assert!( + bloom_total > stats_only_total, + "expected bloom filters to cost more total bytes than stats_only for \ + {query_name} ({bloom_total} <= {stats_only_total}) -- bloom filters \ + pay index bytes but prune nothing for a NOT IN" + ); + } + other => unreachable!("unknown query shape {other}"), + } + } + println!("=== end pruning evidence ===\n"); +} + +fn bench_query( + c: &mut Criterion, + rt: &Runtime, + dataset: &Dataset, + group_name: &str, + sql: &str, +) { + let mut group = c.benchmark_group(group_name); + for variant in &BENCH_VARIANTS { + let path = dataset.path_for(variant.file).to_path_buf(); + let query = sql.to_string(); + let variant = *variant; + group.bench_function(variant.name, |b| { + b.to_async(rt).iter(|| { + let path = path.clone(); + let query = query.clone(); + async move { black_box(run_query(&variant, &path, &query).await) } + }) + }); + } + group.finish(); +} + +fn parquet_dictionary_pruning_query(c: &mut Criterion) { + let rt = Runtime::new().expect("tokio runtime"); + let dataset = &*DATASET; + rt.block_on(print_evidence_table(dataset)); + + let config = &dataset.config; + let trace_lookup_sql = trace_lookup_query(&needle_trace_id(config)); + let trace_lookup_in_sql = trace_lookup_in_query(&needle_trace_ids_in(config)); + let tenant_not_in_sql = tenant_not_in_query(); + + bench_query(c, &rt, dataset, "trace_lookup", &trace_lookup_sql); + bench_query(c, &rt, dataset, "trace_lookup_in", &trace_lookup_in_sql); + bench_query(c, &rt, dataset, "tenant_not_in", &tenant_not_in_sql); +} + +criterion_group! { + name = benches; + config = Criterion::default() + .measurement_time(Duration::from_secs(20)) + .sample_size(10); + targets = parquet_dictionary_pruning_query +} +criterion_main!(benches); diff --git a/datafusion/datasource-parquet/Cargo.toml b/datafusion/datasource-parquet/Cargo.toml index 32424069c17a0..cff417a383f63 100644 --- a/datafusion/datasource-parquet/Cargo.toml +++ b/datafusion/datasource-parquet/Cargo.toml @@ -61,6 +61,7 @@ chrono = { workspace = true } criterion = { workspace = true } datafusion-functions = { workspace = true } datafusion-functions-nested = { workspace = true } +rand = { workspace = true, features = ["small_rng"] } tempfile = { workspace = true } # Note: add additional linter rules in lib.rs. @@ -91,3 +92,7 @@ harness = false [[bench]] name = "parquet_metadata_statistics" harness = false + +[[bench]] +name = "parquet_dictionary_pruning" +harness = false diff --git a/datafusion/datasource-parquet/benches/parquet_dictionary_pruning.rs b/datafusion/datasource-parquet/benches/parquet_dictionary_pruning.rs new file mode 100644 index 0000000000000..f78be2a29cec4 --- /dev/null +++ b/datafusion/datasource-parquet/benches/parquet_dictionary_pruning.rs @@ -0,0 +1,656 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Benchmarks the **cost** of row-group pruning by Parquet dictionaries -- +//! i.e. how expensive the pruning decision itself is, and (in the +//! `prune_and_scan` group) what that decision actually saves. See +//! `datafusion/core/benches/parquet_dictionary_pruning_query.rs` for the +//! same win measured end to end (full SQL queries, real file I/O, and +//! measured `bytes_scanned` rather than the metadata-derived estimate +//! `report_byte_accounting` below prints) against a second, span/log-shaped +//! dataset; this file isolates the pruning stage itself using a single +//! synthetic file with three query shapes. +//! +//! Compares three levels of row-group pruning: +//! +//! - `statistics_only`: min/max statistics alone (the baseline every reader +//! already gets, dictionary or bloom filter disabled). +//! - `bloom_filter`: adds Parquet Split Block Bloom Filters +//! (`bloom_filter_on_read`). +//! - `dictionary`: adds exact Parquet dictionary-page pruning +//! (`dictionary_filter_on_read`), this crate's new row-group index. +//! +//! against three query shapes, all against the same file: +//! +//! - `point_lookup`: `s = `, present in exactly one row group. +//! - `in_list`: `s IN (v0 .. v31)`, 32 literals all drawn from one row +//! group. `Guarantee::In` prunes a row group only when *every* literal is +//! absent, so a bloom filter's false-positive rate compounds per literal: +//! a non-matching row group survives with probability +//! `1 - (1 - fpp)^N`. At the default `fpp = 0.05` and `N = 32` that's +//! ~81% -- bloom filters retain most non-matching row groups here, while +//! dictionaries stay exact at any `N`. +//! - `not_in`: `c NOT IN (v0, v1)` on a second, low-cardinality column. +//! Bloom filters cannot answer "does this row group contain anything +//! *other than* these values" -- they return `None` and prune nothing; +//! dictionaries prune every row group whose value set is a subset of the +//! excluded literals. +//! +//! `s` has `TOTAL_ROW_GROUPS * DISTINCT_VALUES_PER_ROW_GROUP` distinct +//! values overall, each repeated `ROWS_PER_DISTINCT_VALUE` times within its +//! row group -- so a single row group's dictionary is small and never falls +//! back to `PLAIN`, while the column stays high-cardinality overall. `c` +//! mirrors the query bench's `tenant` construction: normal row groups draw +//! from a small set with the extrema pinned so statistics can't prune them, +//! and 3 of every 4 row groups are "noisy" -- drawn only from the two +//! excluded values -- so only dictionary pruning can remove them. +//! +//! Two criterion groups exercise the three strategies over the three query +//! shapes (`{statistics_only,bloom_filter,dictionary}` x +//! `{point_lookup,in_list,not_in}`): +//! +//! - `prune_only` times only the pruning decision itself -- real work +//! (bloom filters and dictionaries must be read and decoded for every +//! surviving row group), but not the data-page reads it saves. +//! - `prune_and_scan` prunes, then actually reads the surviving row groups' +//! data pages via `ParquetRecordBatchReaderBuilder::with_row_groups`. +//! This is the total a user actually pays, and it's where exactness shows +//! up: `dictionary` should win here even where it loses in `prune_only`. +//! +//! Run with `cargo bench -p datafusion-datasource-parquet --bench parquet_dictionary_pruning`. + +use std::hint::black_box; +use std::path::{Path, PathBuf}; +use std::sync::{Arc, LazyLock}; + +use arrow::array::{ArrayRef, RecordBatch, StringArray}; +use arrow::datatypes::{DataType, Field, Schema, SchemaRef}; +use criterion::{BenchmarkId, Criterion, Throughput, criterion_group, criterion_main}; +use datafusion_datasource_parquet::{ + BloomFilterStatistics, DictionaryStatistics, ParquetAccessPlan, ParquetFileMetrics, + RowGroupAccessPlanFilter, is_fully_dictionary_encoded, +}; +use datafusion_expr::{Expr, col, lit}; +use datafusion_physical_expr::planner::logical2physical; +use datafusion_physical_plan::metrics::ExecutionPlanMetricsSet; +use datafusion_pruning::PruningPredicate; +use parquet::arrow::arrow_reader::ParquetRecordBatchReaderBuilder; +use parquet::arrow::{ArrowWriter, parquet_column}; +use parquet::file::metadata::ParquetMetaDataReader; +use parquet::file::properties::WriterProperties; +use parquet::file::reader::{FileReader, SerializedFileReader}; +use parquet::schema::types::SchemaDescriptor; +use rand::rngs::SmallRng; +use rand::{Rng, SeedableRng}; +use tempfile::TempDir; + +/// Fixed seed so the generated `c` column (and its noisy/normal row group +/// assignment) is identical from run to run. +const SEED: u64 = 0xC0FF_EE00_D1C7_5EED; + +const TOTAL_ROW_GROUPS: usize = 200; +const DISTINCT_VALUES_PER_ROW_GROUP: usize = 1_000; +/// Each distinct `s` value is repeated this many times within its row +/// group, so the dictionary indexes real repeated data rather than being +/// the entire column chunk (the pathological, maximally-expensive case). +const ROWS_PER_DISTINCT_VALUE: usize = 8; +const ROWS_PER_ROW_GROUP: usize = DISTINCT_VALUES_PER_ROW_GROUP * ROWS_PER_DISTINCT_VALUE; +const TOTAL_VALUES: usize = TOTAL_ROW_GROUPS * DISTINCT_VALUES_PER_ROW_GROUP; +const COLUMN_NAME: &str = "s"; + +const IN_LIST_LEN: usize = 32; + +const CAT_COLUMN_NAME: &str = "c"; +const NORMAL_CAT_COUNT: u32 = 20; +// These sort between `cat-09` and `cat-10`, keeping them inside every +// normal row group's `[min, max]` range so statistics cannot prune a row +// group for `c NOT IN (...)`, only the dictionary can. +const NOT_IN_VALUE_A: &str = "cat-09-a"; +const NOT_IN_VALUE_B: &str = "cat-09-b"; +/// Every `NORMAL_ROW_GROUP_STRIDE`-th row group draws `c` from the full +/// normal domain; the rest are "noisy" row groups whose `c` column is drawn +/// only from `{NOT_IN_VALUE_A, NOT_IN_VALUE_B}`, so their dictionary is a +/// subset of the excluded values while `min != max` (statistics can't prune +/// them, only the dictionary can). Mirrors the query bench's `tenant` +/// construction: 3 of every 4 row groups here are noisy. +const NORMAL_ROW_GROUP_STRIDE: usize = 4; + +fn is_noisy_row_group(rg: usize) -> bool { + !rg.is_multiple_of(NORMAL_ROW_GROUP_STRIDE) +} + +fn normal_row_group_count() -> usize { + (0..TOTAL_ROW_GROUPS) + .filter(|rg| !is_noisy_row_group(*rg)) + .count() +} + +/// Interleaves (round-robins) values across row groups: row group `rg` gets +/// the values at global positions `rg, rg + TOTAL_ROW_GROUPS, rg + 2 * +/// TOTAL_ROW_GROUPS, ...`. Every row group's `[min, max]` therefore spans +/// almost the entire value domain (min is close to 0, max close to +/// `TOTAL_VALUES`), so plain min/max statistics can't prune any of them -- +/// only the *set* of values actually present (from a bloom filter or exact +/// dictionary) can distinguish row groups. This mirrors data that arrives +/// already shuffled with respect to a low/no-correlation column, e.g. trace +/// IDs or session IDs sharded across row groups by arrival time. +fn value_at(rg: usize, k: usize) -> String { + format!("val-{:06}", rg + k * TOTAL_ROW_GROUPS) +} + +/// The row group and local (per-row-group) index of the needle value used +/// by `point_lookup` and `in_list`, chosen near the middle of the value +/// domain so every row group's `[min, max]` range contains it, but only one +/// row group's dictionary actually does. +fn needle_row_group() -> usize { + (TOTAL_VALUES / 2) % TOTAL_ROW_GROUPS +} + +fn needle_local_index() -> usize { + (TOTAL_VALUES / 2) / TOTAL_ROW_GROUPS +} + +/// Present in exactly one row group. +fn needle() -> String { + value_at(needle_row_group(), needle_local_index()) +} + +/// `IN_LIST_LEN` values, all drawn from the same single row group as +/// [`needle`]. +fn needle_in_list() -> Vec { + let rg = needle_row_group(); + let k0 = needle_local_index(); + assert!( + k0 + IN_LIST_LEN <= DISTINCT_VALUES_PER_ROW_GROUP, + "IN_LIST_LEN ({IN_LIST_LEN}) overruns the needle row group's distinct values" + ); + (0..IN_LIST_LEN).map(|k| value_at(rg, k0 + k)).collect() +} + +struct BenchmarkDataset { + _tempdir: TempDir, + file_path: PathBuf, +} + +impl BenchmarkDataset { + fn path(&self) -> &Path { + &self.file_path + } +} + +static DATASET: LazyLock = LazyLock::new(|| { + create_dataset().expect("failed to prepare parquet benchmark dataset") +}); + +fn schema() -> SchemaRef { + Arc::new(Schema::new(vec![ + Field::new(COLUMN_NAME, DataType::Utf8, false), + Field::new(CAT_COLUMN_NAME, DataType::Utf8, false), + ])) +} + +/// Builds one row group's data for both `s` (high-cardinality overall, but +/// small and fully dictionary-encoded per row group) and `c` (low +/// cardinality, normal vs. noisy per [`is_noisy_row_group`]). +fn generate_row_group(rg: usize) -> RecordBatch { + let mut rng = SmallRng::seed_from_u64(SEED.wrapping_add(rg as u64)); + let noisy = is_noisy_row_group(rg); + + let mut s_values: Vec = Vec::with_capacity(ROWS_PER_ROW_GROUP); + for k in 0..DISTINCT_VALUES_PER_ROW_GROUP { + let v = value_at(rg, k); + for _ in 0..ROWS_PER_DISTINCT_VALUE { + s_values.push(v.clone()); + } + } + + let mut c_values: Vec = Vec::with_capacity(ROWS_PER_ROW_GROUP); + for i in 0..ROWS_PER_ROW_GROUP { + c_values.push(if noisy { + if i % 2 == 0 { + NOT_IN_VALUE_A.to_string() + } else { + NOT_IN_VALUE_B.to_string() + } + } else if i == 0 { + // Pin both extrema so every normal row group contains the + // excluded values within, rather than outside, its statistics. + "cat-00".to_string() + } else if i == 1 { + format!("cat-{:02}", NORMAL_CAT_COUNT - 1) + } else { + format!("cat-{:02}", rng.random_range(0..NORMAL_CAT_COUNT)) + }); + } + + let s_array: ArrayRef = Arc::new(StringArray::from_iter_values( + s_values.iter().map(|v| v.as_str()), + )); + let c_array: ArrayRef = Arc::new(StringArray::from_iter_values( + c_values.iter().map(|v| v.as_str()), + )); + + RecordBatch::try_new(schema(), vec![s_array, c_array]).expect("valid record batch") +} + +fn create_dataset() -> datafusion_common::Result { + let tempdir = TempDir::new()?; + let file_path = tempdir.path().join("dictionary_pruning.parquet"); + + let schema = schema(); + // Dictionary and bloom filter both enabled, so the same file drives all + // three benchmark strategies below. + let writer_props = WriterProperties::builder() + .set_max_row_group_row_count(Some(ROWS_PER_ROW_GROUP)) + .set_dictionary_enabled(true) + .set_bloom_filter_enabled(true) + .build(); + + let mut writer = ArrowWriter::try_new( + std::fs::File::create(&file_path)?, + Arc::clone(&schema), + Some(writer_props), + )?; + + for rg in 0..TOTAL_ROW_GROUPS { + let batch = generate_row_group(rg); + writer.write(&batch)?; + } + writer.close()?; + + let reader = + ParquetRecordBatchReaderBuilder::try_new(std::fs::File::open(&file_path)?)?; + let metadata = reader.metadata(); + assert_eq!(metadata.num_row_groups(), TOTAL_ROW_GROUPS); + + // Load-bearing property: neither column's dictionary may have + // overflowed into a PLAIN fallback. If one had, `is_fully_dictionary_encoded` + // would (correctly) refuse the chunk and the corresponding `dictionary` + // benchmark scenario would silently degrade to "no pruning" -- which + // would look like a regression in the numbers rather than a broken + // dataset. + let s_idx = schema.index_of(COLUMN_NAME).expect("s column"); + let c_idx = schema.index_of(CAT_COLUMN_NAME).expect("c column"); + for rg in 0..TOTAL_ROW_GROUPS { + assert!( + is_fully_dictionary_encoded(metadata.row_group(rg).column(s_idx)), + "s dictionary overflowed to PLAIN in row group {rg}" + ); + assert!( + is_fully_dictionary_encoded(metadata.row_group(rg).column(c_idx)), + "c dictionary overflowed to PLAIN in row group {rg}" + ); + } + + Ok(BenchmarkDataset { + _tempdir: tempdir, + file_path, + }) +} + +/// One of the three query shapes benchmarked below: a name, the column it +/// filters on, and how to build its [`PruningPredicate`] against a given +/// file's schema descriptor. +#[derive(Clone, Copy)] +struct QueryShape { + name: &'static str, + column_name: &'static str, + build_predicate: fn(&SchemaDescriptor) -> (PruningPredicate, usize), +} + +fn build_predicate( + parquet_schema: &SchemaDescriptor, + column_name: &str, + expr: &Expr, +) -> (PruningPredicate, usize) { + let schema = schema(); + let physical_expr = logical2physical(expr, &schema); + let predicate = + PruningPredicate::try_new(physical_expr, schema).expect("valid predicate"); + let (column_idx, _) = parquet_column(parquet_schema, predicate.schema(), column_name) + .expect("column present"); + (predicate, column_idx) +} + +fn point_lookup_predicate( + parquet_schema: &SchemaDescriptor, +) -> (PruningPredicate, usize) { + build_predicate( + parquet_schema, + COLUMN_NAME, + &col(COLUMN_NAME).eq(lit(needle())), + ) +} + +fn in_list_predicate(parquet_schema: &SchemaDescriptor) -> (PruningPredicate, usize) { + let list = needle_in_list().into_iter().map(lit).collect(); + build_predicate( + parquet_schema, + COLUMN_NAME, + &col(COLUMN_NAME).in_list(list, false), + ) +} + +fn not_in_predicate(parquet_schema: &SchemaDescriptor) -> (PruningPredicate, usize) { + let list = vec![lit(NOT_IN_VALUE_A), lit(NOT_IN_VALUE_B)]; + build_predicate( + parquet_schema, + CAT_COLUMN_NAME, + &col(CAT_COLUMN_NAME).in_list(list, true), + ) +} + +const QUERY_SHAPES: [QueryShape; 3] = [ + QueryShape { + name: "point_lookup", + column_name: COLUMN_NAME, + build_predicate: point_lookup_predicate, + }, + QueryShape { + name: "in_list", + column_name: COLUMN_NAME, + build_predicate: in_list_predicate, + }, + QueryShape { + name: "not_in", + column_name: CAT_COLUMN_NAME, + build_predicate: not_in_predicate, + }, +]; + +/// Total compressed bytes of the queried column across `indexes`' row +/// groups -- i.e. the data-page bytes an actual scan would have to read for +/// the row groups that survive pruning. +fn column_bytes( + metadata: &parquet::file::metadata::ParquetMetaData, + column_idx: usize, + indexes: impl Iterator, +) -> i64 { + indexes + .map(|idx| metadata.row_group(idx).column(column_idx).compressed_size()) + .sum() +} + +fn prune_by_statistics_only(path: &Path, shape: &QueryShape) -> Vec { + let file = std::fs::File::open(path).expect("open file"); + let reader = SerializedFileReader::new(file).expect("open reader"); + let metadata = reader.metadata(); + let (predicate, _) = (shape.build_predicate)(metadata.file_metadata().schema_descr()); + + let mut access_plan = RowGroupAccessPlanFilter::new(ParquetAccessPlan::new_all( + metadata.num_row_groups(), + )); + let metrics_set = ExecutionPlanMetricsSet::new(); + let metrics = ParquetFileMetrics::new(0, &path.display().to_string(), &metrics_set); + access_plan.prune_by_statistics( + predicate.schema(), + metadata.file_metadata().schema_descr(), + metadata.row_groups(), + &predicate, + &metrics, + ); + access_plan.row_group_indexes().collect() +} + +fn prune_by_bloom_filter(path: &Path, shape: &QueryShape) -> Vec { + let file = std::fs::File::open(path).expect("open file"); + let builder = ParquetRecordBatchReaderBuilder::try_new(file).expect("open reader"); + let metadata = builder.metadata().clone(); + let (predicate, column_idx) = + (shape.build_predicate)(metadata.file_metadata().schema_descr()); + let physical_type = metadata + .file_metadata() + .schema_descr() + .column(column_idx) + .physical_type(); + let type_length = metadata + .file_metadata() + .schema_descr() + .column(column_idx) + .type_length(); + + let mut access_plan = RowGroupAccessPlanFilter::new(ParquetAccessPlan::new_all( + metadata.num_row_groups(), + )); + let metrics_set = ExecutionPlanMetricsSet::new(); + let metrics = ParquetFileMetrics::new(0, &path.display().to_string(), &metrics_set); + access_plan.prune_by_statistics( + predicate.schema(), + metadata.file_metadata().schema_descr(), + metadata.row_groups(), + &predicate, + &metrics, + ); + + let mut row_group_bloom_filters = + vec![BloomFilterStatistics::new(); metadata.num_row_groups()]; + for idx in access_plan.row_group_indexes() { + let mut stats = BloomFilterStatistics::with_capacity(1); + if let Ok(Some(bf)) = builder.get_row_group_column_bloom_filter(idx, column_idx) { + stats.insert(shape.column_name, bf, physical_type, type_length); + } + row_group_bloom_filters[idx] = stats; + } + access_plan.prune_by_bloom_filters(&predicate, &metrics, &row_group_bloom_filters); + access_plan.row_group_indexes().collect() +} + +fn prune_by_dictionary(path: &Path, shape: &QueryShape) -> Vec { + let file = std::fs::File::open(path).expect("open file"); + let reader = SerializedFileReader::new(file).expect("open reader"); + let metadata = reader.metadata(); + let (predicate, column_idx) = + (shape.build_predicate)(metadata.file_metadata().schema_descr()); + + let mut access_plan = RowGroupAccessPlanFilter::new(ParquetAccessPlan::new_all( + metadata.num_row_groups(), + )); + let metrics_set = ExecutionPlanMetricsSet::new(); + let metrics = ParquetFileMetrics::new(0, &path.display().to_string(), &metrics_set); + access_plan.prune_by_statistics( + predicate.schema(), + metadata.file_metadata().schema_descr(), + metadata.row_groups(), + &predicate, + &metrics, + ); + + let file = std::fs::File::open(path).expect("open file"); + let mut row_group_dictionaries = + vec![DictionaryStatistics::new(); metadata.num_row_groups()]; + for idx in access_plan.row_group_indexes() { + let col_meta = metadata.row_group(idx).column(column_idx); + if !is_fully_dictionary_encoded(col_meta) { + continue; + } + let mut stats = DictionaryStatistics::with_capacity(1); + if let Ok(Some(dict)) = ParquetMetaDataReader::read_column_dictionary( + &file, metadata, idx, column_idx, + ) { + stats + .insert(shape.column_name, &dict) + .expect("decode dictionary"); + } + row_group_dictionaries[idx] = stats; + } + access_plan.prune_by_dictionary(&predicate, &metrics, &row_group_dictionaries); + access_plan.row_group_indexes().collect() +} + +/// Prunes are cheap relative to a scan; this reads and decodes the data +/// pages of exactly the row groups pruning left standing, which is the +/// total cost a real query actually pays. Returns the row count so the +/// result can't be optimized away. +fn scan_survivors(path: &Path, indexes: &[usize]) -> usize { + let file = std::fs::File::open(path).expect("open file"); + let builder = ParquetRecordBatchReaderBuilder::try_new(file).expect("open reader"); + let reader = builder + .with_row_groups(indexes.to_vec()) + .build() + .expect("build reader"); + reader + .map(|batch| batch.expect("read batch").num_rows()) + .sum() +} + +/// Sanity-checks each strategy's pruning result against the shape it should +/// (or structurally cannot) exploit, and reports the column bytes an actual +/// scan would read for each -- the real payoff `prune_only` doesn't time. +fn report_byte_accounting(dataset_path: &Path) { + let file = std::fs::File::open(dataset_path).expect("open file"); + let metadata = SerializedFileReader::new(file) + .expect("open reader") + .metadata() + .clone(); + + for shape in &QUERY_SHAPES { + let (_, column_idx) = + (shape.build_predicate)(metadata.file_metadata().schema_descr()); + + // Every row group's [min, max] range spans the needle(s) (`s` is + // interleaved across row groups, `c`'s excluded values sort inside + // every normal row group's range), so statistics alone can't prune + // any of them for any of the three shapes -- this is the baseline + // the other two strategies are compared against. + let statistics_only = prune_by_statistics_only(dataset_path, shape); + assert_eq!( + statistics_only.len(), + TOTAL_ROW_GROUPS, + "statistics unexpectedly pruned a row group for {} -- the dataset no \ + longer isolates the dictionary's contribution", + shape.name + ); + + let bloom_filter = prune_by_bloom_filter(dataset_path, shape); + let dictionary = prune_by_dictionary(dataset_path, shape); + + match shape.name { + "point_lookup" => { + // Bloom filters are probabilistic but should reliably prune + // this exact, absent-from-most-groups scenario down to + // (approximately) one row group. + assert!(bloom_filter.len() <= TOTAL_ROW_GROUPS); + assert_eq!( + dictionary.len(), + 1, + "expected dictionary pruning to narrow point_lookup down to \ + exactly the one row group containing the needle" + ); + } + "in_list" => { + // The needle's row group must always survive (bloom filters + // never produce false negatives); how many others also + // survive is the compounding-false-positive-rate + // observation from the module doc, not an assertion. + assert!( + !bloom_filter.is_empty(), + "expected bloom filters to retain the needle's row group for in_list" + ); + assert_eq!( + dictionary.len(), + 1, + "expected dictionary pruning to narrow in_list down to exactly \ + the one row group containing all the needles" + ); + } + "not_in" => { + assert_eq!( + bloom_filter.len(), + TOTAL_ROW_GROUPS, + "bloom filters must not prune any row group for not_in -- they \ + cannot prove a row group contains nothing but excluded values" + ); + assert_eq!( + dictionary.len(), + normal_row_group_count(), + "expected dictionary pruning to remove exactly the noisy row \ + groups for not_in" + ); + } + other => unreachable!("unknown query shape {other}"), + } + + let statistics_only_bytes = + column_bytes(&metadata, column_idx, statistics_only.into_iter()); + let bloom_filter_bytes = + column_bytes(&metadata, column_idx, bloom_filter.into_iter()); + let dictionary_bytes = + column_bytes(&metadata, column_idx, dictionary.into_iter()); + eprintln!( + "parquet_dictionary_pruning: {} column data bytes an actual scan would \ + read -- statistics_only: {statistics_only_bytes}, bloom_filter: \ + {bloom_filter_bytes}, dictionary: {dictionary_bytes}", + shape.name + ); + } +} + +fn parquet_dictionary_pruning(c: &mut Criterion) { + let dataset_path = DATASET.path().to_owned(); + + report_byte_accounting(&dataset_path); + + let mut prune_only = c.benchmark_group("prune_only"); + prune_only.throughput(Throughput::Elements(TOTAL_ROW_GROUPS as u64)); + for shape in &QUERY_SHAPES { + let path = dataset_path.clone(); + prune_only.bench_function(BenchmarkId::new("statistics_only", shape.name), |b| { + b.iter(|| black_box(prune_by_statistics_only(&path, shape))); + }); + prune_only.bench_function(BenchmarkId::new("bloom_filter", shape.name), |b| { + b.iter(|| black_box(prune_by_bloom_filter(&path, shape))); + }); + prune_only.bench_function(BenchmarkId::new("dictionary", shape.name), |b| { + b.iter(|| black_box(prune_by_dictionary(&path, shape))); + }); + } + prune_only.finish(); + + let mut prune_and_scan = c.benchmark_group("prune_and_scan"); + prune_and_scan.throughput(Throughput::Elements(TOTAL_ROW_GROUPS as u64)); + for shape in &QUERY_SHAPES { + let path = dataset_path.clone(); + prune_and_scan.bench_function( + BenchmarkId::new("statistics_only", shape.name), + |b| { + b.iter(|| { + let indexes = prune_by_statistics_only(&path, shape); + black_box(scan_survivors(&path, &indexes)) + }); + }, + ); + prune_and_scan.bench_function( + BenchmarkId::new("bloom_filter", shape.name), + |b| { + b.iter(|| { + let indexes = prune_by_bloom_filter(&path, shape); + black_box(scan_survivors(&path, &indexes)) + }); + }, + ); + prune_and_scan.bench_function(BenchmarkId::new("dictionary", shape.name), |b| { + b.iter(|| { + let indexes = prune_by_dictionary(&path, shape); + black_box(scan_survivors(&path, &indexes)) + }); + }); + } + prune_and_scan.finish(); +} + +criterion_group!(benches, parquet_dictionary_pruning); +criterion_main!(benches); diff --git a/datafusion/datasource-parquet/src/dictionary_filter.rs b/datafusion/datasource-parquet/src/dictionary_filter.rs new file mode 100644 index 0000000000000..c5210554bb24e --- /dev/null +++ b/datafusion/datasource-parquet/src/dictionary_filter.rs @@ -0,0 +1,560 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Loaded Parquet dictionary-page data, with a [`PruningStatistics`] adapter +//! so the predicate-pruning machinery in [`datafusion_pruning`] can consume +//! it. +//! +//! Some writers (e.g. Grafana Tempo) use a low-cardinality string column's +//! Parquet dictionary as a makeshift row-group index: the dictionary page is +//! the exact, complete set of the row group's distinct non-null values. If a +//! required value is not in the dictionary, the whole row group can be +//! skipped without reading any data pages. This is a strictly stronger +//! (exact, non-probabilistic) signal than a bloom filter, letting us prune +//! both `IN`/`=` (value absent) and `NOT IN`/`!=` (value is the only one +//! present) directions -- see [`DictionaryStatistics::contained`]. + +use std::collections::{HashMap, HashSet}; + +use arrow::array::{ArrayRef, BinaryArray, BooleanArray, StringArray}; +use arrow::datatypes::DataType; +use datafusion_common::pruning::PruningStatistics; +use datafusion_common::{Column, DataFusionError, Result, ScalarValue, internal_err}; +use parquet::basic::{Encoding, Type}; +use parquet::file::metadata::ColumnChunkMetaData; + +/// In-memory decoded Parquet dictionary-page values, keyed by column name. +/// +/// This structure implements [`PruningStatistics`] and is used to prune +/// Parquet row groups based on the query predicate. Unlike bloom filters, +/// which are probabilistic, the values stored here are assumed to be the +/// *exact* set of distinct values present in the column for the row group -- +/// callers must only [`insert`](Self::insert) dictionaries for column chunks +/// that pass [`is_fully_dictionary_encoded`], or pruning will be unsound. +#[derive(Debug, Clone, Default)] +pub struct DictionaryStatistics { + /// Per-column exact value sets, keyed by predicate column name. Values + /// are stored as raw bytes so `Utf8`- and `Binary`-typed dictionaries + /// compare equally to whichever `ScalarValue` variant the predicate + /// literal happens to use (see [`normalize_literal`]). + column_values: HashMap>>, +} + +impl DictionaryStatistics { + /// Create an empty [`DictionaryStatistics`] + pub fn new() -> Self { + Default::default() + } + + /// Create an empty [`DictionaryStatistics`] with the specified capacity + pub fn with_capacity(capacity: usize) -> Self { + Self { + column_values: HashMap::with_capacity(capacity), + } + } + + /// Record the exact dictionary values for `column`, decoded as a + /// `Utf8` or `Binary` array (e.g. from + /// `ParquetRecordBatchStreamBuilder::get_row_group_column_dictionary`). + /// + /// # Panics / Errors + /// + /// This does not itself verify that the column chunk is fully + /// dictionary-encoded -- see [`is_fully_dictionary_encoded`]. Passing a + /// dictionary that is only a subset of the chunk's actual values (e.g. + /// because part of the chunk fell back to `PLAIN`) will make pruning + /// unsound. + pub fn insert( + &mut self, + column: impl Into, + dictionary: &ArrayRef, + ) -> Result<()> { + let values = match dictionary.data_type() { + DataType::Utf8 => dictionary + .as_any() + .downcast_ref::() + .ok_or_else(|| { + DataFusionError::Internal("Expected a StringArray".to_string()) + })? + .iter() + .flatten() + .map(|v| v.as_bytes().to_vec()) + .collect(), + DataType::Binary => dictionary + .as_any() + .downcast_ref::() + .ok_or_else(|| { + DataFusionError::Internal("Expected a BinaryArray".to_string()) + })? + .iter() + .flatten() + .map(|v| v.to_vec()) + .collect(), + other => { + return internal_err!( + "DictionaryStatistics only supports Utf8/Binary dictionaries, got {other}" + ); + } + }; + self.column_values.insert(column.into(), values); + Ok(()) + } +} + +/// Returns `value` normalized to its raw byte representation, if `value` is +/// a string- or binary-like scalar. Mirrors the literal normalization used +/// by [`crate::bloom_filter::BloomFilterStatistics`] so a predicate literal +/// (which may be `Utf8`, `Utf8View`, `LargeUtf8`, etc. depending on how the +/// query was planned) compares equal to a dictionary entry regardless of +/// which of those variants was used. +fn normalize_literal(value: &ScalarValue) -> Option<&[u8]> { + match value { + ScalarValue::Utf8(Some(v)) + | ScalarValue::Utf8View(Some(v)) + | ScalarValue::LargeUtf8(Some(v)) => Some(v.as_bytes()), + ScalarValue::Binary(Some(v)) + | ScalarValue::BinaryView(Some(v)) + | ScalarValue::LargeBinary(Some(v)) => Some(v.as_slice()), + ScalarValue::Dictionary(_, inner) => normalize_literal(inner), + _ => None, + } +} + +impl PruningStatistics for DictionaryStatistics { + fn min_values(&self, _column: &Column) -> Option { + None + } + + fn max_values(&self, _column: &Column) -> Option { + None + } + + fn num_containers(&self) -> usize { + 1 + } + + fn null_counts(&self, _column: &Column) -> Option { + None + } + + fn row_counts(&self) -> Option { + None + } + + /// Use the exact dictionary values to determine whether the column is + /// definitely disjoint from (`Some(false)`), or definitely a subset of + /// (`Some(true)`), `values`. + /// + /// Because the dictionary is the row group's *complete* set of distinct + /// values (guaranteed by the caller checking + /// [`is_fully_dictionary_encoded`] before inserting it), this can prove + /// both directions exactly, unlike a bloom filter's probabilistic + /// "definitely absent" only. + fn contained( + &self, + column: &Column, + values: &HashSet, + ) -> Option { + let dictionary_values = self.column_values.get(column.name.as_str())?; + + let mut literal_bytes: HashSet<&[u8]> = HashSet::with_capacity(values.len()); + for value in values { + // A literal we can't normalize (e.g. a non-string/binary type + // that somehow ended up compared against this column) makes the + // guarantee inconclusive rather than wrong. + literal_bytes.insert(normalize_literal(value)?); + } + + let none_present = literal_bytes + .iter() + .all(|literal| !dictionary_values.contains(*literal)); + if none_present { + return Some(BooleanArray::from(vec![Some(false)])); + } + + let all_present = dictionary_values + .iter() + .all(|value| literal_bytes.contains(value.as_slice())); + if all_present { + return Some(BooleanArray::from(vec![Some(true)])); + } + + Some(BooleanArray::from(vec![None])) + } +} + +/// Returns `true` if `col_meta`'s column chunk is entirely `BYTE_ARRAY` and +/// dictionary-encoded, i.e. its dictionary page is the exact, complete set of +/// the row group's distinct non-null values for that column. +/// +/// Dictionary encoding is best-effort: writers fall back to `PLAIN` once a +/// column's dictionary grows past a size limit, and once that happens the +/// data pages after the fallback contain values that never entered the +/// dictionary. `dictionary_page_offset().is_some()` alone does not rule this +/// out. Page-encoding statistics +/// () record every +/// encoding used across a column chunk's data pages, so requiring the mask +/// to contain *only* a dictionary encoding proves every data page decoded +/// from the dictionary. +/// +/// First cut: only `BYTE_ARRAY` (`Utf8`/`Binary`) columns are supported. +pub fn is_fully_dictionary_encoded(col_meta: &ColumnChunkMetaData) -> bool { + col_meta.column_descr().physical_type() == Type::BYTE_ARRAY + && col_meta.dictionary_page_offset().is_some() + && col_meta.page_encoding_stats_mask().is_some_and(|mask| { + mask.is_only(Encoding::PLAIN_DICTIONARY) + || mask.is_only(Encoding::RLE_DICTIONARY) + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + use std::sync::Arc; + + use crate::reader::ParquetFileReader; + use crate::test_util::ExpectedPruning; + use crate::{ParquetAccessPlan, ParquetFileMetrics, RowGroupAccessPlanFilter}; + + use arrow::datatypes::{Field, Schema}; + use bytes::{BufMut, BytesMut}; + use datafusion_expr::{Expr, col, lit}; + use datafusion_physical_expr::planner::logical2physical; + use datafusion_physical_plan::metrics::ExecutionPlanMetricsSet; + use datafusion_pruning::PruningPredicate; + use object_store::ObjectStoreExt; + use parquet::arrow::ArrowWriter; + use parquet::arrow::ParquetRecordBatchStreamBuilder; + use parquet::arrow::async_reader::ParquetObjectReader; + use parquet::arrow::parquet_column; + use parquet::basic::EncodingMask; + use parquet::file::metadata::ColumnChunkMetaData; + use parquet::file::properties::WriterProperties; + use parquet::schema::types::{SchemaDescriptor, Type as SchemaType}; + + #[test] + fn is_fully_dictionary_encoded_requires_only_dictionary_encoding() { + let schema_descr = Arc::new(SchemaDescriptor::new(Arc::new( + SchemaType::group_type_builder("schema") + .with_fields(vec![Arc::new( + SchemaType::primitive_type_builder("s", Type::BYTE_ARRAY) + .build() + .unwrap(), + )]) + .build() + .unwrap(), + ))); + + let fully_dictionary_encoded = + ColumnChunkMetaData::builder(schema_descr.column(0)) + .set_dictionary_page_offset(Some(0)) + .set_data_page_offset(100) + .set_page_encoding_stats_mask(EncodingMask::new_from_encodings( + [Encoding::RLE_DICTIONARY].iter(), + )) + .build() + .unwrap(); + assert!(is_fully_dictionary_encoded(&fully_dictionary_encoded)); + + // Plain fallback mid-chunk: a dictionary page exists, but at least + // one data page used PLAIN, so the dictionary is not exhaustive. + let plain_fallback = ColumnChunkMetaData::builder(schema_descr.column(0)) + .set_dictionary_page_offset(Some(0)) + .set_data_page_offset(100) + .set_page_encoding_stats_mask(EncodingMask::new_from_encodings( + [Encoding::RLE_DICTIONARY, Encoding::PLAIN].iter(), + )) + .build() + .unwrap(); + assert!(!is_fully_dictionary_encoded(&plain_fallback)); + + // No dictionary page at all. + let no_dictionary = ColumnChunkMetaData::builder(schema_descr.column(0)) + .set_data_page_offset(100) + .set_page_encoding_stats_mask(EncodingMask::new_from_encodings( + [Encoding::PLAIN].iter(), + )) + .build() + .unwrap(); + assert!(!is_fully_dictionary_encoded(&no_dictionary)); + + // No page encoding stats recorded at all: treat as not prunable. + let no_stats = ColumnChunkMetaData::builder(schema_descr.column(0)) + .set_dictionary_page_offset(Some(0)) + .set_data_page_offset(100) + .build() + .unwrap(); + assert!(!is_fully_dictionary_encoded(&no_stats)); + } + + struct DictionaryFilterTest { + schema: Schema, + post_pruning_row_groups: ExpectedPruning, + } + + impl DictionaryFilterTest { + /// A small dictionary-encoded string column with three distinct + /// values, one row group per value so pruning is easy to verify. + fn new_single_value_row_groups() -> (Self, bytes::Bytes) { + let schema = Schema::new(vec![Field::new("s", DataType::Utf8, false)]); + let arrow_schema = Arc::new(schema.clone()); + let values = ["alpha", "beta", "gamma"]; + let array: ArrayRef = Arc::new(StringArray::from_iter_values(values)); + let batch = + arrow::array::RecordBatch::try_new(arrow_schema.clone(), vec![array]) + .unwrap(); + + let props = WriterProperties::builder() + .set_dictionary_enabled(true) + .set_max_row_group_row_count(Some(1)) + .build(); + let mut out = BytesMut::new().writer(); + { + let mut writer = + ArrowWriter::try_new(&mut out, arrow_schema, Some(props)).unwrap(); + writer.write(&batch).unwrap(); + writer.finish().unwrap(); + } + + ( + Self { + schema, + post_pruning_row_groups: ExpectedPruning::None, + }, + out.into_inner().freeze(), + ) + } + + fn with_expect_all_pruned(mut self) -> Self { + self.post_pruning_row_groups = ExpectedPruning::All; + self + } + + fn with_expect_some_pruned(mut self, remaining: Vec) -> Self { + self.post_pruning_row_groups = ExpectedPruning::Some(remaining); + self + } + + async fn run(self, data: bytes::Bytes, expr: Expr) { + let Self { + schema, + post_pruning_row_groups, + } = self; + + let expr = logical2physical(&expr, &Arc::new(schema)); + let pruning_predicate = PruningPredicate::try_new( + expr, + Arc::new(Schema::new(vec![Field::new("s", DataType::Utf8, false)])), + ) + .unwrap(); + + let pruned_row_groups = + test_row_group_dictionary_pruning_predicate(data, &pruning_predicate) + .await + .unwrap(); + + post_pruning_row_groups.assert(&pruned_row_groups); + } + } + + /// Evaluates the pruning predicate on the specified row groups and + /// returns the row groups that are left, loading dictionaries exactly + /// like [`crate::opener`]'s dictionary-loading stage does. + async fn test_row_group_dictionary_pruning_predicate( + data: bytes::Bytes, + pruning_predicate: &PruningPredicate, + ) -> Result { + use datafusion_datasource::PartitionedFile; + use object_store::ObjectMeta; + + let object_meta = ObjectMeta { + location: object_store::path::Path::parse("test.parquet") + .expect("creating path"), + last_modified: chrono::DateTime::from(std::time::SystemTime::now()), + size: data.len() as u64, + e_tag: None, + version: None, + }; + let in_memory = object_store::memory::InMemory::new(); + in_memory + .put(&object_meta.location, data.into()) + .await + .expect("put parquet file into in memory object store"); + + let metrics = ExecutionPlanMetricsSet::new(); + let file_metrics = + ParquetFileMetrics::new(0, object_meta.location.as_ref(), &metrics); + let inner = + ParquetObjectReader::new(Arc::new(in_memory), object_meta.location.clone()) + .with_file_size(object_meta.size); + + let partitioned_file = PartitionedFile::new_from_meta(object_meta); + + let reader = ParquetFileReader { + inner, + file_metrics: file_metrics.clone(), + partitioned_file, + }; + let mut builder = ParquetRecordBatchStreamBuilder::new(reader).await.unwrap(); + + let access_plan = ParquetAccessPlan::new_all(builder.metadata().num_row_groups()); + let mut pruned_row_groups = RowGroupAccessPlanFilter::new(access_plan); + let literal_columns = pruning_predicate.literal_columns(); + let parquet_columns: Vec<_> = literal_columns + .into_iter() + .filter_map(|column_name| { + let (column_idx, _) = parquet_column( + builder.parquet_schema(), + pruning_predicate.schema(), + &column_name, + )?; + Some((column_name.to_string(), column_idx)) + }) + .collect::>(); + + let num_row_groups = builder.metadata().num_row_groups(); + let mut row_group_dictionaries = Vec::with_capacity(num_row_groups); + row_group_dictionaries.resize_with(num_row_groups, DictionaryStatistics::new); + + for idx in pruned_row_groups.row_group_indexes() { + let mut dict_stats = + DictionaryStatistics::with_capacity(parquet_columns.len()); + for (column_name, column_idx) in &parquet_columns { + let col_meta = builder.metadata().row_group(idx).column(*column_idx); + if !is_fully_dictionary_encoded(col_meta) { + continue; + } + let dict = match builder + .get_row_group_column_dictionary(idx, *column_idx) + .await + { + Ok(Some(dict)) => dict, + Ok(None) => continue, + Err(e) => { + log::debug!("Ignoring error reading dictionary: {e}"); + file_metrics.predicate_evaluation_errors.add(1); + continue; + } + }; + dict_stats.insert(column_name, &dict).unwrap(); + } + row_group_dictionaries[idx] = dict_stats; + } + pruned_row_groups.prune_by_dictionary( + pruning_predicate, + &file_metrics, + &row_group_dictionaries, + ); + + Ok(pruned_row_groups) + } + + #[tokio::test] + async fn test_row_group_dictionary_pruning_predicate_absent_value() { + let (test, data) = DictionaryFilterTest::new_single_value_row_groups(); + test.with_expect_all_pruned() + .run(data, col("s").eq(lit("delta"))) + .await + } + + #[tokio::test] + async fn test_row_group_dictionary_pruning_predicate_in_list_absent() { + let (test, data) = DictionaryFilterTest::new_single_value_row_groups(); + test.with_expect_all_pruned() + .run( + data, + col("s").in_list(vec![lit("delta"), lit("epsilon")], false), + ) + .await + } + + #[tokio::test] + async fn test_row_group_dictionary_pruning_predicate_present_value() { + // Each row group has exactly one distinct value, so `s = 'beta'` only + // survives in the row group whose sole value is "beta"; the other + // two row groups are provably disjoint from {"beta"}. + let (test, data) = DictionaryFilterTest::new_single_value_row_groups(); + test.with_expect_some_pruned(vec![1]) + .run(data, col("s").eq(lit("beta"))) + .await + } + + #[tokio::test] + async fn test_row_group_dictionary_pruning_predicate_not_eq_sole_value() { + // Each row group has exactly one distinct value, so `s != ` can never be true within that row group -- this exercises + // the `NOT IN`/`!=` direction that a bloom filter alone can't prove. + let (test, data) = DictionaryFilterTest::new_single_value_row_groups(); + test.with_expect_some_pruned(vec![1, 2]) + .run(data, col("s").not_eq(lit("alpha"))) + .await + } + + #[tokio::test] + async fn test_row_group_dictionary_pruning_predicate_not_in_list() { + let (test, data) = DictionaryFilterTest::new_single_value_row_groups(); + test.with_expect_some_pruned(vec![2]) + .run( + data, + col("s") + .not_eq(lit("alpha")) + .and(col("s").not_eq(lit("beta"))), + ) + .await + } + + #[tokio::test] + async fn test_row_group_dictionary_pruning_predicate_plain_fallback_not_pruned() { + // A large, high-cardinality dictionary forces PLAIN fallback, so the + // encoding-stats gate should treat the row group as not prunable by + // dictionary even though a query would otherwise be able to prune it + // via exact statistics. + let schema = Schema::new(vec![Field::new("s", DataType::Utf8, false)]); + let arrow_schema = Arc::new(schema.clone()); + let values: Vec = (0..5000).map(|i| format!("value_{i}")).collect(); + let array: ArrayRef = Arc::new(StringArray::from_iter_values( + values.iter().map(|v| v.as_str()), + )); + let batch = arrow::array::RecordBatch::try_new(arrow_schema.clone(), vec![array]) + .unwrap(); + + let props = WriterProperties::builder() + .set_dictionary_enabled(true) + .set_dictionary_page_size_limit(64) + .build(); + let mut out = BytesMut::new().writer(); + { + let mut writer = + ArrowWriter::try_new(&mut out, arrow_schema, Some(props)).unwrap(); + writer.write(&batch).unwrap(); + writer.finish().unwrap(); + } + let data = out.into_inner().freeze(); + + let test = DictionaryFilterTest { + schema, + post_pruning_row_groups: ExpectedPruning::None, + }; + // Query a value written late in the column: fallback happens almost + // immediately given the tiny size limit above, so this value is + // almost certainly PLAIN-encoded. A broken gate that used only the + // (small, abandoned) dictionary would incorrectly consider it absent + // and prune the row group, even though the value is actually present. + test.run(data, col("s").eq(lit("value_4999"))).await + } +} diff --git a/datafusion/datasource-parquet/src/metrics.rs b/datafusion/datasource-parquet/src/metrics.rs index cbdcb73196b17..b653a2d8f4977 100644 --- a/datafusion/datasource-parquet/src/metrics.rs +++ b/datafusion/datasource-parquet/src/metrics.rs @@ -49,6 +49,8 @@ pub struct ParquetFileMetrics { pub predicate_evaluation_errors: Count, /// Number of row groups pruned by bloom filters pub row_groups_pruned_bloom_filter: PruningMetrics, + /// Number of row groups pruned by exact Parquet dictionary-page values + pub row_groups_pruned_dictionary: PruningMetrics, /// Number of row groups pruned due to limit pruning. pub limit_pruned_row_groups: PruningMetrics, /// Number of row groups pruned by statistics @@ -123,6 +125,11 @@ impl ParquetFileMetrics { .with_type(MetricType::Summary) .pruning_metrics("row_groups_pruned_bloom_filter", partition); + let row_groups_pruned_dictionary = builder + .clone() + .with_type(MetricType::Summary) + .pruning_metrics("row_groups_pruned_dictionary", partition); + let limit_pruned_row_groups = builder .clone() .with_type(MetricType::Summary) @@ -215,6 +222,7 @@ impl ParquetFileMetrics { files_ranges_pruned_statistics, predicate_evaluation_errors, row_groups_pruned_bloom_filter, + row_groups_pruned_dictionary, row_groups_pruned_statistics, limit_pruned_row_groups, bytes_scanned, diff --git a/datafusion/datasource-parquet/src/mod.rs b/datafusion/datasource-parquet/src/mod.rs index 25b79a618830c..cfe8660d00cf8 100644 --- a/datafusion/datasource-parquet/src/mod.rs +++ b/datafusion/datasource-parquet/src/mod.rs @@ -27,6 +27,7 @@ pub mod access_plan; mod bloom_filter; mod decoder_projection; +mod dictionary_filter; pub mod file_format; pub mod metadata; mod metrics; @@ -49,6 +50,7 @@ mod writer; pub use access_plan::{ParquetAccessPlan, ParquetRowSelection, RowGroupAccess}; pub use bloom_filter::BloomFilterStatistics; +pub use dictionary_filter::{DictionaryStatistics, is_fully_dictionary_encoded}; pub use file_format::*; pub use metrics::ParquetFileMetrics; pub use page_filter::PagePruningAccessPlanFilter; diff --git a/datafusion/datasource-parquet/src/opener/mod.rs b/datafusion/datasource-parquet/src/opener/mod.rs index af97a192fa7ce..15622db02b154 100644 --- a/datafusion/datasource-parquet/src/opener/mod.rs +++ b/datafusion/datasource-parquet/src/opener/mod.rs @@ -25,6 +25,7 @@ use self::early_stop::EarlyStoppingStream; use self::encryption::EncryptionContext; use crate::access_plan::PreparedAccessPlan; use crate::decoder_projection::DecoderProjection; +use crate::dictionary_filter::is_fully_dictionary_encoded; use crate::page_filter::PagePruningAccessPlanFilter; use crate::push_decoder::{ DecoderBuilderConfig, PushDecoderStreamState, RgPlanEntry, RowGroupPruner, @@ -32,9 +33,9 @@ use crate::push_decoder::{ use crate::row_filter::RowFilterGenerator; use crate::row_group_filter::RowGroupAccessPlanFilter; use crate::{ - BloomFilterStatistics, Int96Coercer, ParquetAccessPlan, ParquetFileMetrics, - ParquetFileReaderFactory, ParquetRowSelection, ParquetVirtualColumn, - apply_file_schema_type_coercions, + BloomFilterStatistics, DictionaryStatistics, Int96Coercer, ParquetAccessPlan, + ParquetFileMetrics, ParquetFileReaderFactory, ParquetRowSelection, + ParquetVirtualColumn, apply_file_schema_type_coercions, }; use arrow::array::RecordBatch; use arrow::datatypes::DataType; @@ -269,6 +270,9 @@ pub(super) struct ParquetMorselizer { /// Should the bloom filter be read from parquet, if present, to skip row /// groups pub enable_bloom_filter: bool, + /// Should fully dictionary-encoded `BYTE_ARRAY` column chunks be used as + /// an exact row-group membership index, if present, to skip row groups + pub enable_dictionary_filter: bool, /// Should row group pruning be applied pub enable_row_group_stats_pruning: bool, /// Coerce INT96 timestamps to specific TimeUnit @@ -305,6 +309,7 @@ impl fmt::Debug for ParquetMorselizer { .field("preserve_order", &self.preserve_order) .field("enable_page_index", &self.enable_page_index) .field("enable_bloom_filter", &self.enable_bloom_filter) + .field("enable_dictionary_filter", &self.enable_dictionary_filter) .finish() } } @@ -350,6 +355,12 @@ impl Morselizer for ParquetMorselizer { /// PruneWithBloomFilters /// | /// v +/// LoadDictionaries +/// | +/// v +/// PruneWithDictionaries +/// | +/// v /// BuildStream /// | /// v @@ -384,6 +395,10 @@ enum ParquetOpenState { LoadBloomFilters(BoxFuture<'static, Result>), /// Pruning with preloaded Bloom Filters PruneWithBloomFilters(Box), + /// Loading Parquet dictionary pages required for row-group pruning + LoadDictionaries(BoxFuture<'static, Result>), + /// Pruning with preloaded dictionary pages + PruneWithDictionaries(Box), /// Builds the final reader stream /// /// TODO: split state as this currently does both I/O and CPU work. @@ -407,6 +422,8 @@ impl fmt::Debug for ParquetOpenState { ParquetOpenState::PruneWithStatistics(_) => "PruneWithStatistics", ParquetOpenState::LoadBloomFilters(_) => "LoadBloomFilters", ParquetOpenState::PruneWithBloomFilters(_) => "PruneWithBloomFilters", + ParquetOpenState::LoadDictionaries(_) => "LoadDictionaries", + ParquetOpenState::PruneWithDictionaries(_) => "PruneWithDictionaries", ParquetOpenState::BuildStream(_) => "BuildStream", ParquetOpenState::Ready(_) => "Ready", ParquetOpenState::Done => "Done", @@ -444,6 +461,7 @@ struct PreparedParquetOpen { force_filter_selections: bool, enable_page_index: bool, enable_bloom_filter: bool, + enable_dictionary_filter: bool, enable_row_group_stats_pruning: bool, limit: Option, coerce_int96: Option, @@ -496,6 +514,18 @@ struct BloomFiltersLoadedParquetOpen { row_group_bloom_filters: Vec, } +/// State of [`ParquetOpenState`] +/// +/// Result of loading dictionary pages needed for row-group pruning. +struct DictionariesLoadedParquetOpen { + prepared: RowGroupsPrunedParquetOpen, + /// Dictionary values loaded for each row group that remains under + /// consideration. + /// + /// indexed by parquet row-group index + row_group_dictionaries: Vec, +} + impl ParquetOpenState { /// Applies one CPU-only state transition. /// @@ -583,8 +613,17 @@ impl ParquetOpenState { ParquetOpenState::LoadBloomFilters(future) => { Ok(ParquetOpenState::LoadBloomFilters(future)) } - ParquetOpenState::PruneWithBloomFilters(loaded) => Ok( - ParquetOpenState::BuildStream(Box::new(loaded.prune_bloom_filters())), + ParquetOpenState::PruneWithBloomFilters(loaded) => { + let prepared_row_groups = loaded.prune_bloom_filters(); + Ok(ParquetOpenState::LoadDictionaries( + prepared_row_groups.load_dictionaries().boxed(), + )) + } + ParquetOpenState::LoadDictionaries(future) => { + Ok(ParquetOpenState::LoadDictionaries(future)) + } + ParquetOpenState::PruneWithDictionaries(loaded) => Ok( + ParquetOpenState::BuildStream(Box::new(loaded.prune_dictionaries())), ), ParquetOpenState::BuildStream(prepared) => { Ok(ParquetOpenState::Ready(prepared.build_stream()?)) @@ -705,6 +744,13 @@ impl MorselPlanner for ParquetMorselPlanner { ))) }))) } + ParquetOpenState::LoadDictionaries(future) => { + Ok(Some(Self::schedule_io(async move { + Ok(ParquetOpenState::PruneWithDictionaries(Box::new( + future.await?, + ))) + }))) + } ParquetOpenState::Ready(stream) => { let morsels: Vec> = vec![Box::new(ParquetStreamMorsel::new(stream))]; @@ -843,6 +889,7 @@ impl ParquetMorselizer { force_filter_selections: self.force_filter_selections, enable_page_index: self.enable_page_index, enable_bloom_filter: self.enable_bloom_filter, + enable_dictionary_filter: self.enable_dictionary_filter, enable_row_group_stats_pruning: self.enable_row_group_stats_pruning, limit: self.limit, coerce_int96: self.coerce_int96, @@ -1124,6 +1171,15 @@ impl FiltersPreparedParquetOpen { .row_groups_pruned_bloom_filter .add_matched(row_groups.remaining_row_group_count()); } + + if !prepared.enable_dictionary_filter || row_groups.is_empty() { + // Update metrics: dictionary filter unavailable, so all row + // groups are matched (not pruned) + prepared + .file_metrics + .row_groups_pruned_dictionary + .add_matched(row_groups.remaining_row_group_count()); + } } else { // Update metrics: no predicate, so all row groups are matched (not pruned) let remaining = row_groups.remaining_row_group_count(); @@ -1135,6 +1191,10 @@ impl FiltersPreparedParquetOpen { .file_metrics .row_groups_pruned_bloom_filter .add_matched(remaining); + prepared + .file_metrics + .row_groups_pruned_dictionary + .add_matched(remaining); } Ok(RowGroupsPrunedParquetOpen { @@ -1248,6 +1308,126 @@ impl RowGroupsPrunedParquetOpen { row_group_bloom_filters, }) } + + /// Load dictionary pages needed for pruning when enabled and a pruning + /// predicate exists. + /// + /// Only column chunks that are fully `BYTE_ARRAY` dictionary-encoded + /// (see [`is_fully_dictionary_encoded`]) are read: their dictionary is + /// the exact set of the row group's distinct values, so partially + /// dictionary-encoded chunks (writers fall back to `PLAIN` past a size + /// limit) are skipped rather than risk unsound pruning. + async fn load_dictionaries(mut self) -> Result { + let num_row_groups = self + .prepared + .loaded + .reader_metadata + .metadata() + .num_row_groups(); + let mut row_group_dictionaries = + vec![DictionaryStatistics::new(); num_row_groups]; + + if let Some(predicate) = + self.prepared.pruning_predicate.as_ref().map(|p| p.as_ref()) + && self.prepared.loaded.prepared.enable_dictionary_filter + && !self.row_groups.is_empty() + { + // Use the existing reader for dictionary I/O; + // replace with a fresh reader for decoding below. + let reader_metadata = self.prepared.loaded.reader_metadata.clone(); + let replacement_reader = { + let prepared = &self.prepared.loaded.prepared; + prepared.parquet_file_reader_factory.create_reader( + prepared.partition_index, + prepared.partitioned_file.clone(), + prepared.metadata_size_hint, + &prepared.metrics, + )? + }; + + let prepared = &mut self.prepared.loaded.prepared; + let mut builder = ParquetRecordBatchStreamBuilder::new_with_metadata( + mem::replace(&mut prepared.async_file_reader, replacement_reader), + reader_metadata, + ); + let parquet_columns: Vec<(String, usize)> = predicate + .literal_columns() + .into_iter() + .filter_map(|column_name| { + let parquet_schema = builder.parquet_schema(); + let (column_idx, _) = parquet_column( + parquet_schema, + &prepared.physical_file_schema, + &column_name, + )?; + Some((column_name, column_idx)) + }) + .collect(); + + let file_metadata = Arc::clone(builder.metadata()); + for idx in self.row_groups.row_group_indexes() { + let mut row_group_dictionary = + DictionaryStatistics::with_capacity(parquet_columns.len()); + for (column_name, column_idx) in &parquet_columns { + let col_meta = file_metadata.row_group(idx).column(*column_idx); + if !is_fully_dictionary_encoded(col_meta) { + continue; + } + let dictionary = match builder + .get_row_group_column_dictionary(idx, *column_idx) + .await + { + Ok(Some(dictionary)) => dictionary, + Ok(None) => continue, + Err(e) => { + debug!("Ignoring error reading dictionary page: {e}"); + prepared.file_metrics.predicate_evaluation_errors.add(1); + continue; + } + }; + if let Err(e) = row_group_dictionary.insert(column_name, &dictionary) + { + debug!("Ignoring error decoding dictionary page: {e}"); + prepared.file_metrics.predicate_evaluation_errors.add(1); + } + } + row_group_dictionaries[idx] = row_group_dictionary; + } + } + + Ok(DictionariesLoadedParquetOpen { + prepared: self, + row_group_dictionaries, + }) + } +} + +impl DictionariesLoadedParquetOpen { + /// Apply dictionary-based pruning using already loaded dictionary values. + fn prune_dictionaries(mut self) -> RowGroupsPrunedParquetOpen { + if let Some(predicate) = self + .prepared + .prepared + .pruning_predicate + .as_ref() + .map(|p| p.as_ref()) + && self + .prepared + .prepared + .loaded + .prepared + .enable_dictionary_filter + && !self.prepared.row_groups.is_empty() + { + self.prepared.row_groups.prune_by_dictionary( + predicate, + &self.prepared.prepared.loaded.prepared.file_metrics, + &self.row_group_dictionaries, + ); + } + + self.prepared + } } impl BloomFiltersLoadedParquetOpen { @@ -1749,6 +1929,7 @@ mod test { force_filter_selections: bool, enable_page_index: bool, enable_bloom_filter: bool, + enable_dictionary_filter: bool, enable_row_group_stats_pruning: bool, coerce_int96: Option, max_predicate_cache_size: Option, @@ -1857,6 +2038,7 @@ mod test { force_filter_selections: false, enable_page_index: false, enable_bloom_filter: false, + enable_dictionary_filter: false, enable_row_group_stats_pruning: false, coerce_int96: None, max_predicate_cache_size: None, @@ -2025,6 +2207,7 @@ mod test { force_filter_selections: self.force_filter_selections, enable_page_index: self.enable_page_index, enable_bloom_filter: self.enable_bloom_filter, + enable_dictionary_filter: self.enable_dictionary_filter, enable_row_group_stats_pruning: self.enable_row_group_stats_pruning, coerce_int96: self.coerce_int96, // End-to-end coercion behavior (including timezone) is diff --git a/datafusion/datasource-parquet/src/row_group_filter.rs b/datafusion/datasource-parquet/src/row_group_filter.rs index 2a2544b99b06c..2c02d9c5d48cd 100644 --- a/datafusion/datasource-parquet/src/row_group_filter.rs +++ b/datafusion/datasource-parquet/src/row_group_filter.rs @@ -20,6 +20,7 @@ use std::sync::Arc; use super::{ParquetAccessPlan, ParquetFileMetrics, RowGroupAccess}; use crate::bloom_filter::BloomFilterStatistics; +use crate::dictionary_filter::DictionaryStatistics; use arrow::array::{ArrayRef, BooleanArray, UInt64Array}; use arrow::datatypes::Schema; use datafusion_common::pruning::PruningStatistics; @@ -453,6 +454,48 @@ impl RowGroupAccessPlanFilter { } } } + + /// Prune remaining row groups using loaded Parquet dictionary pages and + /// the [`PruningPredicate`]. + /// + /// Updates this set with row groups that should not be scanned. + /// `row_group_dictionaries[idx]` contains the exact dictionary values for + /// the parquet row group at index `idx`. + /// + /// # Panics + /// if `row_group_dictionaries` does not have the same number of row groups as this set + pub fn prune_by_dictionary( + &mut self, + predicate: &PruningPredicate, + metrics: &ParquetFileMetrics, + row_group_dictionaries: &[DictionaryStatistics], + ) { + assert_eq!(row_group_dictionaries.len(), self.access_plan.len()); + for (idx, stats) in row_group_dictionaries.iter().enumerate() { + if !self.access_plan.should_scan(idx) { + continue; + } + + // Can this group be pruned? + let prune_group = match predicate.prune(stats) { + Ok(values) => !values[0], + Err(e) => { + log::debug!( + "Error evaluating row group predicate on dictionary: {e}" + ); + metrics.predicate_evaluation_errors.add(1); + false + } + }; + + if prune_group { + metrics.row_groups_pruned_dictionary.add_pruned(1); + self.access_plan.skip(idx) + } else { + metrics.row_groups_pruned_dictionary.add_matched(1); + } + } + } } /// Wraps a slice of [`RowGroupMetaData`] in a way that implements [`PruningStatistics`]. diff --git a/datafusion/datasource-parquet/src/source.rs b/datafusion/datasource-parquet/src/source.rs index 3443b08475e0d..71565b681e5e3 100644 --- a/datafusion/datasource-parquet/src/source.rs +++ b/datafusion/datasource-parquet/src/source.rs @@ -479,6 +479,23 @@ impl ParquetSource { self.table_parquet_options.global.bloom_filter_on_read } + /// If enabled, the reader will use fully dictionary-encoded `BYTE_ARRAY` + /// column chunks as an exact row-group membership index. Defaults to + /// false. + pub fn with_dictionary_filter_on_read( + mut self, + dictionary_filter_on_read: bool, + ) -> Self { + self.table_parquet_options.global.dictionary_filter_on_read = + dictionary_filter_on_read; + self + } + + /// Return the value described in [`Self::with_dictionary_filter_on_read`] + fn dictionary_filter_on_read(&self) -> bool { + self.table_parquet_options.global.dictionary_filter_on_read + } + /// Return the maximum predicate cache size, in bytes, used when /// `pushdown_filters` pub fn max_predicate_cache_size(&self) -> Option { @@ -638,6 +655,7 @@ impl FileSource for ParquetSource { force_filter_selections: self.force_filter_selections(), enable_page_index: self.enable_page_index(), enable_bloom_filter: self.bloom_filter_on_read(), + enable_dictionary_filter: self.dictionary_filter_on_read(), enable_row_group_stats_pruning: self.table_parquet_options.global.pruning, coerce_int96, coerce_int96_tz, diff --git a/datafusion/physical-expr-common/src/metrics/value.rs b/datafusion/physical-expr-common/src/metrics/value.rs index 232fefcc5f47e..cf6d0ff566681 100644 --- a/datafusion/physical-expr-common/src/metrics/value.rs +++ b/datafusion/physical-expr-common/src/metrics/value.rs @@ -1041,29 +1041,30 @@ impl MetricValue { "files_ranges_pruned_statistics" => 4, "row_groups_pruned_statistics" => 5, "row_groups_pruned_bloom_filter" => 6, - "page_index_pages_pruned" => 7, - "page_index_rows_pruned" => 8, - _ => 9, + "row_groups_pruned_dictionary" => 7, + "page_index_pages_pruned" => 8, + "page_index_rows_pruned" => 9, + _ => 10, }, - Self::SpillCount(_) => 10, - Self::SpilledBytes(_) => 11, - Self::SpilledRows(_) => 12, - Self::CurrentMemoryUsage(_) => 13, + Self::SpillCount(_) => 11, + Self::SpilledBytes(_) => 12, + Self::SpilledRows(_) => 13, + Self::CurrentMemoryUsage(_) => 14, Self::Count { name, .. } => match name.as_ref() { // This Parquet page-index metric is a plain Count because it // records pages that skipped page-index evaluation, not a // pruned/matched pair. Keep it grouped with the other // page-index pruning metrics in EXPLAIN output. - "page_index_pages_skipped_by_fully_matched" => 8, - _ => 14, + "page_index_pages_skipped_by_fully_matched" => 9, + _ => 15, }, - Self::PeakMemoryUsage { .. } => 13, - Self::Gauge { .. } => 15, - Self::Time { .. } => 16, - Self::Ratio { .. } => 17, - Self::StartTimestamp(_) => 18, // show timestamps last - Self::EndTimestamp(_) => 19, - Self::Custom { .. } => 20, + Self::PeakMemoryUsage { .. } => 14, + Self::Gauge { .. } => 16, + Self::Time { .. } => 17, + Self::Ratio { .. } => 18, + Self::StartTimestamp(_) => 19, // show timestamps last + Self::EndTimestamp(_) => 20, + Self::Custom { .. } => 21, } } diff --git a/datafusion/proto-common/proto/datafusion_common.proto b/datafusion/proto-common/proto/datafusion_common.proto index 7fff5b6b715ff..9e955bb7919b3 100644 --- a/datafusion/proto-common/proto/datafusion_common.proto +++ b/datafusion/proto-common/proto/datafusion_common.proto @@ -572,6 +572,7 @@ message ParquetOptions { bool schema_force_view_types = 28; // default = false bool binary_as_string = 29; // default = false bool skip_arrow_metadata = 30; // default = false + bool dictionary_filter_on_read = 38; // default = false oneof metadata_size_hint_opt { uint64 metadata_size_hint = 4; diff --git a/datafusion/proto-common/src/from_proto/mod.rs b/datafusion/proto-common/src/from_proto/mod.rs index 97cc9af230105..5e925a91f28b8 100644 --- a/datafusion/proto-common/src/from_proto/mod.rs +++ b/datafusion/proto-common/src/from_proto/mod.rs @@ -1102,6 +1102,7 @@ impl TryFrom<&protobuf::ParquetOptions> for ParquetOptions { }) .unwrap_or(None), bloom_filter_on_read: value.bloom_filter_on_read, + dictionary_filter_on_read: value.dictionary_filter_on_read, bloom_filter_on_write: value.bloom_filter_on_write, bloom_filter_fpp: value.clone() .bloom_filter_fpp_opt diff --git a/datafusion/proto-common/src/generated/pbjson.rs b/datafusion/proto-common/src/generated/pbjson.rs index 963faa5a3e9cb..c5ad6a7f705af 100644 --- a/datafusion/proto-common/src/generated/pbjson.rs +++ b/datafusion/proto-common/src/generated/pbjson.rs @@ -6400,6 +6400,9 @@ impl serde::Serialize for ParquetOptions { if self.skip_arrow_metadata { len += 1; } + if self.dictionary_filter_on_read { + len += 1; + } if self.dictionary_page_size_limit != 0 { len += 1; } @@ -6514,6 +6517,9 @@ impl serde::Serialize for ParquetOptions { if self.skip_arrow_metadata { struct_ser.serialize_field("skipArrowMetadata", &self.skip_arrow_metadata)?; } + if self.dictionary_filter_on_read { + struct_ser.serialize_field("dictionaryFilterOnRead", &self.dictionary_filter_on_read)?; + } if self.dictionary_page_size_limit != 0 { #[allow(clippy::needless_borrow)] #[allow(clippy::needless_borrows_for_generic_args)] @@ -6681,6 +6687,8 @@ impl<'de> serde::Deserialize<'de> for ParquetOptions { "binaryAsString", "skip_arrow_metadata", "skipArrowMetadata", + "dictionary_filter_on_read", + "dictionaryFilterOnRead", "dictionary_page_size_limit", "dictionaryPageSizeLimit", "data_page_row_count_limit", @@ -6736,6 +6744,7 @@ impl<'de> serde::Deserialize<'de> for ParquetOptions { SchemaForceViewTypes, BinaryAsString, SkipArrowMetadata, + DictionaryFilterOnRead, DictionaryPageSizeLimit, DataPageRowCountLimit, MaxRowGroupSize, @@ -6792,6 +6801,7 @@ impl<'de> serde::Deserialize<'de> for ParquetOptions { "schemaForceViewTypes" | "schema_force_view_types" => Ok(GeneratedField::SchemaForceViewTypes), "binaryAsString" | "binary_as_string" => Ok(GeneratedField::BinaryAsString), "skipArrowMetadata" | "skip_arrow_metadata" => Ok(GeneratedField::SkipArrowMetadata), + "dictionaryFilterOnRead" | "dictionary_filter_on_read" => Ok(GeneratedField::DictionaryFilterOnRead), "dictionaryPageSizeLimit" | "dictionary_page_size_limit" => Ok(GeneratedField::DictionaryPageSizeLimit), "dataPageRowCountLimit" | "data_page_row_count_limit" => Ok(GeneratedField::DataPageRowCountLimit), "maxRowGroupSize" | "max_row_group_size" => Ok(GeneratedField::MaxRowGroupSize), @@ -6846,6 +6856,7 @@ impl<'de> serde::Deserialize<'de> for ParquetOptions { let mut schema_force_view_types__ = None; let mut binary_as_string__ = None; let mut skip_arrow_metadata__ = None; + let mut dictionary_filter_on_read__ = None; let mut dictionary_page_size_limit__ = None; let mut data_page_row_count_limit__ = None; let mut max_row_group_size__ = None; @@ -6976,6 +6987,12 @@ impl<'de> serde::Deserialize<'de> for ParquetOptions { } skip_arrow_metadata__ = Some(map_.next_value()?); } + GeneratedField::DictionaryFilterOnRead => { + if dictionary_filter_on_read__.is_some() { + return Err(serde::de::Error::duplicate_field("dictionaryFilterOnRead")); + } + dictionary_filter_on_read__ = Some(map_.next_value()?); + } GeneratedField::DictionaryPageSizeLimit => { if dictionary_page_size_limit__.is_some() { return Err(serde::de::Error::duplicate_field("dictionaryPageSizeLimit")); @@ -7110,6 +7127,7 @@ impl<'de> serde::Deserialize<'de> for ParquetOptions { schema_force_view_types: schema_force_view_types__.unwrap_or_default(), binary_as_string: binary_as_string__.unwrap_or_default(), skip_arrow_metadata: skip_arrow_metadata__.unwrap_or_default(), + dictionary_filter_on_read: dictionary_filter_on_read__.unwrap_or_default(), dictionary_page_size_limit: dictionary_page_size_limit__.unwrap_or_default(), data_page_row_count_limit: data_page_row_count_limit__.unwrap_or_default(), max_row_group_size: max_row_group_size__.unwrap_or_default(), diff --git a/datafusion/proto-common/src/generated/prost.rs b/datafusion/proto-common/src/generated/prost.rs index 93b97c4f1376c..c4db891e8b7ac 100644 --- a/datafusion/proto-common/src/generated/prost.rs +++ b/datafusion/proto-common/src/generated/prost.rs @@ -856,6 +856,9 @@ pub struct ParquetOptions { /// default = false #[prost(bool, tag = "30")] pub skip_arrow_metadata: bool, + /// default = false + #[prost(bool, tag = "38")] + pub dictionary_filter_on_read: bool, #[prost(uint64, tag = "12")] pub dictionary_page_size_limit: u64, #[prost(uint64, tag = "18")] diff --git a/datafusion/proto-common/src/to_proto/mod.rs b/datafusion/proto-common/src/to_proto/mod.rs index d2e1ca50c812d..4b1fa5f4d4f11 100644 --- a/datafusion/proto-common/src/to_proto/mod.rs +++ b/datafusion/proto-common/src/to_proto/mod.rs @@ -926,6 +926,7 @@ impl TryFrom<&ParquetOptions> for protobuf::ParquetOptions { data_page_row_count_limit: value.data_page_row_count_limit as u64, encoding_opt: value.encoding.clone().map(protobuf::parquet_options::EncodingOpt::Encoding), bloom_filter_on_read: value.bloom_filter_on_read, + dictionary_filter_on_read: value.dictionary_filter_on_read, bloom_filter_on_write: value.bloom_filter_on_write, bloom_filter_fpp_opt: value.bloom_filter_fpp.map(protobuf::parquet_options::BloomFilterFppOpt::BloomFilterFpp), bloom_filter_ndv_opt: value.bloom_filter_ndv.map(protobuf::parquet_options::BloomFilterNdvOpt::BloomFilterNdv), diff --git a/datafusion/proto-models/src/generated/datafusion_proto_common.rs b/datafusion/proto-models/src/generated/datafusion_proto_common.rs index 93b97c4f1376c..c4db891e8b7ac 100644 --- a/datafusion/proto-models/src/generated/datafusion_proto_common.rs +++ b/datafusion/proto-models/src/generated/datafusion_proto_common.rs @@ -856,6 +856,9 @@ pub struct ParquetOptions { /// default = false #[prost(bool, tag = "30")] pub skip_arrow_metadata: bool, + /// default = false + #[prost(bool, tag = "38")] + pub dictionary_filter_on_read: bool, #[prost(uint64, tag = "12")] pub dictionary_page_size_limit: u64, #[prost(uint64, tag = "18")] diff --git a/datafusion/proto/src/logical_plan/file_formats.rs b/datafusion/proto/src/logical_plan/file_formats.rs index 8940b16bf83f5..48f71837e9dee 100644 --- a/datafusion/proto/src/logical_plan/file_formats.rs +++ b/datafusion/proto/src/logical_plan/file_formats.rs @@ -436,6 +436,7 @@ mod parquet { parquet_options::EncodingOpt::Encoding(encoding) }), bloom_filter_on_read: global_options.global.bloom_filter_on_read, + dictionary_filter_on_read: global_options.global.dictionary_filter_on_read, bloom_filter_on_write: global_options.global.bloom_filter_on_write, bloom_filter_fpp_opt: global_options.global.bloom_filter_fpp.map(|fpp| { parquet_options::BloomFilterFppOpt::BloomFilterFpp(fpp) @@ -590,6 +591,7 @@ mod parquet { } }), bloom_filter_on_read: proto.bloom_filter_on_read, + dictionary_filter_on_read: proto.dictionary_filter_on_read, bloom_filter_on_write: proto.bloom_filter_on_write, bloom_filter_fpp: proto .bloom_filter_fpp_opt diff --git a/datafusion/proto/tests/cases/roundtrip_logical_plan.rs b/datafusion/proto/tests/cases/roundtrip_logical_plan.rs index 74f7253386764..9f39749595f48 100644 --- a/datafusion/proto/tests/cases/roundtrip_logical_plan.rs +++ b/datafusion/proto/tests/cases/roundtrip_logical_plan.rs @@ -867,6 +867,7 @@ async fn roundtrip_logical_plan_copy_to_writer_options() -> Result<()> { let mut parquet_format = table_options.parquet; parquet_format.global.bloom_filter_on_read = true; + parquet_format.global.dictionary_filter_on_read = true; parquet_format.global.created_by = "DataFusion Test".to_string(); parquet_format.global.writer_version = DFParquetWriterVersion::V2_0; parquet_format.global.write_batch_size = 111; @@ -1256,6 +1257,7 @@ async fn roundtrip_default_codec_parquet() -> Result<()> { TableOptions::default_from_session_config(ctx.state().config_options()); let mut parquet_format = table_options.parquet; parquet_format.global.bloom_filter_on_read = true; + parquet_format.global.dictionary_filter_on_read = true; parquet_format.global.created_by = "DefaultCodecTest".to_string(); let file_type = format_as_file_type(Arc::new( @@ -1290,6 +1292,7 @@ async fn roundtrip_default_codec_parquet() -> Result<()> { .unwrap(); let decoded = pq.options.as_ref().unwrap(); assert!(decoded.global.bloom_filter_on_read); + assert!(decoded.global.dictionary_filter_on_read); assert_eq!("DefaultCodecTest", decoded.global.created_by); } _ => panic!("Expected CopyTo plan"), diff --git a/datafusion/sqllogictest/test_files/dynamic_filter_pushdown_config.slt b/datafusion/sqllogictest/test_files/dynamic_filter_pushdown_config.slt index c58047c4abe10..18e13d031dfb4 100644 --- a/datafusion/sqllogictest/test_files/dynamic_filter_pushdown_config.slt +++ b/datafusion/sqllogictest/test_files/dynamic_filter_pushdown_config.slt @@ -104,7 +104,7 @@ Plan with Metrics 03)----ProjectionExec: expr=[id@0 as id, value@1 as v, value@1 + id@0 as name], metrics=[output_rows=10, ] 04)------FilterExec: value@1 > 3, metrics=[output_rows=10, , selectivity=100% (10/10)] 05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, metrics=[output_rows=10, ] -06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/dynamic_filter_pushdown_config/test_data.parquet]]}, projection=[id, value], file_type=parquet, predicate=value@1 > 3 AND DynamicFilter [ value@1 IS NULL OR value@1 > 800 ], dynamic_rg_pruning=eligible, pruning_predicate=value_null_count@1 != row_count@2 AND value_max@0 > 3 AND (value_null_count@1 > 0 OR value_null_count@1 != row_count@2 AND value_max@0 > 800), required_guarantees=[], metrics=[output_rows=10, elapsed_compute=, output_bytes=80.0 B, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched -> 1 fully matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=210, page_index_load_skipped=1, row_groups_pruned_dynamic_filter=0, metadata_load_time=, scan_efficiency_ratio=18.31% (210/1.15 K)] +06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/dynamic_filter_pushdown_config/test_data.parquet]]}, projection=[id, value], file_type=parquet, predicate=value@1 > 3 AND DynamicFilter [ value@1 IS NULL OR value@1 > 800 ], dynamic_rg_pruning=eligible, pruning_predicate=value_null_count@1 != row_count@2 AND value_max@0 > 3 AND (value_null_count@1 > 0 OR value_null_count@1 != row_count@2 AND value_max@0 > 800), required_guarantees=[], metrics=[output_rows=10, elapsed_compute=, output_bytes=80.0 B, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched -> 1 fully matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=210, page_index_load_skipped=1, row_groups_pruned_dynamic_filter=0, metadata_load_time=, scan_efficiency_ratio=18.31% (210/1.15 K)] statement ok set datafusion.explain.analyze_level = dev; diff --git a/datafusion/sqllogictest/test_files/dynamic_row_group_pruning.slt b/datafusion/sqllogictest/test_files/dynamic_row_group_pruning.slt index 2149cacfc0a55..06221c7915502 100644 --- a/datafusion/sqllogictest/test_files/dynamic_row_group_pruning.slt +++ b/datafusion/sqllogictest/test_files/dynamic_row_group_pruning.slt @@ -98,7 +98,7 @@ explain analyze select v from t order by v desc limit 3; ---- Plan with Metrics 01)SortExec: TopK(fetch=3), expr=[v@0 DESC], preserve_partitioning=[false], filter=[v@0 IS NULL OR v@0 > 12], metrics=[output_rows=3, elapsed_compute=, output_bytes=] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/dynamic_row_group_pruning/data.parquet]]}, projection=[v], file_type=parquet, predicate=DynamicFilter [ v@0 IS NULL OR v@0 > 12 ], sort_order_for_reorder=[v@0 DESC], reverse_row_groups=true, dynamic_rg_pruning=eligible, pruning_predicate=v_null_count@0 > 0 OR v_null_count@0 != row_count@2 AND v_max@1 > 12, required_guarantees=[], metrics=[output_rows=3, elapsed_compute=, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=5 total → 5 matched, row_groups_pruned_bloom_filter=5 total → 5 matched, page_index_pages_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, row_groups_pruned_dynamic_filter=4, metadata_load_time=, scan_efficiency_ratio=] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/dynamic_row_group_pruning/data.parquet]]}, projection=[v], file_type=parquet, predicate=DynamicFilter [ v@0 IS NULL OR v@0 > 12 ], sort_order_for_reorder=[v@0 DESC], reverse_row_groups=true, dynamic_rg_pruning=eligible, pruning_predicate=v_null_count@0 > 0 OR v_null_count@0 != row_count@2 AND v_max@1 > 12, required_guarantees=[], metrics=[output_rows=3, elapsed_compute=, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=5 total → 5 matched, row_groups_pruned_bloom_filter=5 total → 5 matched, row_groups_pruned_dictionary=5 total → 5 matched, page_index_pages_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, row_groups_pruned_dynamic_filter=4, metadata_load_time=, scan_efficiency_ratio=] statement ok drop table t; diff --git a/datafusion/sqllogictest/test_files/explain_analyze.slt b/datafusion/sqllogictest/test_files/explain_analyze.slt index d64efe80ccae5..db5db835bb3fb 100644 --- a/datafusion/sqllogictest/test_files/explain_analyze.slt +++ b/datafusion/sqllogictest/test_files/explain_analyze.slt @@ -247,7 +247,7 @@ explain analyze select * from cat_tracking where species > 'M' AND s >= 50 order ---- Plan with Metrics 01)SortExec: TopK(fetch=3), expr=[species@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[species@0 < Nlpine Sheep], metrics=[output_rows=3] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, scan_efficiency_ratio=21.75% (485/2.23 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, scan_efficiency_ratio=21.75% (485/2.23 K)] statement ok reset datafusion.explain.analyze_categories; @@ -262,7 +262,7 @@ explain analyze select * from cat_tracking where species > 'M' AND s >= 50 order ---- Plan with Metrics 01)SortExec: TopK(fetch=3), expr=[species@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[species@0 < Nlpine Sheep], metrics=[output_rows=3, output_bytes=] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, scan_efficiency_ratio=] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, scan_efficiency_ratio=] statement ok reset datafusion.explain.analyze_categories; @@ -277,7 +277,7 @@ explain analyze select * from cat_tracking where species > 'M' AND s >= 50 order ---- Plan with Metrics 01)SortExec: TopK(fetch=3), expr=[species@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[species@0 < Nlpine Sheep], metrics=[output_rows=3, output_bytes=] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, scan_efficiency_ratio=] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, scan_efficiency_ratio=] statement ok reset datafusion.explain.analyze_categories; @@ -559,7 +559,7 @@ EXPLAIN (ANALYZE, METRICS 'rows', LEVEL summary) select * from cat_tracking wher ---- Plan with Metrics 01)SortExec: TopK(fetch=3), expr=[species@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[species@0 < Nlpine Sheep], metrics=[output_rows=3] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, scan_efficiency_ratio=21.75% (485/2.23 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, scan_efficiency_ratio=21.75% (485/2.23 K)] # ---- Quoted-string METRICS with multiple categories ---- @@ -568,7 +568,7 @@ EXPLAIN (ANALYZE, METRICS 'rows,bytes', LEVEL summary) select * from cat_trackin ---- Plan with Metrics 01)SortExec: TopK(fetch=3), expr=[species@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[species@0 < Nlpine Sheep], metrics=[output_rows=3, output_bytes=] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, scan_efficiency_ratio=] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, scan_efficiency_ratio=] # ---- (METRICS 'timing', LEVEL summary) — timing metrics only ---- @@ -588,7 +588,7 @@ EXPLAIN (ANALYZE, METRICS 'rows,bytes', TIMING off, LEVEL summary) select * from ---- Plan with Metrics 01)SortExec: TopK(fetch=3), expr=[species@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[species@0 < Nlpine Sheep], metrics=[output_rows=3, output_bytes=] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, scan_efficiency_ratio=] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, scan_efficiency_ratio=] # ---- TIMING sugar: `METRICS 'rows', TIMING on` ↔ rows + timing ---- @@ -597,7 +597,7 @@ EXPLAIN (ANALYZE, METRICS 'rows', TIMING on, LEVEL summary) select * from cat_tr ---- Plan with Metrics 01)SortExec: TopK(fetch=3), expr=[species@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[species@0 < Nlpine Sheep], metrics=[output_rows=3, elapsed_compute=] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, elapsed_compute=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, metadata_load_time=, scan_efficiency_ratio=21.75% (485/2.23 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, elapsed_compute=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, metadata_load_time=, scan_efficiency_ratio=21.75% (485/2.23 K)] # ---- SUMMARY sugar: `SUMMARY on` ↔ `LEVEL summary` ---- # Equivalent to METRICS 'rows', LEVEL summary above. @@ -607,7 +607,7 @@ EXPLAIN (ANALYZE, METRICS 'rows', SUMMARY on) select * from cat_tracking where s ---- Plan with Metrics 01)SortExec: TopK(fetch=3), expr=[species@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[species@0 < Nlpine Sheep], metrics=[output_rows=3] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, scan_efficiency_ratio=21.75% (485/2.23 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, scan_efficiency_ratio=21.75% (485/2.23 K)] # ---- Statement option overrides session config ---- # Session says 'timing' but statement-level `METRICS 'rows'` wins. @@ -620,7 +620,7 @@ EXPLAIN (ANALYZE, METRICS 'rows', LEVEL summary) select * from cat_tracking wher ---- Plan with Metrics 01)SortExec: TopK(fetch=3), expr=[species@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[species@0 < Nlpine Sheep], metrics=[output_rows=3] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, scan_efficiency_ratio=21.75% (485/2.23 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, scan_efficiency_ratio=21.75% (485/2.23 K)] # ---- pgjson format: structural golden with no metrics ---- @@ -682,7 +682,7 @@ EXPLAIN (ANALYZE, METRICS rows, LEVEL summary) select * from cat_tracking where ---- Plan with Metrics 01)SortExec: TopK(fetch=3), expr=[species@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[species@0 < Nlpine Sheep], metrics=[output_rows=3] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, scan_efficiency_ratio=21.75% (485/2.23 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/explain_analyze/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, scan_efficiency_ratio=21.75% (485/2.23 K)] statement ok reset datafusion.sql_parser.dialect; diff --git a/datafusion/sqllogictest/test_files/information_schema.slt b/datafusion/sqllogictest/test_files/information_schema.slt index 77acaa4747f9d..102efa51cbec5 100644 --- a/datafusion/sqllogictest/test_files/information_schema.slt +++ b/datafusion/sqllogictest/test_files/information_schema.slt @@ -249,6 +249,7 @@ datafusion.execution.parquet.created_by datafusion datafusion.execution.parquet.data_page_row_count_limit 20000 datafusion.execution.parquet.data_pagesize_limit 1048576 datafusion.execution.parquet.dictionary_enabled true +datafusion.execution.parquet.dictionary_filter_on_read false datafusion.execution.parquet.dictionary_page_size_limit 1048576 datafusion.execution.parquet.enable_page_index true datafusion.execution.parquet.encoding NULL @@ -408,6 +409,7 @@ datafusion.execution.parquet.created_by datafusion (writing) Sets "created by" p datafusion.execution.parquet.data_page_row_count_limit 20000 (writing) Sets best effort maximum number of rows in data page datafusion.execution.parquet.data_pagesize_limit 1048576 (writing) Sets best effort maximum size of data page in bytes datafusion.execution.parquet.dictionary_enabled true (writing) Sets if dictionary encoding is enabled. If NULL, uses default parquet writer setting +datafusion.execution.parquet.dictionary_filter_on_read false (reading) Use fully dictionary-encoded `BYTE_ARRAY` (Utf8/Binary) column chunks as an exact row-group membership index when reading parquet files. Unlike bloom filters, a fully dictionary-encoded column chunk's dictionary is the exact, complete set of the row group's distinct values, so this can prune both `IN`/`=` and `NOT IN`/`!=` predicates. Only column chunks whose page encoding statistics prove every data page came from the dictionary are used; chunks that fell back to `PLAIN` encoding are ignored. datafusion.execution.parquet.dictionary_page_size_limit 1048576 (writing) Sets best effort maximum dictionary page size, in bytes datafusion.execution.parquet.enable_page_index true (reading) If true, reads the Parquet data page level metadata (the Page Index), if present, to reduce the I/O and number of rows decoded. datafusion.execution.parquet.encoding NULL (writing) Sets default encoding for any column. Valid values are: plain, plain_dictionary, rle, bit_packed, delta_binary_packed, delta_length_byte_array, delta_byte_array, rle_dictionary, and byte_stream_split. These values are not case sensitive. If NULL, uses default parquet writer setting diff --git a/datafusion/sqllogictest/test_files/limit_pruning.slt b/datafusion/sqllogictest/test_files/limit_pruning.slt index 4ef0b5c74f3e7..ae1d64f88aae0 100644 --- a/datafusion/sqllogictest/test_files/limit_pruning.slt +++ b/datafusion/sqllogictest/test_files/limit_pruning.slt @@ -63,7 +63,7 @@ set datafusion.explain.analyze_level = summary; query TT explain analyze select * from tracking_data where species > 'M' AND s >= 50 limit 3; ---- -Plan with Metrics DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/limit_pruning/data.parquet]]}, projection=[species, s], limit=3, file_type=parquet, predicate=species@0 > M AND s@1 >= 50, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50, required_guarantees=[], metrics=[output_rows=3, elapsed_compute=, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, page_index_pages_pruned=0 total → 0 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=2 total → 0 matched, bytes_scanned=, metadata_load_time=, scan_efficiency_ratio= (159/2.23 K)] +Plan with Metrics DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/limit_pruning/data.parquet]]}, projection=[species, s], limit=3, file_type=parquet, predicate=species@0 > M AND s@1 >= 50, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50, required_guarantees=[], metrics=[output_rows=3, elapsed_compute=, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 3 matched, page_index_pages_pruned=0 total → 0 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=2 total → 0 matched, bytes_scanned=, metadata_load_time=, scan_efficiency_ratio= (159/2.23 K)] statement ok CREATE TABLE fully_matched_limit_source AS VALUES @@ -120,7 +120,7 @@ explain analyze select * from tracking_data where species > 'M' AND s >= 50 orde ---- Plan with Metrics 01)SortExec: TopK(fetch=3), expr=[species@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[species@0 < Nlpine Sheep], metrics=[output_rows=3, elapsed_compute=, output_bytes=] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/limit_pruning/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, elapsed_compute=, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, row_groups_pruned_dynamic_filter=0, metadata_load_time=, scan_efficiency_ratio= (/)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/limit_pruning/data.parquet]]}, projection=[species, s], file_type=parquet, predicate=species@0 > M AND s@1 >= 50 AND DynamicFilter [ species@0 < Nlpine Sheep ], sort_order_for_reorder=[species@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=species_null_count@1 != row_count@2 AND species_max@0 > M AND s_null_count@4 != row_count@2 AND s_max@3 >= 50 AND species_null_count@1 != row_count@2 AND species_min@5 < Nlpine Sheep, required_guarantees=[], metrics=[output_rows=3, elapsed_compute=, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=4 total → 3 matched -> 1 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 3 matched, page_index_pages_pruned=2 total → 2 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, row_groups_pruned_dynamic_filter=0, metadata_load_time=, scan_efficiency_ratio= (/)] statement ok drop table tracking_data; diff --git a/datafusion/sqllogictest/test_files/parquet_dictionary_pruning.slt b/datafusion/sqllogictest/test_files/parquet_dictionary_pruning.slt new file mode 100644 index 0000000000000..214f270bb12cc --- /dev/null +++ b/datafusion/sqllogictest/test_files/parquet_dictionary_pruning.slt @@ -0,0 +1,258 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at + +# http://www.apache.org/licenses/LICENSE-2.0 + +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +# End-to-end SLT for **row-group pruning by Parquet dictionaries** +# (`datafusion.execution.parquet.dictionary_filter_on_read`). +# +# Some stores use a low-cardinality string column's Parquet dictionary as a +# makeshift row-group index: when the whole column chunk is dictionary +# encoded, the dictionary page is the exact, complete set of that row +# group's distinct values, so a row group can be skipped without reading any +# data pages if a required value isn't in its dictionary. Unlike bloom +# filters, this is exact in both directions, so it can also prune `!=`/`NOT +# IN` when a row group's dictionary is a subset of the excluded values. +# +# The dataset interleaves two values per row group, chosen so every row +# group's [min, max] range spans (almost) the same wide interval -- min/max +# statistics alone can't prune anything below, only the exact per-row-group +# dictionary can: +# RG 0: "a", "z", "a", "z" -> dictionary {a, z}, min='a' max='z' +# RG 1: "b", "y", "b", "y" -> dictionary {b, y}, min='b' max='y' +# RG 2: "c", "x", "c", "x" -> dictionary {c, x}, min='c' max='x' +# This is the same trick as the "interleave across row groups" idea in +# `datafusion/datasource-parquet/benches/parquet_dictionary_pruning.rs`, +# shrunk to 3 row groups so the golden `EXPLAIN ANALYZE` output stays +# readable. A single-distinct-value-per-row-group layout (e.g. all "alpha" +# in RG 0) would let min/max statistics alone prune everything before the +# dictionary stage ever runs, which defeats the point of this file. +# +# `dictionary_filter_on_read` (like other `format.*`/reader options) is +# captured into the table's options at `CREATE EXTERNAL TABLE` time, not +# re-read from the session config on every query. So this file creates +# table `t` *twice* against the same underlying file: once before the `SET` +# below (to show the off-by-default plan) and once after (to show the +# flag actually taking effect) -- toggling the `SET` against a single, +# already-created `t` would silently have no effect on either plan. + +statement ok +set datafusion.explain.analyze_level = summary; + +# A parallel write can interleave/reshuffle rows across row groups, so pin +# target_partitions to 1 for the COPY below to keep row-group boundaries +# (and therefore the RG 0/1/2 layout described above) deterministic. +statement ok +set datafusion.execution.target_partitions = 1; + +statement ok +CREATE TABLE source_data AS VALUES + (0, 'a'), (1, 'z'), (2, 'a'), (3, 'z'), + (4, 'b'), (5, 'y'), (6, 'b'), (7, 'y'), + (8, 'c'), (9, 'x'), (10, 'c'), (11, 'x'); + +statement ok +COPY (SELECT column2 as s FROM source_data ORDER BY column1) +TO 'test_files/scratch/parquet_dictionary_pruning/data.parquet' +STORED AS PARQUET +OPTIONS ( + 'format.max_row_group_size' '4' +); + +statement ok +drop table source_data; + +statement ok +set datafusion.execution.target_partitions = 4; + +######## +# Off by default: no dictionary pruning happens without opting in. `t` is +# created here, before the `SET` further down, so its options capture +# dictionary_filter_on_read=false. +######## + +statement ok +CREATE EXTERNAL TABLE t +STORED AS PARQUET +LOCATION 'test_files/scratch/parquet_dictionary_pruning/data.parquet'; + +# Sanity: query returns the right rows regardless of the config below. +query T rowsort +SELECT s FROM t; +---- +a +a +b +b +c +c +x +x +y +y +z +z + +query error DataFusion error: Error during planning: SHOW \[VARIABLE\] is not supported unless information_schema is enabled +SHOW datafusion.execution.parquet.dictionary_filter_on_read + +query TT +explain analyze select s from t where s = 'm'; +---- +Plan with Metrics +01)FilterExec: s@0 = m, metrics=[output_rows=0, elapsed_compute=, output_bytes=, selectivity=0% (0/12)] +02)--RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, metrics=[output_rows=12, elapsed_compute=, output_bytes=] +03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_dictionary_pruning/data.parquet]]}, projection=[s], file_type=parquet, predicate=s@0 = m, pruning_predicate=s_null_count@2 != row_count@3 AND s_min@0 <= m AND m <= s_max@1, required_guarantees=[s in (m)], metrics=[output_rows=12, elapsed_compute=, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=3 total → 3 matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 3 matched, page_index_pages_pruned=3 total → 3 matched, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, row_groups_pruned_dynamic_filter=0, metadata_load_time=, scan_efficiency_ratio=] + +statement ok +drop table t; + +statement ok +set datafusion.execution.parquet.dictionary_filter_on_read = true; + +######## +# `IN`/`=` direction: a value absent from every row group's dictionary +# prunes all of them without reading any data pages. Unlike the flag-off +# plan just above, min/max statistics alone can't prune any row group here +# (every group's [min, max] contains 'm') -- only the dictionary stage can, +# and does: row_groups_pruned_dictionary goes from "3 total -> 3 matched" to +# "3 total -> 0 matched". `t` is (re)created here, after the `SET`, so its +# options now capture dictionary_filter_on_read=true. +######## + +statement ok +CREATE EXTERNAL TABLE t +STORED AS PARQUET +LOCATION 'test_files/scratch/parquet_dictionary_pruning/data.parquet'; + +query TT +explain analyze select s from t where s = 'm'; +---- +Plan with Metrics +01)FilterExec: s@0 = m, metrics=[output_rows=0, elapsed_compute=, output_bytes=, selectivity=N/A (0/0)] +02)--RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, metrics=[output_rows=0, elapsed_compute=, output_bytes=] +03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_dictionary_pruning/data.parquet]]}, projection=[s], file_type=parquet, predicate=s@0 = m, pruning_predicate=s_null_count@2 != row_count@3 AND s_min@0 <= m AND m <= s_max@1, required_guarantees=[s in (m)], metrics=[output_rows=0, elapsed_compute=, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=3 total → 3 matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 0 matched, page_index_pages_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, row_groups_pruned_dynamic_filter=0, metadata_load_time=, scan_efficiency_ratio=] + +query TT +explain analyze select s from t where s IN ('m', 'n'); +---- +Plan with Metrics +01)FilterExec: s@0 = m OR s@0 = n, metrics=[output_rows=0, elapsed_compute=, output_bytes=, selectivity=N/A (0/0)] +02)--RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, metrics=[output_rows=0, elapsed_compute=, output_bytes=] +03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_dictionary_pruning/data.parquet]]}, projection=[s], file_type=parquet, predicate=s@0 = m OR s@0 = n, pruning_predicate=s_null_count@2 != row_count@3 AND s_min@0 <= m AND m <= s_max@1 OR s_null_count@2 != row_count@3 AND s_min@0 <= n AND n <= s_max@1, required_guarantees=[s in (m, n)], metrics=[output_rows=0, elapsed_compute=, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=3 total → 3 matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 0 matched, page_index_pages_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, row_groups_pruned_dynamic_filter=0, metadata_load_time=, scan_efficiency_ratio=] + +# A value that IS present in one row group's dictionary is not pruned: 'a' +# is only in RG 0's dictionary {a, z}, so RG 1 and RG 2 (dictionaries {b, y} +# and {c, x}) get pruned but RG 0 survives. +query T +select s from t where s = 'a'; +---- +a +a + +######## +# `!=`/`NOT IN` direction: exact dictionaries can prune this too, unlike +# bloom filters. Statistics cannot prune this at all -- every row group has +# min != max, so the `(min != v OR v != max)` check statistics use is true +# for both excluded literals in every row group. RG 0's dictionary {a, z} +# is a subset of the excluded set {a, z}, so it's the only one the +# dictionary stage can prune. +######## + +query T rowsort +select s from t where s NOT IN ('a', 'z'); +---- +b +b +c +c +x +x +y +y + +query TT +explain analyze select s from t where s NOT IN ('a', 'z'); +---- +Plan with Metrics +01)FilterExec: s@0 != a AND s@0 != z, metrics=[output_rows=8, elapsed_compute=, output_bytes=, selectivity=100% (8/8)] +02)--RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, metrics=[output_rows=8, elapsed_compute=, output_bytes=] +03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_dictionary_pruning/data.parquet]]}, projection=[s], file_type=parquet, predicate=s@0 != a AND s@0 != z, pruning_predicate=s_null_count@2 != row_count@3 AND (s_min@0 != a OR a != s_max@1) AND s_null_count@2 != row_count@3 AND (s_min@0 != z OR z != s_max@1), required_guarantees=[s not in (a, z)], metrics=[output_rows=8, elapsed_compute=, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=3 total → 3 matched -> 2 fully matched, row_groups_pruned_bloom_filter=3 total → 3 matched, row_groups_pruned_dictionary=3 total → 2 matched, page_index_pages_pruned=0 total → 0 matched, page_index_pages_skipped_by_fully_matched=2, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, row_groups_pruned_dynamic_filter=0, metadata_load_time=, scan_efficiency_ratio=] + +######## +# Encoding-stats gate: a column with a large, high-cardinality dictionary +# falls back to `PLAIN` for some data pages. Since not every value is +# guaranteed to be drawn from the dictionary, it must not be used for +# pruning even though a value may look absent. +######## + +statement ok +CREATE TABLE wide_source_data AS +SELECT 'value_' || i AS s FROM generate_series(0, 4999) t(i); + +statement ok +COPY wide_source_data +TO 'test_files/scratch/parquet_dictionary_pruning/wide.parquet' +STORED AS PARQUET +OPTIONS ( + 'format.dictionary_page_size_limit' '64' +); + +statement ok +drop table wide_source_data; + +statement ok +CREATE EXTERNAL TABLE wide +STORED AS PARQUET +LOCATION 'test_files/scratch/parquet_dictionary_pruning/wide.parquet'; + +# `value_4999` is written late enough that the tiny dictionary size limit +# above has already forced a `PLAIN` fallback for it; it must still be +# found (not incorrectly pruned). +query T +select s from wide where s = 'value_4999'; +---- +value_4999 + +query TT +explain analyze select s from wide where s = 'value_4999'; +---- +Plan with Metrics +01)FilterExec: s@0 = value_4999, metrics=[output_rows=1, elapsed_compute=, output_bytes=, selectivity=0.02% (1/5.00 K)] +02)--RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, metrics=[output_rows=5.00 K, elapsed_compute=, output_bytes=] +03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_dictionary_pruning/wide.parquet]]}, projection=[s], file_type=parquet, predicate=s@0 = value_4999, pruning_predicate=s_null_count@2 != row_count@3 AND s_min@0 <= value_4999 AND value_4999 <= s_max@1, required_guarantees=[s in (value_4999)], metrics=[output_rows=5.00 K, elapsed_compute=, output_bytes=, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=2 total → 2 matched, limit_pruned_row_groups=0 total → 0 matched, bytes_scanned=, row_groups_pruned_dynamic_filter=0, metadata_load_time=, scan_efficiency_ratio=] + +statement ok +drop table wide; + +######## +# Clean up after the test +######## + +statement ok +drop table t; + +statement ok +RESET datafusion.execution.parquet.dictionary_filter_on_read; + +# Not RESET: that would restore the compiled-in default (available +# parallelism), not the sqllogictest harness's baseline of 4 +# (datafusion/sqllogictest/src/test_context.rs), which every other test +# file in this suite assumes is in effect. +statement ok +set datafusion.execution.target_partitions = 4; + +statement ok +RESET datafusion.explain.analyze_level; diff --git a/datafusion/sqllogictest/test_files/push_down_filter_parquet.slt b/datafusion/sqllogictest/test_files/push_down_filter_parquet.slt index e879947e324bb..3072ac21aafa7 100644 --- a/datafusion/sqllogictest/test_files/push_down_filter_parquet.slt +++ b/datafusion/sqllogictest/test_files/push_down_filter_parquet.slt @@ -206,7 +206,7 @@ EXPLAIN ANALYZE SELECT t FROM topk_pushdown ORDER BY t * t LIMIT 10; ---- Plan with Metrics 01)SortExec: TopK(fetch=10), expr=[t@0 * t@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[t@0 * t@0 < 1884329474306198481], metrics=[output_rows=10, output_batches=1, row_replacements=10] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_pushdown.parquet]]}, projection=[t], output_ordering=[t@0 ASC NULLS LAST], file_type=parquet, predicate=DynamicFilter [ t@0 * t@0 < 1884329474306198481 ], dynamic_rg_pruning=eligible, metrics=[output_rows=128, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=782 total → 782 matched, row_groups_pruned_bloom_filter=782 total → 782 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=128, pushdown_rows_pruned=99.87 K, predicate_cache_inner_records=128, predicate_cache_records=128, scan_efficiency_ratio=64.87% (258.7 K/398.8 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_pushdown.parquet]]}, projection=[t], output_ordering=[t@0 ASC NULLS LAST], file_type=parquet, predicate=DynamicFilter [ t@0 * t@0 < 1884329474306198481 ], dynamic_rg_pruning=eligible, metrics=[output_rows=128, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=782 total → 782 matched, row_groups_pruned_bloom_filter=782 total → 782 matched, row_groups_pruned_dictionary=782 total → 782 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=128, pushdown_rows_pruned=99.87 K, predicate_cache_inner_records=128, predicate_cache_records=128, scan_efficiency_ratio=64.87% (258.7 K/398.8 K)] statement ok reset datafusion.explain.analyze_categories; @@ -268,7 +268,7 @@ EXPLAIN ANALYZE SELECT * FROM topk_single_col ORDER BY b DESC LIMIT 1; ---- Plan with Metrics 01)SortExec: TopK(fetch=1), expr=[b@1 DESC], preserve_partitioning=[false], filter=[b@1 IS NULL OR b@1 > bd], metrics=[output_rows=1, output_batches=1, row_replacements=1] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_single_col.parquet]]}, projection=[a, b, c], file_type=parquet, predicate=DynamicFilter [ b@1 IS NULL OR b@1 > bd ], sort_order_for_reorder=[b@1 DESC], reverse_row_groups=true, dynamic_rg_pruning=eligible, pruning_predicate=b_null_count@0 > 0 OR b_null_count@0 != row_count@2 AND b_max@1 > bd, required_guarantees=[], metrics=[output_rows=4, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=4, pushdown_rows_pruned=0, predicate_cache_inner_records=4, predicate_cache_records=4, scan_efficiency_ratio=21.62% (222/1.03 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_single_col.parquet]]}, projection=[a, b, c], file_type=parquet, predicate=DynamicFilter [ b@1 IS NULL OR b@1 > bd ], sort_order_for_reorder=[b@1 DESC], reverse_row_groups=true, dynamic_rg_pruning=eligible, pruning_predicate=b_null_count@0 > 0 OR b_null_count@0 != row_count@2 AND b_max@1 > bd, required_guarantees=[], metrics=[output_rows=4, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=4, pushdown_rows_pruned=0, predicate_cache_inner_records=4, predicate_cache_records=4, scan_efficiency_ratio=21.62% (222/1.03 K)] statement ok reset datafusion.explain.analyze_categories; @@ -319,7 +319,7 @@ EXPLAIN ANALYZE SELECT * FROM topk_multi_col ORDER BY b ASC NULLS LAST, a DESC L ---- Plan with Metrics 01)SortExec: TopK(fetch=2), expr=[b@1 ASC NULLS LAST, a@0 DESC], preserve_partitioning=[false], filter=[b@1 < bb OR b@1 = bb AND (a@0 IS NULL OR a@0 > ac)], metrics=[output_rows=2, output_batches=1, row_replacements=2] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_multi_col.parquet]]}, projection=[a, b, c], file_type=parquet, predicate=DynamicFilter [ b@1 < bb OR b@1 = bb AND (a@0 IS NULL OR a@0 > ac) ], sort_order_for_reorder=[b@1 ASC NULLS LAST, a@0 DESC], dynamic_rg_pruning=eligible, pruning_predicate=b_null_count@1 != row_count@2 AND b_min@0 < bb OR b_null_count@1 != row_count@2 AND b_min@0 <= bb AND bb <= b_max@3 AND (a_null_count@4 > 0 OR a_null_count@4 != row_count@2 AND a_max@5 > ac), required_guarantees=[], metrics=[output_rows=4, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=4, pushdown_rows_pruned=0, predicate_cache_inner_records=8, predicate_cache_records=8, scan_efficiency_ratio=21.62% (222/1.03 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_multi_col.parquet]]}, projection=[a, b, c], file_type=parquet, predicate=DynamicFilter [ b@1 < bb OR b@1 = bb AND (a@0 IS NULL OR a@0 > ac) ], sort_order_for_reorder=[b@1 ASC NULLS LAST, a@0 DESC], dynamic_rg_pruning=eligible, pruning_predicate=b_null_count@1 != row_count@2 AND b_min@0 < bb OR b_null_count@1 != row_count@2 AND b_min@0 <= bb AND bb <= b_max@3 AND (a_null_count@4 > 0 OR a_null_count@4 != row_count@2 AND a_max@5 > ac), required_guarantees=[], metrics=[output_rows=4, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=4, pushdown_rows_pruned=0, predicate_cache_inner_records=8, predicate_cache_records=8, scan_efficiency_ratio=21.62% (222/1.03 K)] statement ok reset datafusion.explain.analyze_categories; @@ -388,8 +388,8 @@ FROM join_probe p INNER JOIN join_build AS build ---- Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(a@0, a@0), (b@1, b@1)], projection=[a@3, b@4, c@2, e@5], metrics=[output_rows=2, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=1, input_rows=2, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/join_build.parquet]]}, projection=[a, b, c], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=19.58% (196/1.00 K)] -03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/join_probe.parquet]]}, projection=[a, b, e], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= ba AND b@1 <= bb AND struct(a@0, b@1) IN (SET) ([{c0:aa,c1:ba}, {c0:ab,c1:bb}]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= ba AND b_null_count@5 != row_count@2 AND b_min@6 <= bb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=22.05% (228/1.03 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/join_build.parquet]]}, projection=[a, b, c], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=19.58% (196/1.00 K)] +03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/join_probe.parquet]]}, projection=[a, b, e], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= ba AND b@1 <= bb AND struct(a@0, b@1) IN (SET) ([{c0:aa,c1:ba}, {c0:ab,c1:bb}]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= ba AND b_null_count@5 != row_count@2 AND b_min@6 <= bb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=22.05% (228/1.03 K)] statement ok reset datafusion.explain.analyze_categories; @@ -474,9 +474,9 @@ INNER JOIN nested_t3 ON nested_t2.c = nested_t3.d; Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(c@3, d@0)], metrics=[output_rows=2, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=1, input_rows=2, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] 02)--HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(a@0, b@0)], metrics=[output_rows=2, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=1, input_rows=2, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nested_t1.parquet]]}, projection=[a, x], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=17.37% (132/760)] -04)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nested_t2.parquet]]}, projection=[b, c, y], file_type=parquet, predicate=DynamicFilter [ b@0 >= aa AND b@0 <= ab AND b@0 IN (SET) ([aa, ab]) ], dynamic_rg_pruning=eligible, pruning_predicate=b_null_count@1 != row_count@2 AND b_max@0 >= aa AND b_null_count@1 != row_count@2 AND b_min@3 <= ab AND (b_null_count@1 != row_count@2 AND b_min@3 <= aa AND aa <= b_max@0 OR b_null_count@1 != row_count@2 AND b_min@3 <= ab AND ab <= b_max@0), required_guarantees=[b in (aa, ab)], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=5 total → 5 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=3, predicate_cache_inner_records=5, predicate_cache_records=2, scan_efficiency_ratio=22.46% (234/1.04 K)] -05)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nested_t3.parquet]]}, projection=[d, z], file_type=parquet, predicate=DynamicFilter [ d@0 >= ca AND d@0 <= cb AND hash_lookup ], dynamic_rg_pruning=eligible, pruning_predicate=d_null_count@1 != row_count@2 AND d_max@0 >= ca AND d_null_count@1 != row_count@2 AND d_min@3 <= cb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=8 total → 8 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=6, predicate_cache_inner_records=8, predicate_cache_records=2, scan_efficiency_ratio=21.45% (172/802)] +03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nested_t1.parquet]]}, projection=[a, x], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=17.37% (132/760)] +04)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nested_t2.parquet]]}, projection=[b, c, y], file_type=parquet, predicate=DynamicFilter [ b@0 >= aa AND b@0 <= ab AND b@0 IN (SET) ([aa, ab]) ], dynamic_rg_pruning=eligible, pruning_predicate=b_null_count@1 != row_count@2 AND b_max@0 >= aa AND b_null_count@1 != row_count@2 AND b_min@3 <= ab AND (b_null_count@1 != row_count@2 AND b_min@3 <= aa AND aa <= b_max@0 OR b_null_count@1 != row_count@2 AND b_min@3 <= ab AND ab <= b_max@0), required_guarantees=[b in (aa, ab)], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=5 total → 5 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=3, predicate_cache_inner_records=5, predicate_cache_records=2, scan_efficiency_ratio=22.46% (234/1.04 K)] +05)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nested_t3.parquet]]}, projection=[d, z], file_type=parquet, predicate=DynamicFilter [ d@0 >= ca AND d@0 <= cb AND hash_lookup ], dynamic_rg_pruning=eligible, pruning_predicate=d_null_count@1 != row_count@2 AND d_max@0 >= ca AND d_null_count@1 != row_count@2 AND d_min@3 <= cb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=8 total → 8 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=6, predicate_cache_inner_records=8, predicate_cache_records=2, scan_efficiency_ratio=21.45% (172/802)] statement ok reset datafusion.explain.analyze_categories; @@ -605,8 +605,8 @@ LIMIT 2; Plan with Metrics 01)SortExec: TopK(fetch=2), expr=[e@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[e@0 < bb], metrics=[output_rows=2, output_batches=1, row_replacements=2] 02)--HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(a@0, d@0)], projection=[e@2], metrics=[output_rows=2, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=1, input_rows=2, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_join_build.parquet]]}, projection=[a], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=6.39% (64/1.00 K)] -04)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_join_probe.parquet]]}, projection=[d, e], file_type=parquet, predicate=DynamicFilter [ d@0 >= aa AND d@0 <= ab AND d@0 IN (SET) ([aa, ab]) ] AND DynamicFilter [ e@1 < bb ], dynamic_rg_pruning=eligible, pruning_predicate=d_null_count@1 != row_count@2 AND d_max@0 >= aa AND d_null_count@1 != row_count@2 AND d_min@3 <= ab AND (d_null_count@1 != row_count@2 AND d_min@3 <= aa AND aa <= d_max@0 OR d_null_count@1 != row_count@2 AND d_min@3 <= ab AND ab <= d_max@0) AND e_null_count@5 != row_count@2 AND e_min@4 < bb, required_guarantees=[d in (aa, ab)], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=4 total → 4 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=14.89% (154/1.03 K)] +03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_join_build.parquet]]}, projection=[a], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=6.39% (64/1.00 K)] +04)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_join_probe.parquet]]}, projection=[d, e], file_type=parquet, predicate=DynamicFilter [ d@0 >= aa AND d@0 <= ab AND d@0 IN (SET) ([aa, ab]) ] AND DynamicFilter [ e@1 < bb ], dynamic_rg_pruning=eligible, pruning_predicate=d_null_count@1 != row_count@2 AND d_max@0 >= aa AND d_null_count@1 != row_count@2 AND d_min@3 <= ab AND (d_null_count@1 != row_count@2 AND d_min@3 <= aa AND aa <= d_max@0 OR d_null_count@1 != row_count@2 AND d_min@3 <= ab AND ab <= d_max@0) AND e_null_count@5 != row_count@2 AND e_min@4 < bb, required_guarantees=[d in (aa, ab)], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=4 total → 4 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=14.89% (154/1.03 K)] statement ok reset datafusion.explain.analyze_categories; @@ -656,7 +656,7 @@ EXPLAIN ANALYZE SELECT b, a FROM topk_proj ORDER BY a LIMIT 2; Plan with Metrics 01)ProjectionExec: expr=[b@1 as b, a@0 as a], metrics=[output_rows=2, output_batches=1] 02)--SortExec: TopK(fetch=2), expr=[a@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[a@0 < 2], metrics=[output_rows=2, output_batches=1, row_replacements=2] -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_proj.parquet]]}, projection=[a, b], file_type=parquet, predicate=DynamicFilter [ a@0 < 2 ], sort_order_for_reorder=[a@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_min@0 < 2, required_guarantees=[], metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=3, pushdown_rows_pruned=0, predicate_cache_inner_records=3, predicate_cache_records=3, scan_efficiency_ratio=13.21% (141/1.07 K)] +03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_proj.parquet]]}, projection=[a, b], file_type=parquet, predicate=DynamicFilter [ a@0 < 2 ], sort_order_for_reorder=[a@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_min@0 < 2, required_guarantees=[], metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=3, pushdown_rows_pruned=0, predicate_cache_inner_records=3, predicate_cache_records=3, scan_efficiency_ratio=13.21% (141/1.07 K)] # Case 2: prune — `SELECT a` — filter stays as `a < 2` on the scan. query TT @@ -664,7 +664,7 @@ EXPLAIN ANALYZE SELECT a FROM topk_proj ORDER BY a LIMIT 2; ---- Plan with Metrics 01)SortExec: TopK(fetch=2), expr=[a@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[a@0 < 2], metrics=[output_rows=2, output_batches=1, row_replacements=2] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_proj.parquet]]}, projection=[a], file_type=parquet, predicate=DynamicFilter [ a@0 < 2 ], sort_order_for_reorder=[a@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_min@0 < 2, required_guarantees=[], metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=3, pushdown_rows_pruned=0, predicate_cache_inner_records=3, predicate_cache_records=3, scan_efficiency_ratio=6.84% (73/1.07 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_proj.parquet]]}, projection=[a], file_type=parquet, predicate=DynamicFilter [ a@0 < 2 ], sort_order_for_reorder=[a@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_min@0 < 2, required_guarantees=[], metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=3, pushdown_rows_pruned=0, predicate_cache_inner_records=3, predicate_cache_records=3, scan_efficiency_ratio=6.84% (73/1.07 K)] # Case 3: expression — `SELECT a+1 AS a_plus_1` — the TopK filter is on # `a_plus_1`, the scan predicate must read `a@0 + 1`. @@ -673,7 +673,7 @@ EXPLAIN ANALYZE SELECT a + 1 AS a_plus_1, b FROM topk_proj ORDER BY a_plus_1 LIM ---- Plan with Metrics 01)SortExec: TopK(fetch=2), expr=[a_plus_1@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[a_plus_1@0 < 3], metrics=[output_rows=2, output_batches=1, row_replacements=2] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_proj.parquet]]}, projection=[CAST(a@0 AS Int64) + 1 as a_plus_1, b], file_type=parquet, predicate=DynamicFilter [ CAST(a@0 AS Int64) + 1 < 3 ], dynamic_rg_pruning=eligible, metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=3, pushdown_rows_pruned=0, predicate_cache_inner_records=3, predicate_cache_records=3, scan_efficiency_ratio=13.21% (141/1.07 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_proj.parquet]]}, projection=[CAST(a@0 AS Int64) + 1 as a_plus_1, b], file_type=parquet, predicate=DynamicFilter [ CAST(a@0 AS Int64) + 1 < 3 ], dynamic_rg_pruning=eligible, metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=3, pushdown_rows_pruned=0, predicate_cache_inner_records=3, predicate_cache_records=3, scan_efficiency_ratio=13.21% (141/1.07 K)] # Case 4: alias shadowing — `SELECT a+1 AS a` — the projection renames # `a+1` to `a`, so the TopK's `a < 3` must still be rewritten to @@ -683,7 +683,7 @@ EXPLAIN ANALYZE SELECT a + 1 AS a, b FROM topk_proj ORDER BY a LIMIT 2; ---- Plan with Metrics 01)SortExec: TopK(fetch=2), expr=[a@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[a@0 < 3], metrics=[output_rows=2, output_batches=1, row_replacements=2] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_proj.parquet]]}, projection=[CAST(a@0 AS Int64) + 1 as a, b], file_type=parquet, predicate=DynamicFilter [ CAST(a@0 AS Int64) + 1 < 3 ], sort_order_for_reorder=[a@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=3, pushdown_rows_pruned=0, predicate_cache_inner_records=3, predicate_cache_records=3, scan_efficiency_ratio=13.21% (141/1.07 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_proj.parquet]]}, projection=[CAST(a@0 AS Int64) + 1 as a, b], file_type=parquet, predicate=DynamicFilter [ CAST(a@0 AS Int64) + 1 < 3 ], sort_order_for_reorder=[a@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=3, pushdown_rows_pruned=0, predicate_cache_inner_records=3, predicate_cache_records=3, scan_efficiency_ratio=13.21% (141/1.07 K)] statement ok reset datafusion.explain.analyze_categories; @@ -740,12 +740,12 @@ INNER JOIN ( ---- Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(a@0, a@0)], projection=[a@0, min_value@2], metrics=[output_rows=2, output_batches=2, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=2, input_rows=2, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/join_agg_build.parquet]]}, projection=[a], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=14.45% (64/443)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/join_agg_build.parquet]]}, projection=[a], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=14.45% (64/443)] 03)--ProjectionExec: expr=[a@0 as a, min(join_agg_probe.value)@1 as min_value], metrics=[output_rows=2, output_batches=2] 04)----AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[min(join_agg_probe.value)], metrics=[output_rows=2, output_batches=2, spill_count=0, spilled_rows=0] 05)------RepartitionExec: partitioning=Hash([a@0], 4), input_partitions=1, metrics=[output_rows=2, output_batches=2, spill_count=0, spilled_rows=0] 06)--------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[min(join_agg_probe.value)], metrics=[output_rows=2, output_batches=1, spill_count=0, spilled_rows=0, skipped_aggregation_rows=0, reduction_factor=100% (2/2)] -07)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/join_agg_probe.parquet]]}, projection=[a, value], file_type=parquet, predicate=DynamicFilter [ a@0 >= h1 AND a@0 <= h2 AND a@0 IN (SET) ([h1, h2]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= h1 AND a_null_count@1 != row_count@2 AND a_min@3 <= h2 AND (a_null_count@1 != row_count@2 AND a_min@3 <= h1 AND h1 <= a_max@0 OR a_null_count@1 != row_count@2 AND a_min@3 <= h2 AND h2 <= a_max@0), required_guarantees=[a in (h1, h2)], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=4 total → 4 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=4, predicate_cache_records=2, scan_efficiency_ratio=19.07% (151/792)] +07)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/join_agg_probe.parquet]]}, projection=[a, value], file_type=parquet, predicate=DynamicFilter [ a@0 >= h1 AND a@0 <= h2 AND a@0 IN (SET) ([h1, h2]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= h1 AND a_null_count@1 != row_count@2 AND a_min@3 <= h2 AND (a_null_count@1 != row_count@2 AND a_min@3 <= h1 AND h1 <= a_max@0 OR a_null_count@1 != row_count@2 AND a_min@3 <= h2 AND h2 <= a_max@0), required_guarantees=[a in (h1, h2)], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=4 total → 4 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=4, predicate_cache_records=2, scan_efficiency_ratio=19.07% (151/792)] statement ok reset datafusion.explain.analyze_categories; @@ -807,8 +807,8 @@ ON nulls_build.a = nulls_probe.a AND nulls_build.b = nulls_probe.b; ---- Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(a@0, a@0), (b@1, b@1)], metrics=[output_rows=1, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=3, input_batches=1, input_rows=1, avg_fanout=100% (1/1), probe_hit_rate=100% (1/1)] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nulls_build.parquet]]}, projection=[a, b], file_type=parquet, metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=18.6% (144/774)] -03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nulls_probe.parquet]]}, projection=[a, b, c], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= 1 AND b@1 <= 2 AND struct(a@0, b@1) IN (SET) ([{c0:aa,c1:1}, {c0:,c1:2}, {c0:ab,c1:}]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= 1 AND b_null_count@5 != row_count@2 AND b_min@6 <= 2, required_guarantees=[], metrics=[output_rows=1, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=1, pushdown_rows_pruned=3, predicate_cache_inner_records=8, predicate_cache_records=2, scan_efficiency_ratio=20.18% (225/1.11 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nulls_build.parquet]]}, projection=[a, b], file_type=parquet, metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=18.6% (144/774)] +03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nulls_probe.parquet]]}, projection=[a, b, c], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= 1 AND b@1 <= 2 AND struct(a@0, b@1) IN (SET) ([{c0:aa,c1:1}, {c0:,c1:2}, {c0:ab,c1:}]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= 1 AND b_null_count@5 != row_count@2 AND b_min@6 <= 2, required_guarantees=[], metrics=[output_rows=1, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=1, pushdown_rows_pruned=3, predicate_cache_inner_records=8, predicate_cache_records=2, scan_efficiency_ratio=20.18% (225/1.11 K)] statement ok reset datafusion.explain.analyze_categories; @@ -873,8 +873,8 @@ ON lj_build.a = lj_probe.a AND lj_build.b = lj_probe.b; ---- Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=Left, on=[(a@0, a@0), (b@1, b@1)], metrics=[output_rows=2, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=2, input_rows=2, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/lj_build.parquet]]}, projection=[a, b, c], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=19.58% (196/1.00 K)] -03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/lj_probe.parquet]]}, projection=[a, b, e], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= ba AND b@1 <= bb AND struct(a@0, b@1) IN (SET) ([{c0:aa,c1:ba}, {c0:ab,c1:bb}]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= ba AND b_null_count@5 != row_count@2 AND b_min@6 <= bb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=22.05% (228/1.03 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/lj_build.parquet]]}, projection=[a, b, c], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=19.58% (196/1.00 K)] +03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/lj_probe.parquet]]}, projection=[a, b, e], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= ba AND b@1 <= bb AND struct(a@0, b@1) IN (SET) ([{c0:aa,c1:ba}, {c0:ab,c1:bb}]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= ba AND b_null_count@5 != row_count@2 AND b_min@6 <= bb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=22.05% (228/1.03 K)] # LEFT SEMI JOIN: only matching build rows are returned; probe scan still # receives the dynamic filter. @@ -889,8 +889,8 @@ WHERE EXISTS ( ---- Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=LeftSemi, on=[(a@0, a@0), (b@1, b@1)], metrics=[output_rows=2, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=2, input_rows=4, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/lj_build.parquet]]}, projection=[a, b, c], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=19.58% (196/1.00 K)] -03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/lj_probe.parquet]]}, projection=[a, b], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= ba AND b@1 <= bb AND struct(a@0, b@1) IN (SET) ([{c0:aa,c1:ba}, {c0:ab,c1:bb}]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= ba AND b_null_count@5 != row_count@2 AND b_min@6 <= bb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=14.89% (154/1.03 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/lj_build.parquet]]}, projection=[a, b, c], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=19.58% (196/1.00 K)] +03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/lj_probe.parquet]]}, projection=[a, b], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= ba AND b@1 <= bb AND struct(a@0, b@1) IN (SET) ([{c0:aa,c1:ba}, {c0:ab,c1:bb}]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= ba AND b_null_count@5 != row_count@2 AND b_min@6 <= bb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=14.89% (154/1.03 K)] statement ok reset datafusion.explain.analyze_categories; @@ -959,8 +959,8 @@ FROM hl_probe p INNER JOIN hl_build AS build ---- Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(a@0, a@0), (b@1, b@1)], projection=[a@3, b@4, c@2, e@5], metrics=[output_rows=2, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=1, input_rows=2, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/hl_build.parquet]]}, projection=[a, b, c], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=19.58% (196/1.00 K)] -03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/hl_probe.parquet]]}, projection=[a, b, e], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= ba AND b@1 <= bb AND hash_lookup ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= ba AND b_null_count@5 != row_count@2 AND b_min@6 <= bb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=22.05% (228/1.03 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/hl_build.parquet]]}, projection=[a, b, c], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=19.58% (196/1.00 K)] +03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/hl_probe.parquet]]}, projection=[a, b, e], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= ba AND b@1 <= bb AND hash_lookup ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= ba AND b_null_count@5 != row_count@2 AND b_min@6 <= bb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=22.05% (228/1.03 K)] statement ok drop table hl_build; @@ -1008,8 +1008,8 @@ FROM int_build b INNER JOIN int_probe p ---- Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(id1@0, id1@0), (id2@1, id2@1)], projection=[id1@0, id2@1, value@2, data@5], metrics=[output_rows=2, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=1, input_rows=2, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/int_build.parquet]]}, projection=[id1, id2, value], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=18.23% (204/1.12 K)] -03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/int_probe.parquet]]}, projection=[id1, id2, data], file_type=parquet, predicate=DynamicFilter [ id1@0 >= 1 AND id1@0 <= 2 AND id2@1 >= 10 AND id2@1 <= 20 AND hash_lookup ], dynamic_rg_pruning=eligible, pruning_predicate=id1_null_count@1 != row_count@2 AND id1_max@0 >= 1 AND id1_null_count@1 != row_count@2 AND id1_min@3 <= 2 AND id2_null_count@5 != row_count@2 AND id2_max@4 >= 10 AND id2_null_count@5 != row_count@2 AND id2_min@6 <= 20, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=20.67% (221/1.07 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/int_build.parquet]]}, projection=[id1, id2, value], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=18.23% (204/1.12 K)] +03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/int_probe.parquet]]}, projection=[id1, id2, data], file_type=parquet, predicate=DynamicFilter [ id1@0 >= 1 AND id1@0 <= 2 AND id2@1 >= 10 AND id2@1 <= 20 AND hash_lookup ], dynamic_rg_pruning=eligible, pruning_predicate=id1_null_count@1 != row_count@2 AND id1_max@0 >= 1 AND id1_null_count@1 != row_count@2 AND id1_min@3 <= 2 AND id2_null_count@5 != row_count@2 AND id2_max@4 >= 10 AND id2_null_count@5 != row_count@2 AND id2_min@6 <= 20, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, row_groups_pruned_dictionary=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=20.67% (221/1.07 K)] statement ok reset datafusion.explain.analyze_categories; diff --git a/docs/source/user-guide/configs.md b/docs/source/user-guide/configs.md index e01af3476b94c..ad97f6e84f669 100644 --- a/docs/source/user-guide/configs.md +++ b/docs/source/user-guide/configs.md @@ -92,6 +92,7 @@ The following configuration settings are available: | datafusion.execution.parquet.coerce_int96 | NULL | (reading) If true, parquet reader will read columns of physical type int96 as originating from a different resolution than nanosecond. This is useful for reading data from systems like Spark which stores microsecond resolution timestamps in an int96 allowing it to write values with a larger date range than 64-bit timestamps with nanosecond resolution. | | datafusion.execution.parquet.coerce_int96_tz | NULL | (reading) Optional timezone applied to INT96 columns when `coerce_int96` is set. When `Some`, INT96 columns coerce to `Timestamp(, Some())` instead of the default `Timestamp(, None)`. Spark and other systems write INT96 values as UTC-adjusted instants, so callers that need the resulting Arrow type to be timezone-aware (e.g. for Spark `TimestampType` semantics) should set this to `"UTC"`. No effect when `coerce_int96` is `None`. | | datafusion.execution.parquet.bloom_filter_on_read | true | (reading) Use any available bloom filters when reading parquet files | +| datafusion.execution.parquet.dictionary_filter_on_read | false | (reading) Use fully dictionary-encoded `BYTE_ARRAY` (Utf8/Binary) column chunks as an exact row-group membership index when reading parquet files. Unlike bloom filters, a fully dictionary-encoded column chunk's dictionary is the exact, complete set of the row group's distinct values, so this can prune both `IN`/`=` and `NOT IN`/`!=` predicates. Only column chunks whose page encoding statistics prove every data page came from the dictionary are used; chunks that fell back to `PLAIN` encoding are ignored. | | datafusion.execution.parquet.max_predicate_cache_size | NULL | (reading) The maximum predicate cache size, in bytes. When `pushdown_filters` is enabled, sets the maximum memory used to cache the results of predicate evaluation between filter evaluation and output generation. Decreasing this value will reduce memory usage, but may increase IO and CPU usage. None means use the default parquet reader setting. 0 means no caching. | | datafusion.execution.parquet.data_pagesize_limit | 1048576 | (writing) Sets best effort maximum size of data page in bytes | | datafusion.execution.parquet.write_batch_size | 1024 | (writing) Sets write_batch_size in rows |