From fcb04700c6469ad3e1c4e4fd9e62192ef7f02d43 Mon Sep 17 00:00:00 2001 From: Adrian Garcia Badaracco <1755071+adriangb@users.noreply.github.com> Date: Wed, 23 Sep 2026 19:44:12 -0500 Subject: [PATCH 01/42] feat(parquet-datasource): always accept pushable filters, run rejected conjuncts post-scan MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Rebased onto main. The original commit 1 of this PR ("extract DecoderProjection from build_stream") landed independently on main as current `decoder_projection` / single-decoder + `rg_plan` model. Two changes, both applied inside the parquet scan so the parent `FilterExec` can be removed unconditionally for pushable filters: 1. Never drop conjuncts the `RowFilter` cannot place. `build_row_filter` previously `.flatten()`-ed away conjuncts that `FilterCandidateBuilder::build` rejected (whole-struct references, per-file physical-schema mismatches) and swallowed whole-build errors. By the time it runs, `try_pushdown_filters` has already removed the `FilterExec`, so those conjuncts were applied nowhere — wrong results. `build_row_filter` now returns `(Option, Vec)`, `RowFilterGenerator` exposes `rejected_conjuncts()`, and a whole-file build error routes every conjunct to the rejected list rather than relaxing the predicate. 2. Always accept pushable filters and run the remainder post-scan. `try_pushdown_filters` reports each pushable filter as accepted so the `FilterExec` is always removed; the scan owns the predicate. The opener routes conjuncts to two places, applying every one: - pushdown_filters=true -> row-filterable conjuncts via the parquet `RowFilter`; rejected conjuncts via the in-scan post-scan filter. - pushdown_filters=false -> the whole predicate runs as a post-scan filter on decoded batches (behaviorally identical to `FilterExec`). Implementation: - `DecoderProjection` (main's `decoder_projection` module) grows a `post_scan_conjuncts` parameter: it widens the decoder mask over (user projection ∪ post-scan filter columns), rebases the conjuncts onto the stream schema, and returns a `PostScanFilter` applied to every decoded batch with SQL `WHERE` semantics. Virtual-column conjuncts are stripped from the read-plan mask (they aren't file columns) but kept in the post-scan predicate, which sees the reader-appended virtual columns. - `PushDecoderStreamState` applies the post-scan filter in the decoded- batch arm, skips empty batches, and re-introduces a stream-level `remaining_limit` (main enforces LIMIT decoder-locally, which is unsafe once a post-scan filter can reject rows). The opener routes the limit to `remaining_limit` iff a post-scan filter is present. - New `post_scan_rows_pruned` / `post_scan_rows_matched` counters and `post_scan_filter_eval_time` on `ParquetFileMetrics`. Tests: - `build_row_filter_surfaces_rejected_struct_conjunct` (row_filter.rs) asserts the rejected struct conjunct is returned, not dropped. - `rejected_struct_conjunct_runs_post_scan_not_dropped` (opener) is end-to-end: `s IS NOT NULL` over a struct column with pushdown on returns 2 (was 3 before the fix). - Parquet `.slt` files regenerated: `FilterExec` above `DataSourceExec` gone, predicate on the scan, `post_scan_rows_*` metrics on EXPLAIN ANALYZE. Opener / core insta / page_pruning assertions updated for the now-applied predicate. Squashed follow-up commits (see their original messages on the adriangb/parquet-post-scan-filter branch): - perf(parquet-datasource): narrow to the projector's columns before filtering - perf(parquet-datasource): coalesce post-scan filter output back to batch_size - perf(parquet-datasource): compact between conjuncts in the post-scan filter Co-Authored-By: Claude Opus 5 --- datafusion/core/src/dataframe/parquet.rs | 9 +- .../src/datasource/physical_plan/parquet.rs | 77 ++- datafusion/core/src/datasource/view_test.rs | 9 +- datafusion/core/tests/parquet/page_pruning.rs | 53 +- datafusion/core/tests/sql/explain_analyze.rs | 17 +- .../benches/parquet_nested_filter_pushdown.rs | 6 +- .../benches/parquet_struct_filter_pushdown.rs | 6 +- .../src/decoder_projection.rs | 472 +++++++++++++++++- datafusion/datasource-parquet/src/metrics.rs | 24 + .../datasource-parquet/src/opener/mod.rs | 292 +++++++++-- .../datasource-parquet/src/push_decoder.rs | 174 ++++++- .../datasource-parquet/src/row_filter.rs | 191 +++++-- datafusion/datasource-parquet/src/source.rs | 16 +- .../sqllogictest/test_files/clickbench.slt | 144 ++---- datafusion/sqllogictest/test_files/cte.slt | 4 +- .../sqllogictest/test_files/explain_tree.slt | 29 +- .../test_files/grouping_set_repartition.slt | 25 +- .../test_files/monotonic_projection_test.slt | 7 +- .../sqllogictest/test_files/parquet.slt | 20 +- .../test_files/parquet_filter_pushdown.slt | 20 +- .../parquet_max_row_group_bytes.slt | 12 +- .../test_files/parquet_statistics.slt | 35 +- .../test_files/preserve_file_partitioning.slt | 14 +- .../sqllogictest/test_files/projection.slt | 5 +- .../test_files/projection_pushdown.slt | 130 ++--- .../test_files/repartition_scan.slt | 14 +- .../repartition_subset_satisfaction.slt | 10 +- 27 files changed, 1279 insertions(+), 536 deletions(-) diff --git a/datafusion/core/src/dataframe/parquet.rs b/datafusion/core/src/dataframe/parquet.rs index dd751eb97779f..ae55f672b44db 100644 --- a/datafusion/core/src/dataframe/parquet.rs +++ b/datafusion/core/src/dataframe/parquet.rs @@ -162,9 +162,14 @@ mod tests { .select_columns(&["bool_col", "int_col"])?; let plan = df.explain(false, false)?.collect().await?; - // Filters all the way to Parquet + // Filters all the way to Parquet. The parquet scan now accepts the + // pushable filter unconditionally so the `FilterExec` is removed — + // the predicate appears as `predicate=` on the `DataSourceExec`. let formatted = pretty::pretty_format_batches(&plan)?.to_string(); - assert!(formatted.contains("FilterExec: id@0 = 1"), "{formatted}"); + assert!( + formatted.contains("predicate=id@0 = 1"), + "expected predicate=id@0 = 1 in {formatted}" + ); Ok(()) } diff --git a/datafusion/core/src/datasource/physical_plan/parquet.rs b/datafusion/core/src/datasource/physical_plan/parquet.rs index d562bbe8490f4..a8c766c1b020b 100644 --- a/datafusion/core/src/datasource/physical_plan/parquet.rs +++ b/datafusion/core/src/datasource/physical_plan/parquet.rs @@ -752,17 +752,12 @@ mod tests { .await .unwrap(); - insta::assert_snapshot!(batches_to_sort_string(&read),@r" - +-----+----+----+ - | c1 | c3 | c2 | - +-----+----+----+ - | | | | - | | 10 | 1 | - | | 20 | | - | | 20 | 2 | - | Foo | 10 | | - | bar | | | - +-----+----+----+ + insta::assert_snapshot!(batches_to_sort_string(&read),@" + +----+----+----+ + | c1 | c3 | c2 | + +----+----+----+ + | | 20 | 2 | + +----+----+----+ "); } @@ -897,7 +892,9 @@ mod tests { assert_eq!(get_value(&metrics, "pushdown_rows_matched"), 0); assert_eq!(rt.batches.unwrap().len(), 0); - // Predicate should prune no row groups + // Predicate should prune no row groups. The scan now applies the + // predicate post-scan (since `pushdown_filters` is off in this test), + // so only the matching row survives. let filter = col("c1").eq(lit(ScalarValue::Utf8(Some("foo".to_string())))); let rt = RoundTrip::new() .with_predicate(filter) @@ -914,7 +911,7 @@ mod tests { .iter() .map(|b| b.num_rows()) .sum::(); - assert_eq!(read, 2, "Expected 2 rows to match the predicate"); + assert_eq!(read, 1, "Expected 1 row to match the predicate"); } #[tokio::test] @@ -938,7 +935,9 @@ mod tests { assert_eq!(get_value(&metrics, "predicate_evaluation_errors"), 0); assert_eq!(rt.batches.unwrap().len(), 0); - // Predicate should prune no row groups + // Predicate should prune no row groups. The scan now applies the + // predicate post-scan (since `pushdown_filters` is off in this test), + // so only the matching row survives. let filter = col("c1").eq(lit(ScalarValue::UInt64(Some(1)))); let rt = RoundTrip::new() .with_predicate(filter) @@ -954,7 +953,7 @@ mod tests { .iter() .map(|b| b.num_rows()) .sum::(); - assert_eq!(read, 2, "Expected 2 rows to match the predicate"); + assert_eq!(read, 1, "Expected 1 row to match the predicate"); } #[tokio::test] @@ -1052,17 +1051,12 @@ mod tests { // In a real query where this predicate was pushed down from a filter stage instead of created directly in the `DataSourceExec`, // the filter stage would be preserved as a separate execution plan stage so the actual query results would be as expected. - insta::assert_snapshot!(batches_to_sort_string(&read),@r" - +-----+----+ - | c1 | c2 | - +-----+----+ - | | | - | | | - | | 1 | - | | 2 | - | Foo | | - | bar | | - +-----+----+ + insta::assert_snapshot!(batches_to_sort_string(&read),@" + +----+----+ + | c1 | c2 | + +----+----+ + | | 1 | + +----+----+ "); } @@ -1148,23 +1142,14 @@ mod tests { .round_trip(vec![batch1, batch2, batch3, batch4]) .await; - insta::assert_snapshot!(batches_to_sort_string(&rt.batches.unwrap()), @r" - +------+----+ - | c1 | c2 | - +------+----+ - | | 1 | - | | 2 | - | Bar | | - | Bar | 2 | - | Bar | 2 | - | Bar2 | | - | Bar3 | | - | Foo | | - | Foo | 1 | - | Foo | 1 | - | Foo2 | | - | Foo3 | | - +------+----+ + insta::assert_snapshot!(batches_to_sort_string(&rt.batches.unwrap()), @" + +-----+----+ + | c1 | c2 | + +-----+----+ + | | 1 | + | Foo | 1 | + | Foo | 1 | + +-----+----+ "); let metrics = rt.parquet_exec.metrics().unwrap(); @@ -1233,11 +1218,10 @@ mod tests { .await .unwrap(); - insta::assert_snapshot!(batches_to_sort_string(&read),@r" + insta::assert_snapshot!(batches_to_sort_string(&read),@" +-----+----+ | c1 | c2 | +-----+----+ - | | 2 | | Foo | 1 | | bar | | +-----+----+ @@ -1756,12 +1740,11 @@ mod tests { let metrics = rt.parquet_exec.metrics().unwrap(); - assert_snapshot!(batches_to_sort_string(&rt.batches.unwrap()),@r" + assert_snapshot!(batches_to_sort_string(&rt.batches.unwrap()),@" +-----+ | int | +-----+ | 4 | - | 5 | +-----+ "); let (page_index_rows_pruned, page_index_rows_matched) = diff --git a/datafusion/core/src/datasource/view_test.rs b/datafusion/core/src/datasource/view_test.rs index 35418d6dea632..a5b21450d6619 100644 --- a/datafusion/core/src/datasource/view_test.rs +++ b/datafusion/core/src/datasource/view_test.rs @@ -322,11 +322,16 @@ mod tests { let plan = df.explain(false, false)?.collect().await?; - // Filters all the way to Parquet + // Filters all the way to Parquet. The parquet scan now accepts the + // pushable filter unconditionally so the `FilterExec` is removed — + // the predicate appears as `predicate=` on the `DataSourceExec`. let formatted = arrow::util::pretty::pretty_format_batches(&plan) .unwrap() .to_string(); - assert!(formatted.contains("FilterExec: id@0 = 1")); + assert!( + formatted.contains("predicate=id@0 = 1"), + "expected predicate=id@0 = 1 in {formatted}" + ); Ok(()) } diff --git a/datafusion/core/tests/parquet/page_pruning.rs b/datafusion/core/tests/parquet/page_pruning.rs index 70052ec6b7d60..865429efe5d9c 100644 --- a/datafusion/core/tests/parquet/page_pruning.rs +++ b/datafusion/core/tests/parquet/page_pruning.rs @@ -120,13 +120,10 @@ async fn page_index_filter_one_col() { let filter = col("month").eq(lit(1_i32)); let batches = get_filter_results(&state, filter.clone(), false).await; - // `month = 1` from the page index should create below RowSelection - // vec.push(RowSelector::select(312)); - // vec.push(RowSelector::skip(3330)); - // vec.push(RowSelector::select(339)); - // vec.push(RowSelector::skip(3319)); - // total 651 row - assert_eq!(batches[0].num_rows(), 651); + // `month = 1` from the page index narrows IO to ~651 candidate rows + // (RowSelector skip/select pattern), and the in-scan post-scan filter + // then rejects the 31 page-aligned rows that don't actually match. + assert_eq!(batches[0].num_rows(), 620); let batches = get_filter_results(&state, filter, true).await; assert_eq!(batches[0].num_rows(), 620); @@ -134,16 +131,9 @@ async fn page_index_filter_one_col() { // 2. create filter month == 1 or month == 2; let filter = col("month").eq(lit(1_i32)).or(col("month").eq(lit(2_i32))); let batches = get_filter_results(&state, filter.clone(), false).await; - // `month = 1` or `month = 2` from the page index should create below RowSelection - // vec.push(RowSelector::select(312)); - // vec.push(RowSelector::skip(900)); - // vec.push(RowSelector::select(312)); - // vec.push(RowSelector::skip(2118)); - // vec.push(RowSelector::select(339)); - // vec.push(RowSelector::skip(873)); - // vec.push(RowSelector::select(318)); - // vec.push(RowSelector::skip(2128)); - assert_eq!(batches[0].num_rows(), 1281); + // Page-index pruning leaves a 1281-row candidate set; the scan applies + // the predicate (RowFilter or post-scan) to land on 1180 matches. + assert_eq!(batches[0].num_rows(), 1180); let batches = get_filter_results(&state, filter, true).await; assert_eq!(batches[0].num_rows(), 1180); @@ -161,8 +151,8 @@ async fn page_index_filter_one_col() { // 4.create filter 0 < month < 2 ; let filter = col("month").gt(lit(0_i32)).and(col("month").lt(lit(2_i32))); let batches = get_filter_results(&state, filter.clone(), false).await; - // should same with `month = 1` - assert_eq!(batches[0].num_rows(), 651); + // Same matching set as `month = 1`. + assert_eq!(batches[0].num_rows(), 620); let batches = get_filter_results(&state, filter, true).await; assert_eq!(batches[0].num_rows(), 620); @@ -170,14 +160,9 @@ async fn page_index_filter_one_col() { // Note this test doesn't apply type coercion so the literal must match the actual view type let filter = col("date_string_col").eq(lit(ScalarValue::new_utf8view("01/01/09"))); let batches = get_filter_results(&state, filter.clone(), false).await; - assert_eq!(batches[0].num_rows(), 14); - - // there should only two pages match the filter - // min max - // page-20 0 01/01/09 01/02/09 - // page-21 0 01/01/09 01/01/09 - // each 7 rows - assert_eq!(batches[0].num_rows(), 14); + // Page index narrows to two pages of 7 rows each (14 rows); the scan + // then rejects the 4 page-aligned rows that don't actually match. + assert_eq!(batches[0].num_rows(), 10); let batches = get_filter_results(&state, filter, true).await; assert_eq!(batches[0].num_rows(), 10); } @@ -190,11 +175,9 @@ async fn page_index_filter_multi_col() { // create filter month == 1 and year = 2009; let filter = col("month").eq(lit(1_i32)).and(col("year").eq(lit(2009))); let batches = get_filter_results(&state, filter.clone(), false).await; - // `year = 2009` from the page index should create below RowSelection - // vec.push(RowSelector::select(3663)); - // vec.push(RowSelector::skip(3642)); - // combine with `month = 1` total 333 row - assert_eq!(batches[0].num_rows(), 333); + // Page index narrows IO to ~333 candidate rows; the scan then rejects + // the page-aligned rows that don't actually match, landing on 310. + assert_eq!(batches[0].num_rows(), 310); let batches = get_filter_results(&state, filter, true).await; assert_eq!(batches[0].num_rows(), 310); @@ -204,7 +187,7 @@ async fn page_index_filter_multi_col() { .eq(lit(1_i32)) .and(col("year").eq(lit(2009)).or(col("id").eq(lit(1)))); let batches = get_filter_results(&state, filter.clone(), false).await; - assert_eq!(batches[0].num_rows(), 651); + assert_eq!(batches[0].num_rows(), 310); let batches = get_filter_results(&state, filter, true).await; assert_eq!(batches[0].num_rows(), 310); @@ -212,7 +195,7 @@ async fn page_index_filter_multi_col() { // this filter use two columns will not push down let filter = col("year").eq(lit(2009)).or(col("id").eq(lit(1))); let batches = get_filter_results(&state, filter.clone(), false).await; - assert_eq!(batches[0].num_rows(), 7300); + assert_eq!(batches[0].num_rows(), 3650); let batches = get_filter_results(&state, filter, true).await; assert_eq!(batches[0].num_rows(), 3650); @@ -225,7 +208,7 @@ async fn page_index_filter_multi_col() { .and(col("id").eq(lit(1))) .or(col("year").eq(lit(2010))); let batches = get_filter_results(&state, filter.clone(), false).await; - assert_eq!(batches[0].num_rows(), 7300); + assert_eq!(batches[0].num_rows(), 3651); let batches = get_filter_results(&state, filter, true).await; assert_eq!(batches[0].num_rows(), 3651); } diff --git a/datafusion/core/tests/sql/explain_analyze.rs b/datafusion/core/tests/sql/explain_analyze.rs index 67614cfbdea9a..aa355260446e8 100644 --- a/datafusion/core/tests/sql/explain_analyze.rs +++ b/datafusion/core/tests/sql/explain_analyze.rs @@ -877,8 +877,12 @@ async fn parquet_explain_analyze() { .unwrap() .to_string(); - // should contain aggregated stats - assert_contains!(&formatted, "output_rows=8"); + // should contain aggregated stats. The scan now applies the predicate + // post-scan (the `FilterExec` above it is removed), so `output_rows` is + // the matching row count (5) rather than the decoded row count (8). + // The 3 rejected rows show up as `post_scan_rows_pruned=3`. + assert_contains!(&formatted, "output_rows=5"); + assert_contains!(&formatted, "post_scan_rows_pruned=3"); assert_contains!( &formatted, "row_groups_pruned_bloom_filter=1 total \u{2192} 1 matched" @@ -1069,14 +1073,13 @@ async fn parquet_recursive_projection_pushdown() -> Result<()> { assert_snapshot!( actual, - @r" + @" SortExec: expr=[id@0 ASC NULLS LAST], preserve_partitioning=[false] RecursiveQueryExec: name=number_series, is_distinct=false CoalescePartitionsExec - ProjectionExec: expr=[CAST(id@0 AS Int64) as id, CAST(1 AS Int64) as level] - FilterExec: id@0 = 1 - RepartitionExec: partitioning=RoundRobinBatch(NUM_CORES), input_partitions=1 - DataSourceExec: file_groups={1 group: [[TMP_DIR/hierarchy.parquet]]}, projection=[id], file_type=parquet, predicate=id@0 = 1, pruning_predicate=id_null_count@2 != row_count@3 AND id_min@0 <= 1 AND 1 <= id_max@1, required_guarantees=[id in (1)] + ProjectionExec: expr=[CAST(id@0 AS Int64) as id, CAST(level@1 AS Int64) as level] + RepartitionExec: partitioning=RoundRobinBatch(NUM_CORES), input_partitions=1 + DataSourceExec: file_groups={1 group: [[TMP_DIR/hierarchy.parquet]]}, projection=[id, 1 as level], file_type=parquet, predicate=id@0 = 1, pruning_predicate=id_null_count@2 != row_count@3 AND id_min@0 <= 1 AND 1 <= id_max@1, required_guarantees=[id in (1)] CoalescePartitionsExec ProjectionExec: expr=[id@0 + 1 as id, level@1 + 1 as level] FilterExec: id@0 < 10 diff --git a/datafusion/datasource-parquet/benches/parquet_nested_filter_pushdown.rs b/datafusion/datasource-parquet/benches/parquet_nested_filter_pushdown.rs index 02137b5a1d288..6a1ed3bdb25e2 100644 --- a/datafusion/datasource-parquet/benches/parquet_nested_filter_pushdown.rs +++ b/datafusion/datasource-parquet/benches/parquet_nested_filter_pushdown.rs @@ -115,9 +115,9 @@ fn scan_with_predicate( let file_metrics = ParquetFileMetrics::new(0, &path.display().to_string(), &metrics); let builder = if pushdown { - if let Some(row_filter) = - build_row_filter(predicate, file_schema, &metadata, false, &file_metrics)? - { + let (row_filter, _rejected) = + build_row_filter(predicate, file_schema, &metadata, false, &file_metrics)?; + if let Some(row_filter) = row_filter { builder.with_row_filter(row_filter) } else { builder diff --git a/datafusion/datasource-parquet/benches/parquet_struct_filter_pushdown.rs b/datafusion/datasource-parquet/benches/parquet_struct_filter_pushdown.rs index b52408d4222d8..17eab65983720 100644 --- a/datafusion/datasource-parquet/benches/parquet_struct_filter_pushdown.rs +++ b/datafusion/datasource-parquet/benches/parquet_struct_filter_pushdown.rs @@ -210,9 +210,9 @@ fn scan( let mut filter_applied = false; let builder = if pushdown { - if let Some(row_filter) = - build_row_filter(predicate, file_schema, &metadata, false, &file_metrics)? - { + let (row_filter, _rejected) = + build_row_filter(predicate, file_schema, &metadata, false, &file_metrics)?; + if let Some(row_filter) = row_filter { filter_applied = true; builder.with_row_filter(row_filter) } else { diff --git a/datafusion/datasource-parquet/src/decoder_projection.rs b/datafusion/datasource-parquet/src/decoder_projection.rs index 89fdc01af4eda..ac8499fa3f9c9 100644 --- a/datafusion/datasource-parquet/src/decoder_projection.rs +++ b/datafusion/datasource-parquet/src/decoder_projection.rs @@ -33,20 +33,201 @@ use std::sync::Arc; -use arrow::array::{RecordBatch, RecordBatchOptions}; +use arrow::array::{Array, BooleanArray, RecordBatch, RecordBatchOptions}; +use arrow::compute::kernels::boolean::and; +use arrow::compute::kernels::filter::{filter_record_batch, prep_null_mask_filter}; use arrow::datatypes::SchemaRef; -use datafusion_common::Result; +use datafusion_common::cast::as_boolean_array; +use datafusion_common::{Result, internal_err}; use datafusion_physical_expr::projection::{ProjectionExprs, Projector}; -use datafusion_physical_expr::utils::reassign_expr_columns; +use datafusion_physical_expr::split_conjunction; +use datafusion_physical_expr::utils::{collect_columns, reassign_expr_columns}; use datafusion_physical_expr_adapter::replace_columns_with_literals; +use datafusion_physical_expr_common::physical_expr::PhysicalExpr; +use datafusion_physical_plan::metrics::{Count, Time}; use parquet::arrow::ProjectionMask; use parquet::schema::types::SchemaDescriptor; +use crate::ParquetFileMetrics; use crate::opener::{VirtualColumnsState, append_fields}; use crate::projection_read_plan::build_projection_read_plan; +/// Stream-schema column indices referenced by `projection`, in ascending +/// order, or `None` when the projection already reads every column of +/// `stream_schema` (so narrowing to them would be an identity). +/// +/// `projection` must already be rebased onto `stream_schema`. +fn projector_input_indices( + projection: &ProjectionExprs, + stream_schema: &SchemaRef, +) -> Option> { + let mut indices: Vec = projection + .expr_iter() + .flat_map(|expr| collect_columns(&expr)) + .map(|col| col.index()) + .collect(); + indices.sort_unstable(); + indices.dedup(); + + // Nothing to drop: keep the batch (and the projector) as-is. This is the + // no-post-scan-filter case, where the decoder mask is already exactly the + // projection. + (indices.len() < stream_schema.fields().len()).then_some(indices) +} + +/// Compact the working batch to the surviving rows once a conjunct leaves at +/// most this fraction of them alive. Above the threshold the copy would not be +/// repaid by the (smaller) saving on the conjuncts that follow, so the masks are +/// just combined with a cheap bitwise `AND` instead. +/// +/// The threshold is higher than the equivalent constant in `FilterExec`'s +/// adaptive evaluator because the post-scan filter's conjunct mix is skewed: +/// the conjuncts that land here at `pushdown_filters = false` are typically a +/// cheap static range predicate followed by a *much* more expensive dynamic +/// filter (a `CASE` over per-partition hash-table probes), so even a modest +/// reduction in the row count reaching the later conjunct pays for the copy. +const COMPACTION_SELECTIVITY_THRESHOLD: f64 = 0.8; + +/// Outcome of running the post-scan predicate over one decoded batch. +/// +/// The surviving rows are described rather than fully materialized so the +/// caller can drop filter-only columns (see [`DecoderProjection::narrow`]) +/// before paying for the final filter kernel. +pub(crate) enum PostScanSelection { + /// No row survived; nothing to hand downstream. + Empty, + /// Some rows survived. `batch` is the working batch — the input batch, or a + /// compacted copy of it when the loop compacted at least once — and `mask` + /// is the residual selection over `batch`'s rows (`None` when every row of + /// `batch` survived). + Rows { + batch: RecordBatch, + mask: Option, + }, +} + +/// Predicate applied to decoded record batches inside the parquet scan. +/// +/// Semantically identical to a `FilterExec` over the scan: rows where the +/// predicate is not `true` are dropped. `NULL` predicate results drop the row +/// (SQL `WHERE` semantics — `filter_record_batch` treats null mask entries as +/// false, same as `FilterExec`'s `batch_filter`). +/// +/// # Compact-once evaluation +/// +/// The predicate is kept split into its `AND` conjuncts and evaluated +/// sequentially, compacting the working batch to the survivors whenever a +/// conjunct proves selective enough. A fused `BinaryExpr` `AND` does *not* +/// compact between conjuncts, so every conjunct is evaluated on ~every decoded +/// row; with `pushdown_filters = false` the whole predicate lands here, which +/// means an expensive dynamic filter would run on all decoded rows even though +/// a cheap static conjunct ahead of it already rejected most of them. Compacting +/// is what the parquet `RowFilter` path gets for free from arrow-rs (each +/// conjunct is its own `ArrowPredicate`, applied against an accumulating +/// `RowSelection`); this loop is the post-scan equivalent. +/// +/// Holds metric handles so per-batch rows-pruned / matched / time accumulate +/// into [`ParquetFileMetrics`] for `EXPLAIN ANALYZE`. +pub(crate) struct PostScanFilter { + /// The `AND` conjuncts of the predicate, rebased onto the decoder's stream + /// schema, in the order they are evaluated. Never empty. + conjuncts: Vec>, + rows_pruned: Count, + rows_matched: Count, + eval_time: Time, +} + +impl PostScanFilter { + /// Evaluate the conjuncts against `batch` and describe the surviving rows. + /// + /// Takes the batch by value because a compaction replaces it; the caller + /// gets the working batch back inside [`PostScanSelection::Rows`]. + /// + /// Note the batch handed back is still on the *full* stream schema (the + /// decoder mask widened for the predicate's columns): every conjunct may + /// need those columns, so narrowing to the projector's inputs must not + /// happen until the loop is done. + pub(crate) fn evaluate(&self, batch: RecordBatch) -> Result { + // Scoped timer: stops on drop, so the early-return paths still record. + let _timer = self.eval_time.timer(); + + let input_rows = batch.num_rows(); + if input_rows == 0 { + return Ok(PostScanSelection::Rows { batch, mask: None }); + } + + // `working` is what conjuncts are evaluated against; `acc` is the + // accumulated (null-free) selection over `working`'s rows since the last + // compaction, `None` meaning "all of them are still live". + let mut working = batch; + let mut acc: Option = None; + + let last = self.conjuncts.len() - 1; + for (position, conjunct) in self.conjuncts.iter().enumerate() { + let rows_in = working.num_rows(); + let array = conjunct.evaluate(&working)?.into_array(rows_in)?; + let Ok(mask) = as_boolean_array(array.as_ref()) else { + return internal_err!( + "post-scan filter predicate did not evaluate to a BooleanArray" + ); + }; + // `filter_record_batch` treats a null mask entry as false, so a + // surviving row is one that is `true` and non-null. + let mask = match mask.nulls() { + Some(_) => prep_null_mask_filter(mask), + None => mask.clone(), + }; + // An all-true conjunct leaves the accumulated selection untouched. + if mask.true_count() == rows_in { + continue; + } + + let folded = match &acc { + None => mask, + Some(previous) => and(previous, &mask)?, + }; + let alive = folded.true_count(); + if alive == 0 { + self.record(input_rows, 0); + return Ok(PostScanSelection::Empty); + } + + // Compaction only benefits the conjuncts that come *after* this + // one, so never compact on the last: its residual mask goes to the + // caller, which narrows away the filter-only columns before + // applying it. + if position < last + && (alive as f64) <= COMPACTION_SELECTIVITY_THRESHOLD * rows_in as f64 + { + working = filter_record_batch(&working, &folded)?; + acc = None; + } else { + acc = Some(folded); + } + } + + let survivors = match &acc { + Some(mask) => mask.true_count(), + None => working.num_rows(), + }; + self.record(input_rows, survivors); + Ok(PostScanSelection::Rows { + batch: working, + mask: acc, + }) + } + + /// Record one batch's contribution to the rows-matched / rows-pruned + /// metrics. `rows_pruned` stays "total rows in, minus final survivors" + /// regardless of how many intermediate compactions the loop performed. + fn record(&self, input_rows: usize, survivors: usize) { + self.rows_matched.add(survivors); + self.rows_pruned.add(input_rows - survivors); + } +} + /// Per-file decoder projection: the [`ProjectionMask`] installed on the /// parquet decoder, plus the per-batch transform that maps the decoder's /// output onto the scan's `output_schema`. @@ -63,6 +244,23 @@ pub(crate) struct DecoderProjection { /// in metadata / nullability and [`map`](Self::map) must rebuild the batch /// with `output_schema`. replace_schema: bool, + /// Predicate to apply on each decoded batch, after any row-level + /// `RowFilter` and before the projector. Carries conjuncts the `RowFilter` + /// machinery could not evaluate, plus the whole predicate when + /// `pushdown_filters = false`. `None` when no conjunct needs post-scan + /// evaluation, in which case the decoder mask covers exactly the user + /// projection and there is no extra per-batch work. + post_scan_filter: Option, + /// Stream-schema column indices the [`Projector`] reads, used by + /// [`narrow`](Self::narrow) to drop filter-only columns before the filter + /// kernel runs. `None` when the projector already reads every stream + /// column — always the case without a post-scan filter, where the decoder + /// mask covers exactly the projection and narrowing would be an identity. + narrow_indices: Option>, + /// Schema of a narrowed, filtered batch: what [`map`](Self::map) consumes, + /// and the schema any coalescer buffering these batches must be built with. + /// Equals the full stream schema when [`Self::narrow_indices`] is `None`. + filtered_schema: SchemaRef, } impl DecoderProjection { @@ -78,12 +276,25 @@ impl DecoderProjection { /// columns are stripped from the projection fed into /// `build_projection_read_plan` (which only understands file columns) and /// appended to the stream schema so the projector can resolve them. + /// + /// `post_scan_conjuncts` are predicate conjuncts that must be evaluated on + /// decoded batches inside the scan (conjuncts the parquet `RowFilter` + /// machinery could not place, plus the whole predicate when + /// `pushdown_filters = false`). They must reference columns in + /// `physical_file_schema` (virtual-column predicates are never pushed into + /// the scan). When non-empty the decoder mask is widened to include their + /// columns, the conjuncts are rebased onto the (widened) stream schema, and + /// they become [`Self::post_scan_filter`] — kept split so it can compact + /// between them. When empty this is exactly the prior projection-only + /// behaviour. pub(crate) fn try_new( projection: &ProjectionExprs, + post_scan_conjuncts: &[Arc], physical_file_schema: &SchemaRef, parquet_schema: &SchemaDescriptor, output_schema: &SchemaRef, virtual_state: Option<&VirtualColumnsState>, + file_metrics: &ParquetFileMetrics, ) -> Result { // Virtual columns are produced by the reader separately from the // projection mask, so strip them from the expressions we feed into @@ -97,8 +308,33 @@ impl DecoderProjection { replace_columns_with_literals(expr, state.null_replacements()) })?, }; + // Decoder reads (user projection ∪ post-scan filter columns). Row-level + // filter columns live inside the parquet RowFilter's per-predicate + // masks, so they don't need to be in this read plan. + // + // A post-scan conjunct may reference a virtual column (e.g. parquet + // `row_number`): the reader produces those separately, so — like the + // projection — strip them to null literals before feeding the read + // plan, which only understands file columns. The *original* conjuncts + // (with the virtual references intact) are still used below to build + // the post-scan predicate, which is rebased onto the stream schema + // where the reader has appended the virtual columns. + let post_scan_for_read_plan: Vec> = match virtual_state { + None => post_scan_conjuncts.to_vec(), + Some(state) => post_scan_conjuncts + .iter() + .map(|expr| { + replace_columns_with_literals( + Arc::clone(expr), + state.null_replacements(), + ) + }) + .collect::>>()?, + }; let read_plan = build_projection_read_plan( - projection_for_read_plan.expr_iter(), + projection_for_read_plan + .expr_iter() + .chain(post_scan_for_read_plan.iter().map(Arc::clone)), physical_file_schema, parquet_schema, ); @@ -118,18 +354,67 @@ impl DecoderProjection { let rebased_projection = projection .clone() .try_map_exprs(|expr| reassign_expr_columns(expr, &stream_schema))?; - let projector = rebased_projection.make_projector(&stream_schema)?; + + // When the mask was widened for post-scan filter columns, the stream + // batch carries columns the projector never reads. Filtering those + // through the (expensive) filter kernel only to drop them afterwards + // is pure waste, so narrow the batch to the projector's inputs first + // and rebase the projector onto that narrower schema — the same + // project-then-filter order `FilterExec::filter_and_project` uses. + let narrow_indices = projector_input_indices(&rebased_projection, &stream_schema); + let (projector, narrow_indices, filtered_schema) = match narrow_indices { + Some(indices) => { + let narrowed = Arc::new(stream_schema.project(&indices)?); + let renarrowed = rebased_projection + .clone() + .try_map_exprs(|expr| reassign_expr_columns(expr, &narrowed))?; + let projector = renarrowed.make_projector(&narrowed)?; + (projector, Some(indices), narrowed) + } + None => ( + rebased_projection.make_projector(&stream_schema)?, + None, + Arc::clone(&stream_schema), + ), + }; // Compare against the projector's *output* schema rather than the - // stream schema, so future widening of the mask (e.g. for post-scan - // filter columns) does not flip this flag. + // (possibly widened) stream schema, so widening the mask for post-scan + // filter columns does not flip this flag. let replace_schema = projector.output_schema() != output_schema; + // Rebase the post-scan conjuncts onto the same (widened) stream schema + // and conjoin them into a single predicate for per-batch evaluation. + let post_scan_filter = if post_scan_conjuncts.is_empty() { + None + } else { + // Split each conjunct again after rebasing: the caller's list is + // already conjunct-wise, but a single entry may still be a nested + // `AND` (e.g. a `RowFilter`-rejected conjunct). The compact-once + // loop can only compact between the pieces it can see. + let rebased = post_scan_conjuncts + .iter() + .map(|expr| reassign_expr_columns(Arc::clone(expr), &stream_schema)) + .collect::>>()? + .iter() + .flat_map(|expr| split_conjunction(expr).into_iter().map(Arc::clone)) + .collect::>(); + Some(PostScanFilter { + conjuncts: rebased, + rows_pruned: file_metrics.post_scan_rows_pruned.clone(), + rows_matched: file_metrics.post_scan_rows_matched.clone(), + eval_time: file_metrics.post_scan_filter_eval_time.clone(), + }) + }; + Ok(Self { projection_mask: read_plan.projection_mask, projector, output_schema: Arc::clone(output_schema), replace_schema, + post_scan_filter, + narrow_indices, + filtered_schema, }) } @@ -138,6 +423,43 @@ impl DecoderProjection { &self.projection_mask } + /// The post-scan filter for this file, if any conjunct needs per-batch + /// evaluation. Applied by the push-decoder stream to each decoded batch + /// (after any row-level `RowFilter`, before the projector). + pub(crate) fn post_scan_filter(&self) -> Option<&PostScanFilter> { + self.post_scan_filter.as_ref() + } + + /// Drop the columns only the post-scan filter needed, leaving just what + /// the [`Projector`] reads. + /// + /// Call this *after* [`PostScanFilter::evaluate`] (every conjunct may need + /// the filter-only columns) but *before* applying the residual mask it + /// returned: `RecordBatch::project` is a cheap `Arc` reslice, so narrowing + /// first keeps the final filter kernel off columns that would be discarded + /// immediately afterwards. Returns the batch unchanged when the projector + /// already reads every stream column. + pub(crate) fn narrow(&self, batch: RecordBatch) -> Result { + match &self.narrow_indices { + Some(indices) => Ok(batch.project(indices)?), + None => Ok(batch), + } + } + + /// Schema of the batches [`map`](Self::map) consumes, i.e. what + /// [`narrow`](Self::narrow) produces. Used to build the coalescer that + /// reassembles post-filter batches back to the target batch size. + pub(crate) fn filtered_schema(&self) -> &SchemaRef { + &self.filtered_schema + } + + /// Whether this file has a post-scan filter. Used by the opener to decide + /// whether a decoder-local LIMIT is safe (it is not, because the filter + /// can reject rows after the decoder counts them). + pub(crate) fn has_post_scan_filter(&self) -> bool { + self.post_scan_filter.is_some() + } + /// Map a decoded batch onto the scan's output schema. /// /// Applies the [`Projector`] and, when the projector's output schema @@ -159,3 +481,139 @@ impl DecoderProjection { )?) } } + +#[cfg(test)] +mod tests { + use super::*; + + use arrow::array::Int32Array; + use arrow::datatypes::{DataType, Field, Schema}; + use datafusion_common::ScalarValue; + use datafusion_expr::Operator; + use datafusion_physical_expr::expressions::{BinaryExpr, Column, Literal}; + + fn batch(values: Vec>) -> RecordBatch { + let schema = Arc::new(Schema::new(vec![ + Field::new("a", DataType::Int32, true), + Field::new("b", DataType::Int32, true), + ])); + let a = Int32Array::from(values.clone()); + // `b` mirrors `a` so both conjuncts can be expressed over real data + // while still exercising two separate columns. + let b = Int32Array::from(values); + RecordBatch::try_new(schema, vec![Arc::new(a), Arc::new(b)]).unwrap() + } + + fn gt(column: &str, index: usize, value: i32) -> Arc { + Arc::new(BinaryExpr::new( + Arc::new(Column::new(column, index)), + Operator::Gt, + Arc::new(Literal::new(ScalarValue::Int32(Some(value)))), + )) + } + + fn filter(conjuncts: Vec>) -> PostScanFilter { + PostScanFilter { + conjuncts, + rows_pruned: Count::new(), + rows_matched: Count::new(), + eval_time: Time::new(), + } + } + + /// Apply the selection the way the push-decoder does, so the test asserts + /// on the rows that actually reach the coalescer. + fn survivors(filter: &PostScanFilter, input: RecordBatch) -> Vec> { + match filter.evaluate(input).unwrap() { + PostScanSelection::Empty => Vec::new(), + PostScanSelection::Rows { batch, mask } => { + let batch = match mask { + Some(mask) => filter_record_batch(&batch, &mask).unwrap(), + None => batch, + }; + batch + .column(0) + .as_any() + .downcast_ref::() + .unwrap() + .iter() + .collect() + } + } + } + + #[test] + fn compacts_between_conjuncts_without_losing_rows() { + // 100 rows; `a > 79` keeps 20 (20% — below the compaction threshold, so + // the loop compacts), then `b > 89` keeps 10 of those. + let input = batch((0..100).map(Some).collect()); + let f = filter(vec![gt("a", 0, 79), gt("b", 1, 89)]); + assert_eq!( + survivors(&f, input), + (90..100).map(Some).collect::>() + ); + assert_eq!(f.rows_matched.value(), 10); + assert_eq!(f.rows_pruned.value(), 90); + } + + #[test] + fn conjunct_order_does_not_change_the_result() { + let a = filter(vec![gt("a", 0, 79), gt("b", 1, 89)]); + let b = filter(vec![gt("b", 1, 89), gt("a", 0, 79)]); + let expected = (90..100).map(Some).collect::>(); + assert_eq!(survivors(&a, batch((0..100).map(Some).collect())), expected); + assert_eq!(survivors(&b, batch((0..100).map(Some).collect())), expected); + } + + /// A `NULL` predicate result drops the row, matching `filter_record_batch` + /// and `FilterExec`. The null must be dropped even when it is the *first* + /// conjunct that produced it and a later conjunct would have said `true`. + #[test] + fn null_predicate_results_drop_the_row() { + let values: Vec> = (0..100) + .map(|i| if i == 95 { None } else { Some(i) }) + .collect(); + let f = filter(vec![gt("a", 0, 79), gt("b", 1, 89)]); + let expected: Vec> = + (90..100).filter(|i| *i != 95).map(Some).collect(); + assert_eq!(survivors(&f, batch(values)), expected); + assert_eq!(f.rows_matched.value(), 9); + assert_eq!(f.rows_pruned.value(), 91); + } + + #[test] + fn all_rows_rejected_reports_empty() { + let input = batch((0..100).map(Some).collect()); + let f = filter(vec![gt("a", 0, 500), gt("b", 1, 0)]); + assert!(matches!( + f.evaluate(input).unwrap(), + PostScanSelection::Empty + )); + assert_eq!(f.rows_matched.value(), 0); + assert_eq!(f.rows_pruned.value(), 100); + } + + #[test] + fn all_rows_kept_needs_no_mask() { + let input = batch((0..100).map(Some).collect()); + let f = filter(vec![gt("a", 0, -1), gt("b", 1, -1)]); + match f.evaluate(input).unwrap() { + PostScanSelection::Rows { batch, mask } => { + assert!(mask.is_none(), "an all-true predicate needs no mask"); + assert_eq!(batch.num_rows(), 100); + } + PostScanSelection::Empty => panic!("expected every row to survive"), + } + assert_eq!(f.rows_matched.value(), 100); + assert_eq!(f.rows_pruned.value(), 0); + } + + #[test] + fn empty_batch_is_passed_through() { + let input = batch(Vec::new()); + let f = filter(vec![gt("a", 0, 79)]); + assert_eq!(survivors(&f, input), Vec::>::new()); + assert_eq!(f.rows_matched.value(), 0); + assert_eq!(f.rows_pruned.value(), 0); + } +} diff --git a/datafusion/datasource-parquet/src/metrics.rs b/datafusion/datasource-parquet/src/metrics.rs index b82bd54839f4f..4abadd2c4e13a 100644 --- a/datafusion/datasource-parquet/src/metrics.rs +++ b/datafusion/datasource-parquet/src/metrics.rs @@ -94,6 +94,14 @@ pub struct ParquetFileMetrics { pub pushdown_rows_matched: Count, /// Total time spent evaluating row-level pushdown filters pub row_pushdown_eval_time: Time, + /// Total rows filtered out by the in-scan post-scan filter + /// (predicate conjuncts that could not be applied as a parquet + /// `RowFilter` and were instead evaluated on decoded batches). + pub post_scan_rows_pruned: Count, + /// Total rows that passed the in-scan post-scan filter. + pub post_scan_rows_matched: Count, + /// Total time spent evaluating the in-scan post-scan filter. + pub post_scan_filter_eval_time: Time, /// Total time spent evaluating row group-level statistics filters pub statistics_eval_time: Time, /// Total time spent evaluating row group Bloom Filters @@ -267,6 +275,19 @@ impl ParquetFileMetrics { let row_pushdown_eval_time = builder .clone() .subset_time("row_pushdown_eval_time", partition); + + let post_scan_rows_pruned = builder + .clone() + .with_category(MetricCategory::Rows) + .counter("post_scan_rows_pruned", partition); + let post_scan_rows_matched = builder + .clone() + .with_category(MetricCategory::Rows) + .counter("post_scan_rows_matched", partition); + let post_scan_filter_eval_time = builder + .clone() + .subset_time("post_scan_filter_eval_time", partition); + let statistics_eval_time = builder .clone() .subset_time("statistics_eval_time", partition); @@ -307,6 +328,9 @@ impl ParquetFileMetrics { pushdown_rows_pruned, pushdown_rows_matched, row_pushdown_eval_time, + post_scan_rows_pruned, + post_scan_rows_matched, + post_scan_filter_eval_time, statistics_eval_time, bloom_filter_eval_time, page_index_rows_pruned, diff --git a/datafusion/datasource-parquet/src/opener/mod.rs b/datafusion/datasource-parquet/src/opener/mod.rs index 1f4a09a58b2d6..4b97fcd66e657 100644 --- a/datafusion/datasource-parquet/src/opener/mod.rs +++ b/datafusion/datasource-parquet/src/opener/mod.rs @@ -38,6 +38,7 @@ use crate::{ apply_file_schema_type_coercions, }; use arrow::array::RecordBatch; +use arrow::compute::BatchCoalescer; use arrow::datatypes::DataType; use datafusion_datasource::morsel::{Morsel, MorselPlan, MorselPlanner, Morselizer}; use datafusion_physical_expr::projection::ProjectionExprs; @@ -540,37 +541,63 @@ impl DecoderReadPlans { prepared: &PreparedParquetOpen, metadata: &ArrowReaderMetadata, ) -> Result { - // Build the decoder projection (mask + per-batch transform) in a - // single call. Encapsulating it behind `DecoderProjection` keeps the - // opener's orchestration body focused on filter / decoder / stream - // wiring. The file-column projection excludes virtual columns and - // respects nested field projections. + // --------------------------------------------------------------- + // Filter placement + // + // The scan accepts every pushable filter (the parent `FilterExec` + // is gone), so the predicate must be applied here. Each conjunct is + // routed to one of two places: + // + // * the parquet `RowFilter` (during decode, only when + // `pushdown_filters = true`), or + // * the in-scan post-scan filter (otherwise, plus any conjunct the + // `RowFilter` machinery cannot evaluate on this file — the rejected + // conjuncts returned by `RowFilterContext::try_new`). + // + // Either way every conjunct is applied; nothing is silently dropped. + // --------------------------------------------------------------- + let (row_filter_context, post_scan_conjuncts) = + match (prepared.pushdown_filters, prepared.predicate.as_ref()) { + // Pushdown enabled: precompute the candidate list once per file. + // Both the initial `RowFilter` and any per-RG rebuilds (via + // `RowFilterContext::build_row_filter`) reuse it, so tree walks + // (`reassign_expr_columns`) and column resolution only run once — + // not once per row group. Only what the `RowFilter` could not + // place falls through to post-scan. + (true, Some(predicate)) => RowFilterContext::try_new( + predicate, + &prepared.physical_file_schema, + metadata.metadata(), + prepared.reorder_predicates, + prepared.file_metrics.clone(), + prepared.max_predicate_cache_size, + ), + // Pushdown disabled: the whole predicate runs post-scan (in-scan + // equivalent of a `FilterExec`). + (false, Some(predicate)) => ( + None, + datafusion_physical_expr::split_conjunction(predicate) + .into_iter() + .cloned() + .collect(), + ), + (_, None) => (None, Vec::new()), + }; + + // Build the decoder projection (mask + per-batch transform + optional + // post-scan filter) in a single call. Encapsulating it behind + // `DecoderProjection` keeps the opener's orchestration body focused on + // filter / decoder / stream wiring. The file-column projection + // excludes virtual columns and respects nested field projections. let projection = DecoderProjection::try_new( &prepared.projection, + &post_scan_conjuncts, &prepared.physical_file_schema, metadata.parquet_schema(), &prepared.output_schema, prepared.virtual_state.as_deref(), + &prepared.file_metrics, )?; - let pushdown_predicate = prepared - .pushdown_filters - .then_some(prepared.predicate.as_ref()) - .flatten(); - // Precompute the candidate list once per file. Both the initial - // `RowFilter` and any per-RG rebuilds (via - // `RowFilterContext::build_row_filter`) reuse it, so tree walks - // (`reassign_expr_columns`) and column resolution only run once — - // not once per row group. - let row_filter_context = pushdown_predicate.and_then(|predicate| { - RowFilterContext::try_new( - predicate, - &prepared.physical_file_schema, - metadata.metadata(), - prepared.reorder_predicates, - prepared.file_metrics.clone(), - prepared.max_predicate_cache_size, - ) - }); Ok(Self { projection, row_filter_context, @@ -1711,6 +1738,14 @@ impl RowGroupsPrunedParquetOpen { prepared.partition_index, &prepared.file_name, ); + // Decoder-local LIMIT is only safe when no post-decode work can reject + // rows. A post-scan filter can — so when one is present the limit is + // enforced at the stream level via `remaining_limit` and kept out of + // the decoder; otherwise it is pushed into the decoder. + let has_post_scan_filter = decoder_projection.has_post_scan_filter(); + let decoder_limit = prepared.limit.filter(|_| !has_post_scan_filter); + let remaining_limit = prepared.limit.filter(|_| has_post_scan_filter); + let InitialDecoderState { decoder, rg_plan, @@ -1730,7 +1765,7 @@ impl RowGroupsPrunedParquetOpen { batch_size: prepared.batch_size, arrow_reader_metrics: &arrow_reader_metrics, force_filter_selections: prepared.force_filter_selections, - decoder_limit: prepared.limit, + decoder_limit, }; let prepared_access_plan = prepare_access_plan(access_plan)?; @@ -1870,6 +1905,10 @@ impl RowGroupsPrunedParquetOpen { .file_metrics .row_groups_pruned_dynamic_filter .clone(); + + // Captured before `decoder_projection` is moved into the stream state. + let filtered_schema = Arc::clone(decoder_projection.filtered_schema()); + let stream = PushDecoderStreamState { decoder: Some(decoder), active_reader: None, @@ -1886,6 +1925,12 @@ impl RowGroupsPrunedParquetOpen { filter_installed, row_filter_skipped_fully_matched, byte_progress, + remaining_limit, + // A post-scan filter can leave only a few rows per decoded batch; + // reassemble them so the operator above sees full-size batches. + batch_coalescer: has_post_scan_filter + .then(|| BatchCoalescer::new(filtered_schema, prepared.batch_size)), + flushed: false, } .into_stream(); @@ -3164,14 +3209,17 @@ mod test { .build() }; - // A filter on "a" should not exclude any rows even if it matches the data + // A filter on "a" cannot be excluded by file-level stats (no stats on + // column 0). The scan now accepts the filter and applies it post-scan + // (in-scan equivalent of `FilterExec`), so only the matching row + // survives. let expr = col("a").eq(lit(1)); let predicate = logical2physical(&expr, &schema); let opener = make_opener(predicate); let stream = open_file(&opener, file.clone()).await.unwrap(); let (num_batches, num_rows) = count_batches_and_rows(stream).await; assert_eq!(num_batches, 1); - assert_eq!(num_rows, 3); + assert_eq!(num_rows, 1); // A filter on `b = 5.0` should exclude all rows let expr = col("b").eq(lit(ScalarValue::Float32(Some(5.0)))); @@ -3292,14 +3340,16 @@ mod test { .build() }; - // Filter should match the partition value and file statistics + // Filter should match the partition value and file statistics (i.e. no + // file-level pruning). The scan now accepts the filter and applies it + // post-scan, leaving only the single row where `b = 1.0`. let expr = col("part").eq(lit(1)).and(col("b").eq(lit(1.0))); let predicate = logical2physical(&expr, &table_schema); let opener = make_opener(predicate); let stream = open_file(&opener, file.clone()).await.unwrap(); let (num_batches, num_rows) = count_batches_and_rows(stream).await; assert_eq!(num_batches, 1); - assert_eq!(num_rows, 3); + assert_eq!(num_rows, 1); // Should prune based on partition value but not file statistics let expr = col("part").eq(lit(2)).and(col("b").eq(lit(1.0))); @@ -3816,10 +3866,12 @@ mod test { Ok(count_batches_and_rows(stream).await) }; + // The scan accepts the `a = 1` filter and applies it (RowFilter or + // post-scan), so only the matching row survives (data is a=[1, 2, 2]). let (num_batches, num_rows) = query_file(schema.clone()).await.expect("query_file"); assert_eq!(num_batches, 1); - assert_eq!(num_rows, 3); + assert_eq!(num_rows, 1); let mismatching_schema = Schema::new(vec![ Field::new("a", DataType::Int32, true), @@ -4560,13 +4612,17 @@ mod test { ) .await; + // The scan now always applies the predicate (RowFilter or post-scan), + // so both paths return the same matching rows. The page index only + // affects IO — it decides whether the 90 non-matching rows are + // physically read before being rejected. assert_eq!( rows_with_page_index, 10, "page index should prune 9 of 10 pages" ); assert_eq!( - rows_without_page_index, 100, - "without page index all rows are returned" + rows_without_page_index, 10, + "without page index the post-scan filter still rejects non-matching rows" ); } @@ -5229,6 +5285,167 @@ mod test { assert_eq!(values, vec![7, 4, 5, 6, 3]); } + /// A selective post-scan filter must not fragment the output stream. + /// + /// The decoder hands back one batch per `batch_size` rows; a predicate + /// matching a single row in each of them used to be emitted as one + /// sliver-sized batch per decoded batch, leaving every operator above the + /// scan to pay per-batch overhead on batches of one row. `FilterExec` + /// coalesces for exactly this reason, and the in-scan filter now does too. + #[tokio::test] + async fn selective_post_scan_filter_coalesces_output_batches() { + use arrow::array::Int32Array; + + let store = Arc::new(InMemory::new()) as Arc; + let schema = + Arc::new(Schema::new(vec![Field::new("id", DataType::Int32, false)])); + + // 8 write batches of 1000 rows. The default test `batch_size` is 1024, + // so the decoder yields several batches regardless of how the writer + // chunked the data. + let batches: Vec = (0..8) + .map(|chunk: i32| { + let base = chunk * 1000; + RecordBatch::try_new( + Arc::clone(&schema), + vec![Arc::new(Int32Array::from( + (base..base + 1000).collect::>(), + ))], + ) + .unwrap() + }) + .collect(); + + let data_size = + write_parquet_batches(Arc::clone(&store), "coalesce.parquet", batches, None) + .await; + let file = PartitionedFile::new("coalesce.parquet".to_string(), data_size as u64); + + // One match per ~1024-row decoded batch, spread across the file, so an + // uncoalesced stream would emit one batch per match. + let matches = [0i32, 1500, 3000, 4500, 6000, 7500]; + let predicate = logical2physical( + &matches + .iter() + .map(|v| col("id").eq(lit(*v))) + .reduce(|acc, e| acc.or(e)) + .unwrap(), + &schema, + ); + + let morselizer = ParquetMorselizerBuilder::new() + .with_store(Arc::clone(&store)) + .with_schema(Arc::clone(&schema)) + .with_predicate(predicate) + // pushdown_filters=false routes the whole predicate to the + // post-scan filter — the path this test is about. + .with_pushdown_filters(false) + .build(); + + let stream = open_file(&morselizer, file).await.unwrap(); + let (num_batches, num_rows) = count_batches_and_rows(stream).await; + + assert_eq!(num_rows, matches.len(), "every match must survive"); + // All survivors fit in one target-size batch. Without coalescing this + // was one batch per match. + assert_eq!( + num_batches, 1, + "expected the survivors to be coalesced into a single batch, got \ + {num_batches} batches for {num_rows} rows" + ); + } + + /// End-to-end regression test for the "drop-on-floor" bug fixed by + /// `build_row_filter` now returning rejected conjuncts and the opener + /// routing them to the post-scan filter. + /// + /// Setup: a parquet file with a struct column where some rows have a NULL + /// struct. Predicate `s IS NOT NULL` is set on the source with + /// `pushdown_filters = true`. `ParquetSource::try_pushdown_filters` would + /// have already removed the parent `FilterExec` (the conjunct is pushable + /// at table schema level). Inside `build_row_filter`, + /// `FilterCandidateBuilder::build` rejects the whole-struct reference as + /// non-primitive. + /// + /// Before the fix the rejected conjunct was silently dropped, leaving the + /// scan with no `RowFilter` and no post-scan filter, so every row was + /// returned — i.e. the predicate was relaxed and the query returned wrong + /// results. After the fix the conjunct is surfaced and applied as a + /// post-scan filter, so only the rows with a non-null struct survive. + #[tokio::test] + async fn rejected_struct_conjunct_runs_post_scan_not_dropped() { + use arrow::array::{Int32Array, StringArray, StructArray}; + use arrow::buffer::NullBuffer; + use arrow::datatypes::Fields; + + let store = Arc::new(InMemory::new()) as Arc; + + // Schema: id (Int32), s (Struct{value: Int32, label: Utf8}). + let struct_fields: Fields = vec![ + Arc::new(Field::new("value", DataType::Int32, true)), + Arc::new(Field::new("label", DataType::Utf8, true)), + ] + .into(); + let schema = Arc::new(Schema::new(vec![ + Field::new("id", DataType::Int32, false), + Field::new("s", DataType::Struct(struct_fields.clone()), true), + ])); + + // Data: rows 0 and 2 have a non-null struct, row 1 is null. + let batch = RecordBatch::try_new( + Arc::clone(&schema), + vec![ + Arc::new(Int32Array::from(vec![1, 2, 3])), + Arc::new(StructArray::new( + struct_fields, + vec![ + Arc::new(Int32Array::from(vec![Some(10), None, Some(30)])) as _, + Arc::new(StringArray::from(vec![Some("a"), None, Some("c")])) + as _, + ], + Some(NullBuffer::from(vec![true, false, true])), + )), + ], + ) + .unwrap(); + + let data_size = write_parquet_batches( + Arc::clone(&store), + "rejected.parquet", + vec![batch], + None, + ) + .await; + + let file = PartitionedFile::new("rejected.parquet".to_string(), data_size as u64); + + // `s IS NOT NULL` references a whole struct, which `PushdownChecker` + // flags as non-primitive — `FilterCandidateBuilder::build` returns + // `Ok(None)` and the conjunct lands in `rejected`. + let predicate = logical2physical(&col("s").is_not_null(), &schema); + + let morselizer = ParquetMorselizerBuilder::new() + .with_store(Arc::clone(&store)) + .with_schema(Arc::clone(&schema)) + .with_predicate(predicate) + // The RowFilter path: emulates the post-`try_pushdown_filters` + // state where the parent `FilterExec` has already been removed + // and the scan owns the conjunct. + .with_pushdown_filters(true) + .build(); + + let stream = open_file(&morselizer, file).await.unwrap(); + let (_, rows) = count_batches_and_rows(stream).await; + + // 2 rows have a non-null struct. Before the fix this returned 3 + // (the conjunct was silently dropped). + assert_eq!( + rows, 2, + "expected 2 rows with non-null struct; the rejected conjunct must \ + be applied post-scan, not silently dropped" + ); + } + /// Helpers for tests that exercise parquet virtual columns /// (e.g. `row_number`) plumbed through `TableSchema`/`ParquetOpener`. mod virtual_columns { @@ -5694,9 +5911,12 @@ mod test { #[tokio::test] async fn test_row_index_predicate_allowed_when_pushdown_disabled() { - // Guards the `pushdown_filters=false` path: the predicate is only - // used for stats pruning (a no-op for row_number) and must not - // trip the virtual-column check. + // Guards the `pushdown_filters=false` path with a virtual-column + // predicate set directly on the opener: it must not trip the + // virtual-column check in the read-plan mask. With always-accept + // semantics the predicate runs as an in-scan post-scan filter + // (evaluated against the reader-appended `row_number` column), so + // only the single matching row survives. let store = Arc::new(InMemory::new()) as Arc; let expr = col("row_number").eq(lit(2i64)); let (morselizer, file) = @@ -5706,7 +5926,7 @@ mod test { let stream = open_file(&morselizer, file).await.unwrap(); let (_batches, rows) = count_batches_and_rows(stream).await; - assert_eq!(rows, 5); + assert_eq!(rows, 1); } } } diff --git a/datafusion/datasource-parquet/src/push_decoder.rs b/datafusion/datasource-parquet/src/push_decoder.rs index e5c03df4860fa..a811212a3dfe3 100644 --- a/datafusion/datasource-parquet/src/push_decoder.rs +++ b/datafusion/datasource-parquet/src/push_decoder.rs @@ -39,6 +39,7 @@ use std::collections::VecDeque; use std::sync::Arc; use arrow::array::RecordBatch; +use arrow::compute::BatchCoalescer; use arrow::datatypes::SchemaRef; use futures::StreamExt; use futures::stream::BoxStream; @@ -57,12 +58,13 @@ use parquet::file::metadata::ParquetMetaData; use datafusion_common::{DataFusionError, Result, internal_err}; use datafusion_physical_expr::expressions::DynamicFilterTracking; +use datafusion_physical_expr::split_conjunction; use datafusion_physical_expr_common::physical_expr::PhysicalExpr; use datafusion_physical_plan::metrics::{BaselineMetrics, Count, Gauge}; use datafusion_pruning::{PruningPredicate, PruningPredicateBuilder}; use crate::ParquetFileMetrics; -use crate::decoder_projection::DecoderProjection; +use crate::decoder_projection::{DecoderProjection, PostScanSelection}; use crate::metrics::{ByteProgress, RowFilterSkippedFullyMatchedMetric}; use crate::row_filter::{ PrebuiltRowFilterCandidate, prebuild_row_filter_candidates, row_filter_from_prebuilt, @@ -320,6 +322,26 @@ pub(crate) struct PushDecoderStreamState { /// group at a time as they are decoded or skipped, and topped up to the /// full range when the stream is dropped. pub(crate) byte_progress: ByteProgress, + /// Stream-level remaining row limit, enforced *after* the post-scan + /// filter. `Some` only when the file has a post-scan filter (which makes + /// the decoder-local `with_limit` unsafe — the decoder would short-circuit + /// before the filter rejects enough rows); `None` otherwise, in which case + /// the limit is enforced inside the decoder via `DecoderBuilderConfig`. + pub(crate) remaining_limit: Option, + /// Reassembles post-filter batches back to the target batch size. + /// + /// `Some` exactly when the file has a post-scan filter. A selective + /// predicate leaves only a handful of rows per decoded batch (TPC-H q3 + /// yields ~41 rows from each 8192-row batch), and without this every one + /// of those slivers would be handed to the operator above as its own + /// batch. `FilterExec` coalesces for the same reason. `None` when there is + /// no post-scan filter: decoder batches are already full size, so routing + /// them through the coalescer would only add a copy. + pub(crate) batch_coalescer: Option, + /// Set once [`BatchCoalescer::finish_buffered_batch`] has been called, so + /// end-of-input flushing happens exactly once no matter which terminal + /// path reached it. + pub(crate) flushed: bool, } /// A reusable, `Arc`-shared list of prebuilt row-filter candidates. @@ -359,8 +381,14 @@ pub(crate) struct RowFilterContext { impl RowFilterContext { /// Precompute the candidate list from the raw predicate + file schema + - /// metadata. Returns `None` when the predicate has no push-downable - /// conjuncts (mirrors the file-open path behaviour). + /// metadata. + /// + /// The first element is `None` when the predicate has no push-downable + /// conjuncts (mirrors the file-open path behaviour). The second element + /// holds the conjuncts the `RowFilter` machinery could not place on this + /// file. The caller must evaluate them elsewhere (post-scan), otherwise + /// the predicate is silently relaxed. On a whole-file build error every + /// conjunct is returned in that list rather than being dropped. pub(crate) fn try_new( predicate: &Arc, physical_file_schema: &SchemaRef, @@ -368,22 +396,31 @@ impl RowFilterContext { reorder_predicates: bool, file_metrics: ParquetFileMetrics, max_predicate_cache_size: Option, - ) -> Option { + ) -> (Option, Vec>) { match prebuild_row_filter_candidates( predicate, physical_file_schema, file_metadata.as_ref(), ) { - Ok(Some(prebuilt)) => Some(Self { - prebuilt: PrebuiltRowFilterCandidateList::new(prebuilt), - reorder_predicates, - file_metrics, - max_predicate_cache_size, - }), - Ok(None) => None, + Ok((prebuilt, rejected)) => { + let context = prebuilt.map(|prebuilt| Self { + prebuilt: PrebuiltRowFilterCandidateList::new(prebuilt), + reorder_predicates, + file_metrics, + max_predicate_cache_size, + }); + (context, rejected) + } Err(e) => { - debug!("Ignoring error prebuilding row filter candidates: {e}"); - None + // Whole-file build failure: route every conjunct post-scan + // rather than silently dropping the predicate. + debug!( + "Ignoring error prebuilding row filter candidates: {e}; \ + all conjuncts will be evaluated post-scan" + ); + let rejected = + split_conjunction(predicate).into_iter().cloned().collect(); + (None, rejected) } } } @@ -445,11 +482,72 @@ impl PushDecoderStreamState { let elapsed_compute = self.baseline_metrics.elapsed_compute().clone(); let mut timer = elapsed_compute.timer(); loop { + // Hand out anything the coalescer has already assembled into a + // full-size batch before doing more decoding work. + if self + .batch_coalescer + .as_ref() + .is_some_and(BatchCoalescer::has_completed_batch) + { + return self.emit_completed(); + } + + // The stream-level limit (set only when a post-scan filter made + // the decoder-local limit unsafe) is exhausted — stop. Anything + // still buffered is beyond the limit, so it is dropped rather + // than flushed. + if self.remaining_limit == Some(0) { + return None; + } + // Step 1: drain a batch from the active reader if any. if let Some(reader) = self.active_reader.as_mut() { match reader.next() { Some(Ok(batch)) => { self.copy_arrow_reader_metrics(); + + // Apply the in-scan post-scan filter (if any). The + // decoder's projection mask already covers the + // predicate's columns; the filter's compact-once loop + // needs them for every conjunct, but once it is done + // those the projector does not also read are dropped by + // `narrow` before the residual mask is applied, so we + // never filter a column just to discard it. Survivors go + // into the coalescer rather than straight downstream, so + // a selective predicate does not fragment the stream into + // slivers; the limit and the projection are applied to + // the full-size batches the coalescer hands back. + if let Some(filter) = self.decoder_projection.post_scan_filter() { + let pushed = filter.evaluate(batch).and_then(|selection| { + let (batch, mask) = match selection { + PostScanSelection::Empty => return Ok(()), + PostScanSelection::Rows { batch, mask } => { + (batch, mask) + } + }; + let narrowed = self.decoder_projection.narrow(batch)?; + let coalescer = self + .batch_coalescer + .as_mut() + .expect("coalescer present with a post-scan filter"); + match mask { + Some(mask) => { + coalescer + .push_batch_with_filter(narrowed, &mask)?; + } + None => coalescer.push_batch(narrowed)?, + } + Ok(()) + }); + if let Err(e) = pushed { + return Some((Err(e), self)); + } + continue; + } + + // No post-scan filter: the decoder's batches are + // already the right shape, so project and yield + // directly. The limit was pushed into the decoder. let result = self.project_batch(&batch); return Some((result, self)); } @@ -502,7 +600,7 @@ impl PushDecoderStreamState { if at_boundary && !self.rg_plan.is_empty() { let pruned_count = self.prune_boundary_row_groups(); match self.rebuild_decoder_at_boundary(pruned_count) { - Ok(true) => return None, + Ok(true) => return self.finish(), Ok(false) => {} Err(e) => return Some((Err(e), self)), } @@ -549,7 +647,7 @@ impl PushDecoderStreamState { } self.active_reader = Some(reader); } - Ok(DecodeResult::Finished) => return None, + Ok(DecodeResult::Finished) => return self.finish(), Err(e) => { return Some((Err(DataFusionError::from(e)), self)); } @@ -716,6 +814,52 @@ impl PushDecoderStreamState { } } + /// Pop one assembled batch from the coalescer, apply the stream-level + /// limit, and project it onto the scan's output schema. + /// + /// Returns `None` only when the coalescer has nothing left, which ends the + /// stream. Called from within [`Self::transition`], whose + /// `elapsed_compute` timer covers this work. + fn emit_completed(mut self) -> Option<(Result, Self)> { + let batch = self.batch_coalescer.as_mut()?.next_completed_batch()?; + + // Enforce the stream-level limit here rather than in the decoder: the + // post-scan filter rejects rows the decoder has already counted, so a + // decoder-local limit would stop short. + let batch = if let Some(remaining) = self.remaining_limit { + if batch.num_rows() > remaining { + self.remaining_limit = Some(0); + batch.slice(0, remaining) + } else { + self.remaining_limit = Some(remaining - batch.num_rows()); + batch + } + } else { + batch + }; + let result = self.project_batch(&batch); + Some((result, self)) + } + + /// End of input: flush the partial batch the coalescer is still holding, + /// then drain it one batch at a time. Idempotent — every terminal path in + /// `transition` routes through here, but the flush happens once. + fn finish(mut self) -> Option<(Result, Self)> { + self.batch_coalescer.as_ref()?; + if !self.flushed { + self.flushed = true; + if let Err(e) = self + .batch_coalescer + .as_mut() + .expect("coalescer checked present") + .finish_buffered_batch() + { + return Some((Err(DataFusionError::from(e)), self)); + } + } + self.emit_completed() + } + fn project_batch(&self, batch: &RecordBatch) -> Result { self.decoder_projection.map(batch) } diff --git a/datafusion/datasource-parquet/src/row_filter.rs b/datafusion/datasource-parquet/src/row_filter.rs index 463f9f84daf5a..96c304a38f661 100644 --- a/datafusion/datasource-parquet/src/row_filter.rs +++ b/datafusion/datasource-parquet/src/row_filter.rs @@ -412,36 +412,41 @@ fn size_of_columns(columns: &[usize], metadata: &ParquetMetaData) -> Result, file_schema: &SchemaRef, metadata: &ParquetMetaData, reorder_predicates: bool, file_metrics: &ParquetFileMetrics, -) -> Result> { +) -> Result<(Option, Vec>)> { // Implemented on top of the prebuild split so there is a single place // that splits conjuncts, orders candidates, and wires metrics — callers // that build once per file go through the same code as the per-row-group // rebuild path in `RowFilterContext`. - let Some(prebuilt) = prebuild_row_filter_candidates(expr, file_schema, metadata)? - else { - return Ok(None); - }; - Ok(Some(row_filter_from_prebuilt( - &prebuilt, - reorder_predicates, - file_metrics, - ))) + let (prebuilt, rejected) = + prebuild_row_filter_candidates(expr, file_schema, metadata)?; + let row_filter = prebuilt.map(|prebuilt| { + row_filter_from_prebuilt(&prebuilt, reorder_predicates, file_metrics) + }); + Ok((row_filter, rejected)) } /// A precomputed [`FilterCandidate`] with its expression column-reassigned to @@ -483,29 +488,40 @@ impl PrebuiltRowFilterCandidate { /// walks and `Arc` allocations that showed up as top hot spots /// in TPCH profiles. /// -/// Returns `Ok(None)` when the predicate has no push-downable conjuncts, in -/// which case callers should skip installing a `RowFilter` entirely. +/// The first element is `None` when the predicate has no push-downable +/// conjuncts, in which case callers should skip installing a `RowFilter` +/// entirely. The second element holds the conjuncts that cannot be evaluated +/// as an `ArrowPredicate` on this file; see [`build_row_filter`] for why the +/// caller must apply them elsewhere. +#[expect(clippy::type_complexity)] pub(crate) fn prebuild_row_filter_candidates( expr: &Arc, file_schema: &SchemaRef, metadata: &ParquetMetaData, -) -> Result>> { +) -> Result<( + Option>, + Vec>, +)> { // Split into conjuncts: // `a = 1 AND b = 2 AND c = 3` -> [`a = 1`, `b = 2`, `c = 3`] let predicates = split_conjunction(expr); - let candidates: Vec = predicates - .into_iter() - .map(|expr| { - FilterCandidateBuilder::new(Arc::clone(expr), Arc::clone(file_schema)) - .build(metadata) - }) - .collect::, _>>()? - .into_iter() - .flatten() - .collect(); + + // Partition conjuncts into those that can be evaluated as ArrowPredicates + // and those that cannot. Rejected conjuncts are returned to the caller so + // they are never silently dropped. + let mut candidates: Vec = Vec::with_capacity(predicates.len()); + let mut rejected: Vec> = Vec::new(); + for predicate in predicates { + match FilterCandidateBuilder::new(Arc::clone(predicate), Arc::clone(file_schema)) + .build(metadata)? + { + Some(candidate) => candidates.push(candidate), + None => rejected.push(Arc::clone(predicate)), + } + } if candidates.is_empty() { - return Ok(None); + return Ok((None, rejected)); } let prebuilt: Vec = candidates @@ -523,7 +539,7 @@ pub(crate) fn prebuild_row_filter_candidates( }) .collect::>>()?; - Ok(Some(prebuilt)) + Ok((Some(prebuilt), rejected)) } /// Wrap a list of prebuilt candidates into a fresh [`RowFilter`], assigning @@ -934,10 +950,14 @@ mod test { let file_metrics = ParquetFileMetrics::new(0, &format!("{func_name}.parquet"), &metrics); - let row_filter = + let (row_filter, rejected) = build_row_filter(&expr, &file_schema, &metadata, false, &file_metrics) - .expect("building row filter") - .expect("row filter should exist"); + .expect("building row filter"); + assert!( + rejected.is_empty(), + "expected no rejected conjuncts, got {rejected:?}" + ); + let row_filter = row_filter.expect("row filter should exist"); let reader = parquet_reader_builder .with_row_filter(row_filter) @@ -1648,10 +1668,14 @@ mod test { let metrics = ExecutionPlanMetricsSet::new(); let file_metrics = ParquetFileMetrics::new(0, "struct_e2e.parquet", &metrics); - let row_filter = + let (row_filter, rejected) = build_row_filter(&expr, &file_schema, &metadata, false, &file_metrics) - .expect("building row filter") - .expect("row filter should exist"); + .expect("building row filter"); + assert!( + rejected.is_empty(), + "expected no rejected conjuncts, got {rejected:?}" + ); + let row_filter = row_filter.expect("row filter should exist"); let reader = parquet_reader_builder .with_row_filter(row_filter) @@ -2104,10 +2128,14 @@ mod test { let file_metrics = ParquetFileMetrics::new(0, "shared_prefix_e2e.parquet", &metrics); - let row_filter = + let (row_filter, rejected) = build_row_filter(&expr, &file_schema, &metadata, false, &file_metrics) - .expect("building row filter") - .expect("row filter should exist"); + .expect("building row filter"); + assert!( + rejected.is_empty(), + "expected no rejected conjuncts, got {rejected:?}" + ); + let row_filter = row_filter.expect("row filter should exist"); let reader = parquet_reader_builder .with_row_filter(row_filter) @@ -2127,4 +2155,85 @@ mod test { assert_eq!(file_metrics.pushdown_rows_pruned.value(), 2); assert_eq!(file_metrics.pushdown_rows_matched.value(), 2); } + + /// Regression test: a predicate `(s IS NOT NULL) AND (id = 1)` mixes a + /// conjunct that the `RowFilter` machinery cannot evaluate (whole-struct + /// reference — [`PushdownChecker`] flags it as non-primitive) with one + /// that it can. `build_row_filter` must return the rejected conjunct in + /// its second tuple element so the caller can re-route it to a post-scan + /// filter; before this was fixed the rejected conjunct was silently + /// dropped on the floor while the parent `FilterExec` had already been + /// removed, relaxing the predicate and returning wrong results. + #[test] + fn build_row_filter_surfaces_rejected_struct_conjunct() { + let struct_fields: Fields = vec![ + Arc::new(Field::new("value", DataType::Int32, false)), + Arc::new(Field::new("label", DataType::Utf8, false)), + ] + .into(); + + let schema = Arc::new(Schema::new(vec![ + Field::new("id", DataType::Int32, false), + Field::new("s", DataType::Struct(struct_fields.clone()), false), + ])); + + let batch = RecordBatch::try_new( + Arc::clone(&schema), + vec![ + Arc::new(Int32Array::from(vec![1, 2, 3])), + Arc::new(StructArray::new( + struct_fields, + vec![ + Arc::new(Int32Array::from(vec![10, 20, 30])) as _, + Arc::new(StringArray::from(vec!["a", "b", "c"])) as _, + ], + None, + )), + ], + ) + .unwrap(); + + let file = NamedTempFile::new().expect("temp file"); + let mut writer = + ArrowWriter::try_new(file.reopen().unwrap(), Arc::clone(&schema), None) + .expect("writer"); + writer.write(&batch).expect("write batch"); + writer.close().expect("close writer"); + + let reader_file = file.reopen().expect("reopen file"); + let parquet_reader_builder = + ParquetRecordBatchReaderBuilder::try_new(reader_file) + .expect("reader builder"); + let metadata = parquet_reader_builder.metadata().clone(); + let file_schema = parquet_reader_builder.schema().clone(); + + // (s IS NOT NULL) AND (id = 1) + // The first conjunct references a whole struct -> RowFilter rejects. + // The second is a plain Int32 equality -> RowFilter accepts. + let predicate_expr = col("s") + .is_not_null() + .and(col("id").eq(Expr::Literal(ScalarValue::Int32(Some(1)), None))); + let expr = logical2physical(&predicate_expr, &file_schema); + + let metrics = ExecutionPlanMetricsSet::new(); + let file_metrics = + ParquetFileMetrics::new(0, "build_row_filter_rejected.parquet", &metrics); + + let (row_filter, rejected) = + build_row_filter(&expr, &file_schema, &metadata, false, &file_metrics) + .expect("building row filter"); + + // The plain id = 1 conjunct produced a RowFilter… + assert!( + row_filter.is_some(), + "id = 1 should have produced a RowFilter" + ); + // …and the struct IS NOT NULL conjunct must be surfaced as rejected, + // never silently dropped. + assert_eq!( + rejected.len(), + 1, + "expected exactly one rejected conjunct (s IS NOT NULL), got {rejected:?}" + ); + } } diff --git a/datafusion/datasource-parquet/src/source.rs b/datafusion/datasource-parquet/src/source.rs index 32fc8c9f3da33..c1e7e47b459e3 100644 --- a/datafusion/datasource-parquet/src/source.rs +++ b/datafusion/datasource-parquet/src/source.rs @@ -896,14 +896,14 @@ impl FileSource for ParquetSource { source.predicate = Some(predicate); source = source.with_pushdown_filters(pushdown_filters); let source = Arc::new(source); - // If pushdown_filters is false we tell our parents that they still have to handle the filters, - // even if we updated the predicate to include the filters (they will only be used for stats pruning). - if !pushdown_filters { - return Ok(FilterPushdownPropagation::with_parent_pushdown_result( - vec![PushedDown::No; filters.len()], - ) - .with_updated_node(source)); - } + // The parquet scan always accepts pushable filters: report each + // pushable filter as accepted (`Yes`) so the parent `FilterExec` is + // removed. The scan now owns these filters and guarantees they are + // applied — as a parquet `RowFilter` when `pushdown_filters` is + // enabled, or as the in-scan post-scan filter otherwise (and for any + // conjunct the `RowFilter` cannot evaluate on a given file). The + // `pushdown_filters` config is preserved because it still controls the + // `RowFilter` vs. post-scan placement downstream. Ok(FilterPushdownPropagation::with_parent_pushdown_result( filters.iter().map(|f| f.discriminant).collect(), ) diff --git a/datafusion/sqllogictest/test_files/clickbench.slt b/datafusion/sqllogictest/test_files/clickbench.slt index 7cb5547383c38..d2ee7f8def96a 100644 --- a/datafusion/sqllogictest/test_files/clickbench.slt +++ b/datafusion/sqllogictest/test_files/clickbench.slt @@ -88,9 +88,8 @@ physical_plan 02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] 03)----CoalescePartitionsExec 04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] -05)--------FilterExec: AdvEngineID@0 != 0, projection=[] -06)----------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -07)------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[AdvEngineID], file_type=parquet, predicate=AdvEngineID@40 != 0, pruning_predicate=AdvEngineID_null_count@2 != row_count@3 AND (AdvEngineID_min@0 != 0 OR 0 != AdvEngineID_max@1), required_guarantees=[AdvEngineID not in (0)] +05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 +06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, file_type=parquet, predicate=AdvEngineID@40 != 0, pruning_predicate=AdvEngineID_null_count@2 != row_count@3 AND (AdvEngineID_min@0 != 0 OR 0 != AdvEngineID_max@1), required_guarantees=[AdvEngineID not in (0)] query I SELECT COUNT(*) FROM hits WHERE "AdvEngineID" <> 0; @@ -218,9 +217,8 @@ physical_plan 04)------AggregateExec: mode=FinalPartitioned, gby=[AdvEngineID@0 as AdvEngineID], aggr=[count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([AdvEngineID@0], 4), input_partitions=4 06)----------AggregateExec: mode=Partial, gby=[AdvEngineID@0 as AdvEngineID], aggr=[count(Int64(1))] -07)------------FilterExec: AdvEngineID@0 != 0 -08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[AdvEngineID], file_type=parquet, predicate=AdvEngineID@40 != 0, pruning_predicate=AdvEngineID_null_count@2 != row_count@3 AND (AdvEngineID_min@0 != 0 OR 0 != AdvEngineID_max@1), required_guarantees=[AdvEngineID not in (0)] +07)------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 +08)--------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[AdvEngineID], file_type=parquet, predicate=AdvEngineID@40 != 0, pruning_predicate=AdvEngineID_null_count@2 != row_count@3 AND (AdvEngineID_min@0 != 0 OR 0 != AdvEngineID_max@1), required_guarantees=[AdvEngineID not in (0)] query II SELECT "AdvEngineID", COUNT(*) FROM hits WHERE "AdvEngineID" <> 0 GROUP BY "AdvEngineID" ORDER BY COUNT(*) DESC; @@ -306,9 +304,8 @@ physical_plan 07)------------AggregateExec: mode=FinalPartitioned, gby=[MobilePhoneModel@0 as MobilePhoneModel, alias1@1 as alias1], aggr=[] 08)--------------RepartitionExec: partitioning=Hash([MobilePhoneModel@0, alias1@1], 4), input_partitions=4 09)----------------AggregateExec: mode=Partial, gby=[MobilePhoneModel@1 as MobilePhoneModel, UserID@0 as alias1], aggr=[] -10)------------------FilterExec: MobilePhoneModel@1 != -11)--------------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -12)----------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[UserID, MobilePhoneModel], file_type=parquet, predicate=MobilePhoneModel@34 != , pruning_predicate=MobilePhoneModel_null_count@2 != row_count@3 AND (MobilePhoneModel_min@0 != OR != MobilePhoneModel_max@1), required_guarantees=[MobilePhoneModel not in ()] +10)------------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 +11)--------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[UserID, MobilePhoneModel], file_type=parquet, predicate=MobilePhoneModel@34 != , pruning_predicate=MobilePhoneModel_null_count@2 != row_count@3 AND (MobilePhoneModel_min@0 != OR != MobilePhoneModel_max@1), required_guarantees=[MobilePhoneModel not in ()] query TI SELECT "MobilePhoneModel", COUNT(DISTINCT "UserID") AS u FROM hits WHERE "MobilePhoneModel" <> '' GROUP BY "MobilePhoneModel" ORDER BY u DESC LIMIT 10; @@ -336,9 +333,8 @@ physical_plan 07)------------AggregateExec: mode=FinalPartitioned, gby=[MobilePhone@0 as MobilePhone, MobilePhoneModel@1 as MobilePhoneModel, alias1@2 as alias1], aggr=[] 08)--------------RepartitionExec: partitioning=Hash([MobilePhone@0, MobilePhoneModel@1, alias1@2], 4), input_partitions=4 09)----------------AggregateExec: mode=Partial, gby=[MobilePhone@1 as MobilePhone, MobilePhoneModel@2 as MobilePhoneModel, UserID@0 as alias1], aggr=[] -10)------------------FilterExec: MobilePhoneModel@2 != -11)--------------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -12)----------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[UserID, MobilePhone, MobilePhoneModel], file_type=parquet, predicate=MobilePhoneModel@34 != , pruning_predicate=MobilePhoneModel_null_count@2 != row_count@3 AND (MobilePhoneModel_min@0 != OR != MobilePhoneModel_max@1), required_guarantees=[MobilePhoneModel not in ()] +10)------------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 +11)--------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[UserID, MobilePhone, MobilePhoneModel], file_type=parquet, predicate=MobilePhoneModel@34 != , pruning_predicate=MobilePhoneModel_null_count@2 != row_count@3 AND (MobilePhoneModel_min@0 != OR != MobilePhoneModel_max@1), required_guarantees=[MobilePhoneModel not in ()] query ITI SELECT "MobilePhone", "MobilePhoneModel", COUNT(DISTINCT "UserID") AS u FROM hits WHERE "MobilePhoneModel" <> '' GROUP BY "MobilePhone", "MobilePhoneModel" ORDER BY u DESC LIMIT 10; @@ -362,9 +358,8 @@ physical_plan 04)------AggregateExec: mode=FinalPartitioned, gby=[SearchPhrase@0 as SearchPhrase], aggr=[count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([SearchPhrase@0], 4), input_partitions=4 06)----------AggregateExec: mode=Partial, gby=[SearchPhrase@0 as SearchPhrase], aggr=[count(Int64(1))] -07)------------FilterExec: SearchPhrase@0 != -08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != , pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] +07)------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 +08)--------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != , pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] query TI SELECT "SearchPhrase", COUNT(*) AS c FROM hits WHERE "SearchPhrase" <> '' GROUP BY "SearchPhrase" ORDER BY c DESC LIMIT 10; @@ -392,9 +387,8 @@ physical_plan 07)------------AggregateExec: mode=FinalPartitioned, gby=[SearchPhrase@0 as SearchPhrase, alias1@1 as alias1], aggr=[] 08)--------------RepartitionExec: partitioning=Hash([SearchPhrase@0, alias1@1], 4), input_partitions=4 09)----------------AggregateExec: mode=Partial, gby=[SearchPhrase@1 as SearchPhrase, UserID@0 as alias1], aggr=[] -10)------------------FilterExec: SearchPhrase@1 != -11)--------------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -12)----------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[UserID, SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != , pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] +10)------------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 +11)--------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[UserID, SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != , pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] query TI SELECT "SearchPhrase", COUNT(DISTINCT "UserID") AS u FROM hits WHERE "SearchPhrase" <> '' GROUP BY "SearchPhrase" ORDER BY u DESC LIMIT 10; @@ -418,9 +412,8 @@ physical_plan 04)------AggregateExec: mode=FinalPartitioned, gby=[SearchEngineID@0 as SearchEngineID, SearchPhrase@1 as SearchPhrase], aggr=[count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([SearchEngineID@0, SearchPhrase@1], 4), input_partitions=4 06)----------AggregateExec: mode=Partial, gby=[SearchEngineID@0 as SearchEngineID, SearchPhrase@1 as SearchPhrase], aggr=[count(Int64(1))] -07)------------FilterExec: SearchPhrase@1 != -08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[SearchEngineID, SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != , pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] +07)------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 +08)--------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[SearchEngineID, SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != , pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] query ITI SELECT "SearchEngineID", "SearchPhrase", COUNT(*) AS c FROM hits WHERE "SearchPhrase" <> '' GROUP BY "SearchEngineID", "SearchPhrase" ORDER BY c DESC LIMIT 10; @@ -550,10 +543,7 @@ logical_plan 01)SubqueryAlias: hits 02)--Filter: hits_raw.UserID = Int64(435090932899640449) 03)----TableScan: hits_raw projection=[UserID], partial_filters=[hits_raw.UserID = Int64(435090932899640449)] -physical_plan -01)FilterExec: UserID@0 = 435090932899640449 -02)--RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[UserID], file_type=parquet, predicate=UserID@9 = 435090932899640449, pruning_predicate=UserID_null_count@2 != row_count@3 AND UserID_min@0 <= 435090932899640449 AND 435090932899640449 <= UserID_max@1, required_guarantees=[UserID in (435090932899640449)] +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[UserID], file_type=parquet, predicate=UserID@9 = 435090932899640449, pruning_predicate=UserID_null_count@2 != row_count@3 AND UserID_min@0 <= 435090932899640449 AND 435090932899640449 <= UserID_max@1, required_guarantees=[UserID in (435090932899640449)] query I SELECT "UserID" FROM hits WHERE "UserID" = 435090932899640449; @@ -575,9 +565,8 @@ physical_plan 02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] 03)----CoalescePartitionsExec 04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] -05)--------FilterExec: URL@0 LIKE %google%, projection=[] -06)----------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -07)------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[URL], file_type=parquet, predicate=URL@13 LIKE %google% +05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 +06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, file_type=parquet, predicate=URL@13 LIKE %google% query I SELECT COUNT(*) FROM hits WHERE "URL" LIKE '%google%'; @@ -602,9 +591,8 @@ physical_plan 04)------AggregateExec: mode=FinalPartitioned, gby=[SearchPhrase@0 as SearchPhrase], aggr=[min(hits.URL), count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([SearchPhrase@0], 4), input_partitions=4 06)----------AggregateExec: mode=Partial, gby=[SearchPhrase@1 as SearchPhrase], aggr=[min(hits.URL), count(Int64(1))] -07)------------FilterExec: SearchPhrase@1 != AND URL@0 LIKE %google% -08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[URL, SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != AND URL@13 LIKE %google%, pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] +07)------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 +08)--------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[URL, SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != AND URL@13 LIKE %google%, pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] query TTI SELECT "SearchPhrase", MIN("URL"), COUNT(*) AS c FROM hits WHERE "URL" LIKE '%google%' AND "SearchPhrase" <> '' GROUP BY "SearchPhrase" ORDER BY c DESC LIMIT 10; @@ -628,9 +616,8 @@ physical_plan 04)------AggregateExec: mode=FinalPartitioned, gby=[SearchPhrase@0 as SearchPhrase], aggr=[min(hits.URL), min(hits.Title), count(Int64(1)), count(DISTINCT hits.UserID)] 05)--------RepartitionExec: partitioning=Hash([SearchPhrase@0], 4), input_partitions=4 06)----------AggregateExec: mode=Partial, gby=[SearchPhrase@3 as SearchPhrase], aggr=[min(hits.URL), min(hits.Title), count(Int64(1)), count(DISTINCT hits.UserID)] -07)------------FilterExec: SearchPhrase@3 != AND Title@0 LIKE %Google% AND URL@2 NOT LIKE %.google.% -08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[Title, UserID, URL, SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != AND Title@2 LIKE %Google% AND URL@13 NOT LIKE %.google.%, pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] +07)------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 +08)--------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[Title, UserID, URL, SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != AND Title@2 LIKE %Google% AND URL@13 NOT LIKE %.google.%, pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] query TTTII SELECT "SearchPhrase", MIN("URL"), MIN("Title"), COUNT(*) AS c, COUNT(DISTINCT "UserID") FROM hits WHERE "Title" LIKE '%Google%' AND "URL" NOT LIKE '%.google.%' AND "SearchPhrase" <> '' GROUP BY "SearchPhrase" ORDER BY c DESC LIMIT 10; @@ -650,9 +637,8 @@ physical_plan 01)SortPreservingMergeExec: [EventTime@4 ASC NULLS LAST], fetch=10 02)--SortExec: TopK(fetch=10), expr=[EventTime@4 ASC NULLS LAST], preserve_partitioning=[true] 03)----ProjectionExec: expr=[WatchID@0 as WatchID, JavaEnable@1 as JavaEnable, Title@2 as Title, GoodEvent@3 as GoodEvent, EventTime@4 as EventTime, CounterID@6 as CounterID, ClientIP@7 as ClientIP, RegionID@8 as RegionID, UserID@9 as UserID, CounterClass@10 as CounterClass, OS@11 as OS, UserAgent@12 as UserAgent, URL@13 as URL, Referer@14 as Referer, IsRefresh@15 as IsRefresh, RefererCategoryID@16 as RefererCategoryID, RefererRegionID@17 as RefererRegionID, URLCategoryID@18 as URLCategoryID, URLRegionID@19 as URLRegionID, ResolutionWidth@20 as ResolutionWidth, ResolutionHeight@21 as ResolutionHeight, ResolutionDepth@22 as ResolutionDepth, FlashMajor@23 as FlashMajor, FlashMinor@24 as FlashMinor, FlashMinor2@25 as FlashMinor2, NetMajor@26 as NetMajor, NetMinor@27 as NetMinor, UserAgentMajor@28 as UserAgentMajor, UserAgentMinor@29 as UserAgentMinor, CookieEnable@30 as CookieEnable, JavascriptEnable@31 as JavascriptEnable, IsMobile@32 as IsMobile, MobilePhone@33 as MobilePhone, MobilePhoneModel@34 as MobilePhoneModel, Params@35 as Params, IPNetworkID@36 as IPNetworkID, TraficSourceID@37 as TraficSourceID, SearchEngineID@38 as SearchEngineID, SearchPhrase@39 as SearchPhrase, AdvEngineID@40 as AdvEngineID, IsArtifical@41 as IsArtifical, WindowClientWidth@42 as WindowClientWidth, WindowClientHeight@43 as WindowClientHeight, ClientTimeZone@44 as ClientTimeZone, ClientEventTime@45 as ClientEventTime, SilverlightVersion1@46 as SilverlightVersion1, SilverlightVersion2@47 as SilverlightVersion2, SilverlightVersion3@48 as SilverlightVersion3, SilverlightVersion4@49 as SilverlightVersion4, PageCharset@50 as PageCharset, CodeVersion@51 as CodeVersion, IsLink@52 as IsLink, IsDownload@53 as IsDownload, IsNotBounce@54 as IsNotBounce, FUniqID@55 as FUniqID, OriginalURL@56 as OriginalURL, HID@57 as HID, IsOldCounter@58 as IsOldCounter, IsEvent@59 as IsEvent, IsParameter@60 as IsParameter, DontCountHits@61 as DontCountHits, WithHash@62 as WithHash, HitColor@63 as HitColor, LocalEventTime@64 as LocalEventTime, Age@65 as Age, Sex@66 as Sex, Income@67 as Income, Interests@68 as Interests, Robotness@69 as Robotness, RemoteIP@70 as RemoteIP, WindowName@71 as WindowName, OpenerName@72 as OpenerName, HistoryLength@73 as HistoryLength, BrowserLanguage@74 as BrowserLanguage, BrowserCountry@75 as BrowserCountry, SocialNetwork@76 as SocialNetwork, SocialAction@77 as SocialAction, HTTPError@78 as HTTPError, SendTiming@79 as SendTiming, DNSTiming@80 as DNSTiming, ConnectTiming@81 as ConnectTiming, ResponseStartTiming@82 as ResponseStartTiming, ResponseEndTiming@83 as ResponseEndTiming, FetchTiming@84 as FetchTiming, SocialSourceNetworkID@85 as SocialSourceNetworkID, SocialSourcePage@86 as SocialSourcePage, ParamPrice@87 as ParamPrice, ParamOrderID@88 as ParamOrderID, ParamCurrency@89 as ParamCurrency, ParamCurrencyID@90 as ParamCurrencyID, OpenstatServiceName@91 as OpenstatServiceName, OpenstatCampaignID@92 as OpenstatCampaignID, OpenstatAdID@93 as OpenstatAdID, OpenstatSourceID@94 as OpenstatSourceID, UTMSource@95 as UTMSource, UTMMedium@96 as UTMMedium, UTMCampaign@97 as UTMCampaign, UTMContent@98 as UTMContent, UTMTerm@99 as UTMTerm, FromTag@100 as FromTag, HasGCLID@101 as HasGCLID, RefererHash@102 as RefererHash, URLHash@103 as URLHash, CLID@104 as CLID, CAST(CAST(EventDate@5 AS Int32) AS Date32) as EventDate] -04)------FilterExec: URL@13 LIKE %google% -05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[WatchID, JavaEnable, Title, GoodEvent, EventTime, EventDate, CounterID, ClientIP, RegionID, UserID, CounterClass, OS, UserAgent, URL, Referer, IsRefresh, RefererCategoryID, RefererRegionID, URLCategoryID, URLRegionID, ResolutionWidth, ResolutionHeight, ResolutionDepth, FlashMajor, FlashMinor, FlashMinor2, NetMajor, NetMinor, UserAgentMajor, UserAgentMinor, CookieEnable, JavascriptEnable, IsMobile, MobilePhone, MobilePhoneModel, Params, IPNetworkID, TraficSourceID, SearchEngineID, SearchPhrase, AdvEngineID, IsArtifical, WindowClientWidth, WindowClientHeight, ClientTimeZone, ClientEventTime, SilverlightVersion1, SilverlightVersion2, SilverlightVersion3, SilverlightVersion4, PageCharset, CodeVersion, IsLink, IsDownload, IsNotBounce, FUniqID, OriginalURL, HID, IsOldCounter, IsEvent, IsParameter, DontCountHits, WithHash, HitColor, LocalEventTime, Age, Sex, Income, Interests, Robotness, RemoteIP, WindowName, OpenerName, HistoryLength, BrowserLanguage, BrowserCountry, SocialNetwork, SocialAction, HTTPError, SendTiming, DNSTiming, ConnectTiming, ResponseStartTiming, ResponseEndTiming, FetchTiming, SocialSourceNetworkID, SocialSourcePage, ParamPrice, ParamOrderID, ParamCurrency, ParamCurrencyID, OpenstatServiceName, OpenstatCampaignID, OpenstatAdID, OpenstatSourceID, UTMSource, UTMMedium, UTMCampaign, UTMContent, UTMTerm, FromTag, HasGCLID, RefererHash, URLHash, CLID], file_type=parquet, predicate=URL@13 LIKE %google% AND DynamicFilter [ empty ], dynamic_rg_pruning=eligible +04)------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 +05)--------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[WatchID, JavaEnable, Title, GoodEvent, EventTime, EventDate, CounterID, ClientIP, RegionID, UserID, CounterClass, OS, UserAgent, URL, Referer, IsRefresh, RefererCategoryID, RefererRegionID, URLCategoryID, URLRegionID, ResolutionWidth, ResolutionHeight, ResolutionDepth, FlashMajor, FlashMinor, FlashMinor2, NetMajor, NetMinor, UserAgentMajor, UserAgentMinor, CookieEnable, JavascriptEnable, IsMobile, MobilePhone, MobilePhoneModel, Params, IPNetworkID, TraficSourceID, SearchEngineID, SearchPhrase, AdvEngineID, IsArtifical, WindowClientWidth, WindowClientHeight, ClientTimeZone, ClientEventTime, SilverlightVersion1, SilverlightVersion2, SilverlightVersion3, SilverlightVersion4, PageCharset, CodeVersion, IsLink, IsDownload, IsNotBounce, FUniqID, OriginalURL, HID, IsOldCounter, IsEvent, IsParameter, DontCountHits, WithHash, HitColor, LocalEventTime, Age, Sex, Income, Interests, Robotness, RemoteIP, WindowName, OpenerName, HistoryLength, BrowserLanguage, BrowserCountry, SocialNetwork, SocialAction, HTTPError, SendTiming, DNSTiming, ConnectTiming, ResponseStartTiming, ResponseEndTiming, FetchTiming, SocialSourceNetworkID, SocialSourcePage, ParamPrice, ParamOrderID, ParamCurrency, ParamCurrencyID, OpenstatServiceName, OpenstatCampaignID, OpenstatAdID, OpenstatSourceID, UTMSource, UTMMedium, UTMCampaign, UTMContent, UTMTerm, FromTag, HasGCLID, RefererHash, URLHash, CLID], file_type=parquet, predicate=URL@13 LIKE %google% AND DynamicFilter [ empty ], sort_order_for_reorder=[EventTime@4 ASC NULLS LAST], dynamic_rg_pruning=eligible query IITIIIIIIIIITTIIIIIIIIIITIIITIIIITTIIITIIIIIIIIIITIIIIITIIIIIITIIIIIIIIIITTTTIIIIIIIITITTITTTTTTTTTTIIIID SELECT * FROM hits WHERE "URL" LIKE '%google%' ORDER BY "EventTime" LIMIT 10; @@ -670,13 +656,9 @@ logical_plan 05)--------Filter: hits_raw.SearchPhrase != Utf8View("") 06)----------TableScan: hits_raw projection=[EventTime, SearchPhrase], partial_filters=[hits_raw.SearchPhrase != Utf8View("")] physical_plan -01)ProjectionExec: expr=[SearchPhrase@0 as SearchPhrase] -02)--SortPreservingMergeExec: [EventTime@1 ASC NULLS LAST], fetch=10 -03)----ProjectionExec: expr=[SearchPhrase@1 as SearchPhrase, EventTime@0 as EventTime] -04)------SortExec: TopK(fetch=10), expr=[EventTime@0 ASC NULLS LAST], preserve_partitioning=[true] -05)--------FilterExec: SearchPhrase@1 != -06)----------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -07)------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[EventTime, SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != AND DynamicFilter [ empty ], dynamic_rg_pruning=eligible, pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] +01)ProjectionExec: expr=[SearchPhrase@1 as SearchPhrase] +02)--SortExec: TopK(fetch=10), expr=[EventTime@0 ASC NULLS LAST], preserve_partitioning=[false] +03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[EventTime, SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != AND DynamicFilter [ empty ], sort_order_for_reorder=[EventTime@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] query T SELECT "SearchPhrase" FROM hits WHERE "SearchPhrase" <> '' ORDER BY "EventTime" LIMIT 10; @@ -692,11 +674,8 @@ logical_plan 03)----Filter: hits_raw.SearchPhrase != Utf8View("") 04)------TableScan: hits_raw projection=[SearchPhrase], partial_filters=[hits_raw.SearchPhrase != Utf8View("")] physical_plan -01)SortPreservingMergeExec: [SearchPhrase@0 ASC NULLS LAST], fetch=10 -02)--SortExec: TopK(fetch=10), expr=[SearchPhrase@0 ASC NULLS LAST], preserve_partitioning=[true] -03)----FilterExec: SearchPhrase@0 != -04)------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -05)--------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != AND DynamicFilter [ empty ], dynamic_rg_pruning=eligible, pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] +01)SortExec: TopK(fetch=10), expr=[SearchPhrase@0 ASC NULLS LAST], preserve_partitioning=[false] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != AND DynamicFilter [ empty ], sort_order_for_reorder=[SearchPhrase@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] query T SELECT "SearchPhrase" FROM hits WHERE "SearchPhrase" <> '' ORDER BY "SearchPhrase" LIMIT 10; @@ -714,13 +693,9 @@ logical_plan 05)--------Filter: hits_raw.SearchPhrase != Utf8View("") 06)----------TableScan: hits_raw projection=[EventTime, SearchPhrase], partial_filters=[hits_raw.SearchPhrase != Utf8View("")] physical_plan -01)ProjectionExec: expr=[SearchPhrase@0 as SearchPhrase] -02)--SortPreservingMergeExec: [EventTime@1 ASC NULLS LAST, SearchPhrase@0 ASC NULLS LAST], fetch=10 -03)----ProjectionExec: expr=[SearchPhrase@1 as SearchPhrase, EventTime@0 as EventTime] -04)------SortExec: TopK(fetch=10), expr=[EventTime@0 ASC NULLS LAST, SearchPhrase@1 ASC NULLS LAST], preserve_partitioning=[true] -05)--------FilterExec: SearchPhrase@1 != -06)----------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -07)------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[EventTime, SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != AND DynamicFilter [ empty ], dynamic_rg_pruning=eligible, pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] +01)ProjectionExec: expr=[SearchPhrase@1 as SearchPhrase] +02)--SortExec: TopK(fetch=10), expr=[EventTime@0 ASC NULLS LAST, SearchPhrase@1 ASC NULLS LAST], preserve_partitioning=[false] +03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[EventTime, SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != AND DynamicFilter [ empty ], sort_order_for_reorder=[EventTime@0 ASC NULLS LAST, SearchPhrase@1 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] query T SELECT "SearchPhrase" FROM hits WHERE "SearchPhrase" <> '' ORDER BY "EventTime", "SearchPhrase" LIMIT 10; @@ -746,9 +721,8 @@ physical_plan 05)--------AggregateExec: mode=FinalPartitioned, gby=[CounterID@0 as CounterID], aggr=[avg(octet_length(hits.URL)), count(Int64(1))] 06)----------RepartitionExec: partitioning=Hash([CounterID@0], 4), input_partitions=4 07)------------AggregateExec: mode=Partial, gby=[CounterID@0 as CounterID], aggr=[avg(octet_length(hits.URL)), count(Int64(1))] -08)--------------FilterExec: URL@1 != -09)----------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -10)------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[CounterID, URL], file_type=parquet, predicate=URL@13 != , pruning_predicate=URL_null_count@2 != row_count@3 AND (URL_min@0 != OR != URL_max@1), required_guarantees=[URL not in ()] +08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 +09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[CounterID, URL], file_type=parquet, predicate=URL@13 != , pruning_predicate=URL_null_count@2 != row_count@3 AND (URL_min@0 != OR != URL_max@1), required_guarantees=[URL not in ()] query IRI SELECT "CounterID", AVG(octet_length("URL")) AS l, COUNT(*) AS c FROM hits WHERE "URL" <> '' GROUP BY "CounterID" HAVING COUNT(*) > 100000 ORDER BY l DESC LIMIT 25; @@ -774,9 +748,8 @@ physical_plan 05)--------AggregateExec: mode=FinalPartitioned, gby=[regexp_replace(hits.Referer,Utf8("^https?://(?:www\.)?([^/]+)/.*$"),Utf8("\1"))@0 as regexp_replace(hits.Referer,Utf8("^https?://(?:www\.)?([^/]+)/.*$"),Utf8("\1"))], aggr=[avg(octet_length(hits.Referer)), count(Int64(1)), min(hits.Referer)] 06)----------RepartitionExec: partitioning=Hash([regexp_replace(hits.Referer,Utf8("^https?://(?:www\.)?([^/]+)/.*$"),Utf8("\1"))@0], 4), input_partitions=4 07)------------AggregateExec: mode=Partial, gby=[regexp_replace(Referer@0, ^https?://(?:www\.)?([^/]+)/.*$, \1) as regexp_replace(hits.Referer,Utf8("^https?://(?:www\.)?([^/]+)/.*$"),Utf8("\1"))], aggr=[avg(octet_length(hits.Referer)), count(Int64(1)), min(hits.Referer)] -08)--------------FilterExec: Referer@0 != -09)----------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -10)------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[Referer], file_type=parquet, predicate=Referer@14 != , pruning_predicate=Referer_null_count@2 != row_count@3 AND (Referer_min@0 != OR != Referer_max@1), required_guarantees=[Referer not in ()] +08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 +09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[Referer], file_type=parquet, predicate=Referer@14 != , pruning_predicate=Referer_null_count@2 != row_count@3 AND (Referer_min@0 != OR != Referer_max@1), required_guarantees=[Referer not in ()] query TRIT SELECT REGEXP_REPLACE("Referer", '^https?://(?:www\.)?([^/]+)/.*$', '\1') AS k, AVG(octet_length("Referer")) AS l, COUNT(*) AS c, MIN("Referer") FROM hits WHERE "Referer" <> '' GROUP BY k HAVING COUNT(*) > 100000 ORDER BY l DESC LIMIT 25; @@ -822,9 +795,8 @@ physical_plan 04)------AggregateExec: mode=FinalPartitioned, gby=[SearchEngineID@0 as SearchEngineID, ClientIP@1 as ClientIP], aggr=[count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] 05)--------RepartitionExec: partitioning=Hash([SearchEngineID@0, ClientIP@1], 4), input_partitions=4 06)----------AggregateExec: mode=Partial, gby=[SearchEngineID@3 as SearchEngineID, ClientIP@0 as ClientIP], aggr=[count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] -07)------------FilterExec: SearchPhrase@4 != , projection=[ClientIP@0, IsRefresh@1, ResolutionWidth@2, SearchEngineID@3] -08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[ClientIP, IsRefresh, ResolutionWidth, SearchEngineID, SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != , pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] +07)------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 +08)--------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[ClientIP, IsRefresh, ResolutionWidth, SearchEngineID], file_type=parquet, predicate=SearchPhrase@39 != , pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] query IIIIR SELECT "SearchEngineID", "ClientIP", COUNT(*) AS c, SUM("IsRefresh"), AVG("ResolutionWidth") FROM hits WHERE "SearchPhrase" <> '' GROUP BY "SearchEngineID", "ClientIP" ORDER BY c DESC LIMIT 10; @@ -849,9 +821,8 @@ physical_plan 04)------AggregateExec: mode=FinalPartitioned, gby=[WatchID@0 as WatchID, ClientIP@1 as ClientIP], aggr=[count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] 05)--------RepartitionExec: partitioning=Hash([WatchID@0, ClientIP@1], 4), input_partitions=4 06)----------AggregateExec: mode=Partial, gby=[WatchID@0 as WatchID, ClientIP@1 as ClientIP], aggr=[count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] -07)------------FilterExec: SearchPhrase@4 != , projection=[WatchID@0, ClientIP@1, IsRefresh@2, ResolutionWidth@3] -08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[WatchID, ClientIP, IsRefresh, ResolutionWidth, SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != , pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] +07)------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 +08)--------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[WatchID, ClientIP, IsRefresh, ResolutionWidth], file_type=parquet, predicate=SearchPhrase@39 != , pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] query IIIIR SELECT "WatchID", "ClientIP", COUNT(*) AS c, SUM("IsRefresh"), AVG("ResolutionWidth") FROM hits WHERE "SearchPhrase" <> '' GROUP BY "WatchID", "ClientIP" ORDER BY c DESC LIMIT 10; @@ -996,9 +967,8 @@ physical_plan 04)------AggregateExec: mode=FinalPartitioned, gby=[URL@0 as URL], aggr=[count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([URL@0], 4), input_partitions=4 06)----------AggregateExec: mode=Partial, gby=[URL@0 as URL], aggr=[count(Int64(1))] -07)------------FilterExec: CounterID@1 = 62 AND EventDate@0 >= 15887 AND EventDate@0 <= 15917 AND DontCountHits@4 = 0 AND IsRefresh@3 = 0 AND URL@2 != , projection=[URL@2] -08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[EventDate, CounterID, URL, IsRefresh, DontCountHits], file_type=parquet, predicate=CounterID@6 = 62 AND EventDate@5 >= 15887 AND EventDate@5 <= 15917 AND DontCountHits@61 = 0 AND IsRefresh@15 = 0 AND URL@13 != , pruning_predicate=CounterID_null_count@2 != row_count@3 AND CounterID_min@0 <= 62 AND 62 <= CounterID_max@1 AND EventDate_null_count@5 != row_count@3 AND EventDate_max@4 >= 15887 AND EventDate_null_count@5 != row_count@3 AND EventDate_min@6 <= 15917 AND DontCountHits_null_count@9 != row_count@3 AND DontCountHits_min@7 <= 0 AND 0 <= DontCountHits_max@8 AND IsRefresh_null_count@12 != row_count@3 AND IsRefresh_min@10 <= 0 AND 0 <= IsRefresh_max@11 AND URL_null_count@15 != row_count@3 AND (URL_min@13 != OR != URL_max@14), required_guarantees=[CounterID in (62), DontCountHits in (0), IsRefresh in (0), URL not in ()] +07)------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 +08)--------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[URL], file_type=parquet, predicate=CounterID@6 = 62 AND EventDate@5 >= 15887 AND EventDate@5 <= 15917 AND DontCountHits@61 = 0 AND IsRefresh@15 = 0 AND URL@13 != , pruning_predicate=CounterID_null_count@2 != row_count@3 AND CounterID_min@0 <= 62 AND 62 <= CounterID_max@1 AND EventDate_null_count@5 != row_count@3 AND EventDate_max@4 >= 15887 AND EventDate_null_count@5 != row_count@3 AND EventDate_min@6 <= 15917 AND DontCountHits_null_count@9 != row_count@3 AND DontCountHits_min@7 <= 0 AND 0 <= DontCountHits_max@8 AND IsRefresh_null_count@12 != row_count@3 AND IsRefresh_min@10 <= 0 AND 0 <= IsRefresh_max@11 AND URL_null_count@15 != row_count@3 AND (URL_min@13 != OR != URL_max@14), required_guarantees=[CounterID in (62), DontCountHits in (0), IsRefresh in (0), URL not in ()] query TI SELECT "URL", COUNT(*) AS PageViews FROM hits WHERE "CounterID" = 62 AND "EventDate" >= '2013-07-01' AND "EventDate" <= '2013-07-31' AND "DontCountHits" = 0 AND "IsRefresh" = 0 AND "URL" <> '' GROUP BY "URL" ORDER BY PageViews DESC LIMIT 10; @@ -1023,9 +993,8 @@ physical_plan 04)------AggregateExec: mode=FinalPartitioned, gby=[Title@0 as Title], aggr=[count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([Title@0], 4), input_partitions=4 06)----------AggregateExec: mode=Partial, gby=[Title@0 as Title], aggr=[count(Int64(1))] -07)------------FilterExec: CounterID@2 = 62 AND EventDate@1 >= 15887 AND EventDate@1 <= 15917 AND DontCountHits@4 = 0 AND IsRefresh@3 = 0 AND Title@0 != , projection=[Title@0] -08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[Title, EventDate, CounterID, IsRefresh, DontCountHits], file_type=parquet, predicate=CounterID@6 = 62 AND EventDate@5 >= 15887 AND EventDate@5 <= 15917 AND DontCountHits@61 = 0 AND IsRefresh@15 = 0 AND Title@2 != , pruning_predicate=CounterID_null_count@2 != row_count@3 AND CounterID_min@0 <= 62 AND 62 <= CounterID_max@1 AND EventDate_null_count@5 != row_count@3 AND EventDate_max@4 >= 15887 AND EventDate_null_count@5 != row_count@3 AND EventDate_min@6 <= 15917 AND DontCountHits_null_count@9 != row_count@3 AND DontCountHits_min@7 <= 0 AND 0 <= DontCountHits_max@8 AND IsRefresh_null_count@12 != row_count@3 AND IsRefresh_min@10 <= 0 AND 0 <= IsRefresh_max@11 AND Title_null_count@15 != row_count@3 AND (Title_min@13 != OR != Title_max@14), required_guarantees=[CounterID in (62), DontCountHits in (0), IsRefresh in (0), Title not in ()] +07)------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 +08)--------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[Title], file_type=parquet, predicate=CounterID@6 = 62 AND EventDate@5 >= 15887 AND EventDate@5 <= 15917 AND DontCountHits@61 = 0 AND IsRefresh@15 = 0 AND Title@2 != , pruning_predicate=CounterID_null_count@2 != row_count@3 AND CounterID_min@0 <= 62 AND 62 <= CounterID_max@1 AND EventDate_null_count@5 != row_count@3 AND EventDate_max@4 >= 15887 AND EventDate_null_count@5 != row_count@3 AND EventDate_min@6 <= 15917 AND DontCountHits_null_count@9 != row_count@3 AND DontCountHits_min@7 <= 0 AND 0 <= DontCountHits_max@8 AND IsRefresh_null_count@12 != row_count@3 AND IsRefresh_min@10 <= 0 AND 0 <= IsRefresh_max@11 AND Title_null_count@15 != row_count@3 AND (Title_min@13 != OR != Title_max@14), required_guarantees=[CounterID in (62), DontCountHits in (0), IsRefresh in (0), Title not in ()] query TI SELECT "Title", COUNT(*) AS PageViews FROM hits WHERE "CounterID" = 62 AND "EventDate" >= '2013-07-01' AND "EventDate" <= '2013-07-31' AND "DontCountHits" = 0 AND "IsRefresh" = 0 AND "Title" <> '' GROUP BY "Title" ORDER BY PageViews DESC LIMIT 10; @@ -1052,9 +1021,8 @@ physical_plan 05)--------AggregateExec: mode=FinalPartitioned, gby=[URL@0 as URL], aggr=[count(Int64(1))] 06)----------RepartitionExec: partitioning=Hash([URL@0], 4), input_partitions=4 07)------------AggregateExec: mode=Partial, gby=[URL@0 as URL], aggr=[count(Int64(1))] -08)--------------FilterExec: CounterID@1 = 62 AND EventDate@0 >= 15887 AND EventDate@0 <= 15917 AND IsRefresh@3 = 0 AND IsLink@4 != 0 AND IsDownload@5 = 0, projection=[URL@2] -09)----------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -10)------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[EventDate, CounterID, URL, IsRefresh, IsLink, IsDownload], file_type=parquet, predicate=CounterID@6 = 62 AND EventDate@5 >= 15887 AND EventDate@5 <= 15917 AND IsRefresh@15 = 0 AND IsLink@52 != 0 AND IsDownload@53 = 0, pruning_predicate=CounterID_null_count@2 != row_count@3 AND CounterID_min@0 <= 62 AND 62 <= CounterID_max@1 AND EventDate_null_count@5 != row_count@3 AND EventDate_max@4 >= 15887 AND EventDate_null_count@5 != row_count@3 AND EventDate_min@6 <= 15917 AND IsRefresh_null_count@9 != row_count@3 AND IsRefresh_min@7 <= 0 AND 0 <= IsRefresh_max@8 AND IsLink_null_count@12 != row_count@3 AND (IsLink_min@10 != 0 OR 0 != IsLink_max@11) AND IsDownload_null_count@15 != row_count@3 AND IsDownload_min@13 <= 0 AND 0 <= IsDownload_max@14, required_guarantees=[CounterID in (62), IsDownload in (0), IsLink not in (0), IsRefresh in (0)] +08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 +09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[URL], file_type=parquet, predicate=CounterID@6 = 62 AND EventDate@5 >= 15887 AND EventDate@5 <= 15917 AND IsRefresh@15 = 0 AND IsLink@52 != 0 AND IsDownload@53 = 0, pruning_predicate=CounterID_null_count@2 != row_count@3 AND CounterID_min@0 <= 62 AND 62 <= CounterID_max@1 AND EventDate_null_count@5 != row_count@3 AND EventDate_max@4 >= 15887 AND EventDate_null_count@5 != row_count@3 AND EventDate_min@6 <= 15917 AND IsRefresh_null_count@9 != row_count@3 AND IsRefresh_min@7 <= 0 AND 0 <= IsRefresh_max@8 AND IsLink_null_count@12 != row_count@3 AND (IsLink_min@10 != 0 OR 0 != IsLink_max@11) AND IsDownload_null_count@15 != row_count@3 AND IsDownload_min@13 <= 0 AND 0 <= IsDownload_max@14, required_guarantees=[CounterID in (62), IsDownload in (0), IsLink not in (0), IsRefresh in (0)] query TI SELECT "URL", COUNT(*) AS PageViews FROM hits WHERE "CounterID" = 62 AND "EventDate" >= '2013-07-01' AND "EventDate" <= '2013-07-31' AND "IsRefresh" = 0 AND "IsLink" <> 0 AND "IsDownload" = 0 GROUP BY "URL" ORDER BY PageViews DESC LIMIT 10 OFFSET 1000; @@ -1081,9 +1049,8 @@ physical_plan 05)--------AggregateExec: mode=FinalPartitioned, gby=[TraficSourceID@0 as TraficSourceID, SearchEngineID@1 as SearchEngineID, AdvEngineID@2 as AdvEngineID, CASE WHEN hits.SearchEngineID = Int64(0) AND hits.AdvEngineID = Int64(0) THEN hits.Referer ELSE Utf8("") END@3 as CASE WHEN hits.SearchEngineID = Int64(0) AND hits.AdvEngineID = Int64(0) THEN hits.Referer ELSE Utf8("") END, URL@4 as URL], aggr=[count(Int64(1))] 06)----------RepartitionExec: partitioning=Hash([TraficSourceID@0, SearchEngineID@1, AdvEngineID@2, CASE WHEN hits.SearchEngineID = Int64(0) AND hits.AdvEngineID = Int64(0) THEN hits.Referer ELSE Utf8("") END@3, URL@4], 4), input_partitions=4 07)------------AggregateExec: mode=Partial, gby=[TraficSourceID@2 as TraficSourceID, SearchEngineID@3 as SearchEngineID, AdvEngineID@4 as AdvEngineID, CASE WHEN SearchEngineID@3 = 0 AND AdvEngineID@4 = 0 THEN Referer@1 ELSE END as CASE WHEN hits.SearchEngineID = Int64(0) AND hits.AdvEngineID = Int64(0) THEN hits.Referer ELSE Utf8("") END, URL@0 as URL], aggr=[count(Int64(1))] -08)--------------FilterExec: CounterID@1 = 62 AND EventDate@0 >= 15887 AND EventDate@0 <= 15917 AND IsRefresh@4 = 0, projection=[URL@2, Referer@3, TraficSourceID@5, SearchEngineID@6, AdvEngineID@7] -09)----------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -10)------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[EventDate, CounterID, URL, Referer, IsRefresh, TraficSourceID, SearchEngineID, AdvEngineID], file_type=parquet, predicate=CounterID@6 = 62 AND EventDate@5 >= 15887 AND EventDate@5 <= 15917 AND IsRefresh@15 = 0, pruning_predicate=CounterID_null_count@2 != row_count@3 AND CounterID_min@0 <= 62 AND 62 <= CounterID_max@1 AND EventDate_null_count@5 != row_count@3 AND EventDate_max@4 >= 15887 AND EventDate_null_count@5 != row_count@3 AND EventDate_min@6 <= 15917 AND IsRefresh_null_count@9 != row_count@3 AND IsRefresh_min@7 <= 0 AND 0 <= IsRefresh_max@8, required_guarantees=[CounterID in (62), IsRefresh in (0)] +08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 +09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[URL, Referer, TraficSourceID, SearchEngineID, AdvEngineID], file_type=parquet, predicate=CounterID@6 = 62 AND EventDate@5 >= 15887 AND EventDate@5 <= 15917 AND IsRefresh@15 = 0, pruning_predicate=CounterID_null_count@2 != row_count@3 AND CounterID_min@0 <= 62 AND 62 <= CounterID_max@1 AND EventDate_null_count@5 != row_count@3 AND EventDate_max@4 >= 15887 AND EventDate_null_count@5 != row_count@3 AND EventDate_min@6 <= 15917 AND IsRefresh_null_count@9 != row_count@3 AND IsRefresh_min@7 <= 0 AND 0 <= IsRefresh_max@8, required_guarantees=[CounterID in (62), IsRefresh in (0)] query IIITTI SELECT "TraficSourceID", "SearchEngineID", "AdvEngineID", CASE WHEN ("SearchEngineID" = 0 AND "AdvEngineID" = 0) THEN "Referer" ELSE '' END AS Src, "URL" AS Dst, COUNT(*) AS PageViews FROM hits WHERE "CounterID" = 62 AND "EventDate" >= '2013-07-01' AND "EventDate" <= '2013-07-31' AND "IsRefresh" = 0 GROUP BY "TraficSourceID", "SearchEngineID", "AdvEngineID", Src, Dst ORDER BY PageViews DESC LIMIT 10 OFFSET 1000; @@ -1110,10 +1077,9 @@ physical_plan 05)--------AggregateExec: mode=FinalPartitioned, gby=[URLHash@0 as URLHash, EventDate@1 as EventDate], aggr=[count(Int64(1))] 06)----------RepartitionExec: partitioning=Hash([URLHash@0, EventDate@1], 4), input_partitions=4 07)------------AggregateExec: mode=Partial, gby=[URLHash@0 as URLHash, EventDate@1 as EventDate], aggr=[count(Int64(1))] -08)--------------ProjectionExec: expr=[URLHash@0 as URLHash, CAST(CAST(EventDate@1 AS Int32) AS Date32) as EventDate] -09)----------------FilterExec: CounterID@1 = 62 AND EventDate@0 >= 15887 AND EventDate@0 <= 15917 AND IsRefresh@2 = 0 AND (TraficSourceID@3 = -1 OR TraficSourceID@3 = 6) AND RefererHash@4 = 3594120000172545465, projection=[URLHash@5, EventDate@0] -10)------------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -11)--------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[EventDate, CounterID, IsRefresh, TraficSourceID, RefererHash, URLHash], file_type=parquet, predicate=CounterID@6 = 62 AND EventDate@5 >= 15887 AND EventDate@5 <= 15917 AND IsRefresh@15 = 0 AND (TraficSourceID@37 = -1 OR TraficSourceID@37 = 6) AND RefererHash@102 = 3594120000172545465, pruning_predicate=CounterID_null_count@2 != row_count@3 AND CounterID_min@0 <= 62 AND 62 <= CounterID_max@1 AND EventDate_null_count@5 != row_count@3 AND EventDate_max@4 >= 15887 AND EventDate_null_count@5 != row_count@3 AND EventDate_min@6 <= 15917 AND IsRefresh_null_count@9 != row_count@3 AND IsRefresh_min@7 <= 0 AND 0 <= IsRefresh_max@8 AND (TraficSourceID_null_count@12 != row_count@3 AND TraficSourceID_min@10 <= -1 AND -1 <= TraficSourceID_max@11 OR TraficSourceID_null_count@12 != row_count@3 AND TraficSourceID_min@10 <= 6 AND 6 <= TraficSourceID_max@11) AND RefererHash_null_count@15 != row_count@3 AND RefererHash_min@13 <= 3594120000172545465 AND 3594120000172545465 <= RefererHash_max@14, required_guarantees=[CounterID in (62), IsRefresh in (0), RefererHash in (3594120000172545465), TraficSourceID in (-1, 6)] +08)--------------ProjectionExec: expr=[URLHash@5 as URLHash, CAST(CAST(EventDate@0 AS Int32) AS Date32) as EventDate] +09)----------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 +10)------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[EventDate, CounterID, IsRefresh, TraficSourceID, RefererHash, URLHash], file_type=parquet, predicate=CounterID@6 = 62 AND EventDate@5 >= 15887 AND EventDate@5 <= 15917 AND IsRefresh@15 = 0 AND (TraficSourceID@37 = -1 OR TraficSourceID@37 = 6) AND RefererHash@102 = 3594120000172545465, pruning_predicate=CounterID_null_count@2 != row_count@3 AND CounterID_min@0 <= 62 AND 62 <= CounterID_max@1 AND EventDate_null_count@5 != row_count@3 AND EventDate_max@4 >= 15887 AND EventDate_null_count@5 != row_count@3 AND EventDate_min@6 <= 15917 AND IsRefresh_null_count@9 != row_count@3 AND IsRefresh_min@7 <= 0 AND 0 <= IsRefresh_max@8 AND (TraficSourceID_null_count@12 != row_count@3 AND TraficSourceID_min@10 <= -1 AND -1 <= TraficSourceID_max@11 OR TraficSourceID_null_count@12 != row_count@3 AND TraficSourceID_min@10 <= 6 AND 6 <= TraficSourceID_max@11) AND RefererHash_null_count@15 != row_count@3 AND RefererHash_min@13 <= 3594120000172545465 AND 3594120000172545465 <= RefererHash_max@14, required_guarantees=[CounterID in (62), IsRefresh in (0), RefererHash in (3594120000172545465), TraficSourceID in (-1, 6)] query IDI SELECT "URLHash", "EventDate", COUNT(*) AS PageViews FROM hits WHERE "CounterID" = 62 AND "EventDate" >= '2013-07-01' AND "EventDate" <= '2013-07-31' AND "IsRefresh" = 0 AND "TraficSourceID" IN (-1, 6) AND "RefererHash" = 3594120000172545465 GROUP BY "URLHash", "EventDate" ORDER BY PageViews DESC LIMIT 10 OFFSET 100; @@ -1140,9 +1106,8 @@ physical_plan 05)--------AggregateExec: mode=FinalPartitioned, gby=[WindowClientWidth@0 as WindowClientWidth, WindowClientHeight@1 as WindowClientHeight], aggr=[count(Int64(1))] 06)----------RepartitionExec: partitioning=Hash([WindowClientWidth@0, WindowClientHeight@1], 4), input_partitions=4 07)------------AggregateExec: mode=Partial, gby=[WindowClientWidth@0 as WindowClientWidth, WindowClientHeight@1 as WindowClientHeight], aggr=[count(Int64(1))] -08)--------------FilterExec: CounterID@1 = 62 AND EventDate@0 >= 15887 AND EventDate@0 <= 15917 AND IsRefresh@2 = 0 AND DontCountHits@5 = 0 AND URLHash@6 = 2868770270353813622, projection=[WindowClientWidth@3, WindowClientHeight@4] -09)----------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -10)------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[EventDate, CounterID, IsRefresh, WindowClientWidth, WindowClientHeight, DontCountHits, URLHash], file_type=parquet, predicate=CounterID@6 = 62 AND EventDate@5 >= 15887 AND EventDate@5 <= 15917 AND IsRefresh@15 = 0 AND DontCountHits@61 = 0 AND URLHash@103 = 2868770270353813622, pruning_predicate=CounterID_null_count@2 != row_count@3 AND CounterID_min@0 <= 62 AND 62 <= CounterID_max@1 AND EventDate_null_count@5 != row_count@3 AND EventDate_max@4 >= 15887 AND EventDate_null_count@5 != row_count@3 AND EventDate_min@6 <= 15917 AND IsRefresh_null_count@9 != row_count@3 AND IsRefresh_min@7 <= 0 AND 0 <= IsRefresh_max@8 AND DontCountHits_null_count@12 != row_count@3 AND DontCountHits_min@10 <= 0 AND 0 <= DontCountHits_max@11 AND URLHash_null_count@15 != row_count@3 AND URLHash_min@13 <= 2868770270353813622 AND 2868770270353813622 <= URLHash_max@14, required_guarantees=[CounterID in (62), DontCountHits in (0), IsRefresh in (0), URLHash in (2868770270353813622)] +08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 +09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[WindowClientWidth, WindowClientHeight], file_type=parquet, predicate=CounterID@6 = 62 AND EventDate@5 >= 15887 AND EventDate@5 <= 15917 AND IsRefresh@15 = 0 AND DontCountHits@61 = 0 AND URLHash@103 = 2868770270353813622, pruning_predicate=CounterID_null_count@2 != row_count@3 AND CounterID_min@0 <= 62 AND 62 <= CounterID_max@1 AND EventDate_null_count@5 != row_count@3 AND EventDate_max@4 >= 15887 AND EventDate_null_count@5 != row_count@3 AND EventDate_min@6 <= 15917 AND IsRefresh_null_count@9 != row_count@3 AND IsRefresh_min@7 <= 0 AND 0 <= IsRefresh_max@8 AND DontCountHits_null_count@12 != row_count@3 AND DontCountHits_min@10 <= 0 AND 0 <= DontCountHits_max@11 AND URLHash_null_count@15 != row_count@3 AND URLHash_min@13 <= 2868770270353813622 AND 2868770270353813622 <= URLHash_max@14, required_guarantees=[CounterID in (62), DontCountHits in (0), IsRefresh in (0), URLHash in (2868770270353813622)] query III SELECT "WindowClientWidth", "WindowClientHeight", COUNT(*) AS PageViews FROM hits WHERE "CounterID" = 62 AND "EventDate" >= '2013-07-01' AND "EventDate" <= '2013-07-31' AND "IsRefresh" = 0 AND "DontCountHits" = 0 AND "URLHash" = 2868770270353813622 GROUP BY "WindowClientWidth", "WindowClientHeight" ORDER BY PageViews DESC LIMIT 10 OFFSET 10000; @@ -1169,9 +1134,8 @@ physical_plan 05)--------AggregateExec: mode=FinalPartitioned, gby=[date_trunc(Utf8("minute"),to_timestamp_seconds(hits.EventTime))@0 as date_trunc(Utf8("minute"),to_timestamp_seconds(hits.EventTime))], aggr=[count(Int64(1))] 06)----------RepartitionExec: partitioning=Hash([date_trunc(Utf8("minute"),to_timestamp_seconds(hits.EventTime))@0], 4), input_partitions=4 07)------------AggregateExec: mode=Partial, gby=[date_trunc(minute, to_timestamp_seconds(EventTime@0)) as date_trunc(Utf8("minute"),to_timestamp_seconds(hits.EventTime))], aggr=[count(Int64(1))] -08)--------------FilterExec: CounterID@2 = 62 AND EventDate@1 >= 15900 AND EventDate@1 <= 15901 AND IsRefresh@3 = 0 AND DontCountHits@4 = 0, projection=[EventTime@0] -09)----------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -10)------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[EventTime, EventDate, CounterID, IsRefresh, DontCountHits], file_type=parquet, predicate=CounterID@6 = 62 AND EventDate@5 >= 15900 AND EventDate@5 <= 15901 AND IsRefresh@15 = 0 AND DontCountHits@61 = 0, pruning_predicate=CounterID_null_count@2 != row_count@3 AND CounterID_min@0 <= 62 AND 62 <= CounterID_max@1 AND EventDate_null_count@5 != row_count@3 AND EventDate_max@4 >= 15900 AND EventDate_null_count@5 != row_count@3 AND EventDate_min@6 <= 15901 AND IsRefresh_null_count@9 != row_count@3 AND IsRefresh_min@7 <= 0 AND 0 <= IsRefresh_max@8 AND DontCountHits_null_count@12 != row_count@3 AND DontCountHits_min@10 <= 0 AND 0 <= DontCountHits_max@11, required_guarantees=[CounterID in (62), DontCountHits in (0), IsRefresh in (0)] +08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 +09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[EventTime], file_type=parquet, predicate=CounterID@6 = 62 AND EventDate@5 >= 15900 AND EventDate@5 <= 15901 AND IsRefresh@15 = 0 AND DontCountHits@61 = 0, pruning_predicate=CounterID_null_count@2 != row_count@3 AND CounterID_min@0 <= 62 AND 62 <= CounterID_max@1 AND EventDate_null_count@5 != row_count@3 AND EventDate_max@4 >= 15900 AND EventDate_null_count@5 != row_count@3 AND EventDate_min@6 <= 15901 AND IsRefresh_null_count@9 != row_count@3 AND IsRefresh_min@7 <= 0 AND 0 <= IsRefresh_max@8 AND DontCountHits_null_count@12 != row_count@3 AND DontCountHits_min@10 <= 0 AND 0 <= DontCountHits_max@11, required_guarantees=[CounterID in (62), DontCountHits in (0), IsRefresh in (0)] query PI SELECT DATE_TRUNC('minute', to_timestamp_seconds("EventTime")) AS M, COUNT(*) AS PageViews FROM hits WHERE "CounterID" = 62 AND "EventDate" >= '2013-07-14' AND "EventDate" <= '2013-07-15' AND "IsRefresh" = 0 AND "DontCountHits" = 0 GROUP BY DATE_TRUNC('minute', to_timestamp_seconds("EventTime")) ORDER BY DATE_TRUNC('minute', M) LIMIT 10 OFFSET 1000; diff --git a/datafusion/sqllogictest/test_files/cte.slt b/datafusion/sqllogictest/test_files/cte.slt index 289134d9f140b..c23501d65af42 100644 --- a/datafusion/sqllogictest/test_files/cte.slt +++ b/datafusion/sqllogictest/test_files/cte.slt @@ -1699,9 +1699,7 @@ physical_plan 07)----------FilterExec: k@0 < scalar_subquery() 08)------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 09)--------------WorkTableExec: name=r -10)--FilterExec: k@0 = 2, projection=[v2@1] -11)----RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -12)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/cte/test.parquet]]}, projection=[k, v2], output_ordering=[k@0 ASC NULLS LAST], file_type=parquet, predicate=k@0 = 2, pruning_predicate=k_null_count@2 != row_count@3 AND k_min@0 <= 2 AND 2 <= k_max@1, required_guarantees=[k in (2)] +10)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/cte/test.parquet]]}, projection=[v2], file_type=parquet, predicate=k@0 = 2, pruning_predicate=k_null_count@2 != row_count@3 AND k_min@0 <= 2 AND 2 <= k_max@1, required_guarantees=[k in (2)] query II with recursive r as ( diff --git a/datafusion/sqllogictest/test_files/explain_tree.slt b/datafusion/sqllogictest/test_files/explain_tree.slt index fcac86aa21a2e..17dc5d26c99ea 100644 --- a/datafusion/sqllogictest/test_files/explain_tree.slt +++ b/datafusion/sqllogictest/test_files/explain_tree.slt @@ -530,29 +530,14 @@ explain SELECT int_col FROM table2 WHERE string_col != 'foo'; ---- physical_plan 01)┌───────────────────────────┐ -02)│ FilterExec │ +02)│ DataSourceExec │ 03)│ -------------------- │ -04)│ predicate: │ -05)│ string_col != foo │ -06)└─────────────┬─────────────┘ -07)┌─────────────┴─────────────┐ -08)│ RepartitionExec │ -09)│ -------------------- │ -10)│ partition_count(in->out): │ -11)│ 1 -> 4 │ -12)│ │ -13)│ partitioning_scheme: │ -14)│ RoundRobinBatch(4) │ -15)└─────────────┬─────────────┘ -16)┌─────────────┴─────────────┐ -17)│ DataSourceExec │ -18)│ -------------------- │ -19)│ files: 1 │ -20)│ format: parquet │ -21)│ │ -22)│ predicate: │ -23)│ string_col != foo │ -24)└───────────────────────────┘ +04)│ files: 1 │ +05)│ format: parquet │ +06)│ │ +07)│ predicate: │ +08)│ string_col != foo │ +09)└───────────────────────────┘ # Query with filter on memory query TT diff --git a/datafusion/sqllogictest/test_files/grouping_set_repartition.slt b/datafusion/sqllogictest/test_files/grouping_set_repartition.slt index 16ab90651c8b3..78122b1dd2ed0 100644 --- a/datafusion/sqllogictest/test_files/grouping_set_repartition.slt +++ b/datafusion/sqllogictest/test_files/grouping_set_repartition.slt @@ -150,20 +150,17 @@ physical_plan 09)----------------AggregateExec: mode=FinalPartitioned, gby=[brand@0 as brand], aggr=[sum(sales.amount)] 10)------------------RepartitionExec: partitioning=Hash([brand@0], 4), input_partitions=4 11)--------------------AggregateExec: mode=Partial, gby=[brand@0 as brand], aggr=[sum(sales.amount)] -12)----------------------FilterExec: channel@0 = store, projection=[brand@1, amount@2] -13)------------------------DataSourceExec: file_groups={4 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/grouping_set_repartition/part=1/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/grouping_set_repartition/part=2/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/grouping_set_repartition/part=3/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/grouping_set_repartition/part=4/data.parquet]]}, projection=[channel, brand, amount], file_type=parquet, predicate=channel@0 = store, pruning_predicate=channel_null_count@2 != row_count@3 AND channel_min@0 <= store AND store <= channel_max@1, required_guarantees=[channel in (store)] -14)--------------ProjectionExec: expr=[web as channel, brand@0 as brand, sum(sales.amount)@1 as total] -15)----------------AggregateExec: mode=FinalPartitioned, gby=[brand@0 as brand], aggr=[sum(sales.amount)] -16)------------------RepartitionExec: partitioning=Hash([brand@0], 4), input_partitions=4 -17)--------------------AggregateExec: mode=Partial, gby=[brand@0 as brand], aggr=[sum(sales.amount)] -18)----------------------FilterExec: channel@0 = web, projection=[brand@1, amount@2] -19)------------------------DataSourceExec: file_groups={4 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/grouping_set_repartition/part=1/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/grouping_set_repartition/part=2/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/grouping_set_repartition/part=3/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/grouping_set_repartition/part=4/data.parquet]]}, projection=[channel, brand, amount], file_type=parquet, predicate=channel@0 = web, pruning_predicate=channel_null_count@2 != row_count@3 AND channel_min@0 <= web AND web <= channel_max@1, required_guarantees=[channel in (web)] -20)--------------ProjectionExec: expr=[catalog as channel, brand@0 as brand, sum(sales.amount)@1 as total] -21)----------------AggregateExec: mode=FinalPartitioned, gby=[brand@0 as brand], aggr=[sum(sales.amount)] -22)------------------RepartitionExec: partitioning=Hash([brand@0], 4), input_partitions=4 -23)--------------------AggregateExec: mode=Partial, gby=[brand@0 as brand], aggr=[sum(sales.amount)] -24)----------------------FilterExec: channel@0 = catalog, projection=[brand@1, amount@2] -25)------------------------DataSourceExec: file_groups={4 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/grouping_set_repartition/part=1/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/grouping_set_repartition/part=2/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/grouping_set_repartition/part=3/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/grouping_set_repartition/part=4/data.parquet]]}, projection=[channel, brand, amount], file_type=parquet, predicate=channel@0 = catalog, pruning_predicate=channel_null_count@2 != row_count@3 AND channel_min@0 <= catalog AND catalog <= channel_max@1, required_guarantees=[channel in (catalog)] +12)----------------------DataSourceExec: file_groups={4 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/grouping_set_repartition/part=1/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/grouping_set_repartition/part=2/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/grouping_set_repartition/part=3/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/grouping_set_repartition/part=4/data.parquet]]}, projection=[brand, amount], file_type=parquet, predicate=channel@0 = store, pruning_predicate=channel_null_count@2 != row_count@3 AND channel_min@0 <= store AND store <= channel_max@1, required_guarantees=[channel in (store)] +13)--------------ProjectionExec: expr=[web as channel, brand@0 as brand, sum(sales.amount)@1 as total] +14)----------------AggregateExec: mode=FinalPartitioned, gby=[brand@0 as brand], aggr=[sum(sales.amount)] +15)------------------RepartitionExec: partitioning=Hash([brand@0], 4), input_partitions=4 +16)--------------------AggregateExec: mode=Partial, gby=[brand@0 as brand], aggr=[sum(sales.amount)] +17)----------------------DataSourceExec: file_groups={4 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/grouping_set_repartition/part=1/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/grouping_set_repartition/part=2/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/grouping_set_repartition/part=3/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/grouping_set_repartition/part=4/data.parquet]]}, projection=[brand, amount], file_type=parquet, predicate=channel@0 = web, pruning_predicate=channel_null_count@2 != row_count@3 AND channel_min@0 <= web AND web <= channel_max@1, required_guarantees=[channel in (web)] +18)--------------ProjectionExec: expr=[catalog as channel, brand@0 as brand, sum(sales.amount)@1 as total] +19)----------------AggregateExec: mode=FinalPartitioned, gby=[brand@0 as brand], aggr=[sum(sales.amount)] +20)------------------RepartitionExec: partitioning=Hash([brand@0], 4), input_partitions=4 +21)--------------------AggregateExec: mode=Partial, gby=[brand@0 as brand], aggr=[sum(sales.amount)] +22)----------------------DataSourceExec: file_groups={4 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/grouping_set_repartition/part=1/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/grouping_set_repartition/part=2/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/grouping_set_repartition/part=3/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/grouping_set_repartition/part=4/data.parquet]]}, projection=[brand, amount], file_type=parquet, predicate=channel@0 = catalog, pruning_predicate=channel_null_count@2 != row_count@3 AND channel_min@0 <= catalog AND catalog <= channel_max@1, required_guarantees=[channel in (catalog)] query TTI rowsort SELECT channel, brand, SUM(total) as grand_total diff --git a/datafusion/sqllogictest/test_files/monotonic_projection_test.slt b/datafusion/sqllogictest/test_files/monotonic_projection_test.slt index 4ecdbca24a116..944eac1ce025a 100644 --- a/datafusion/sqllogictest/test_files/monotonic_projection_test.slt +++ b/datafusion/sqllogictest/test_files/monotonic_projection_test.slt @@ -238,11 +238,8 @@ logical_plan 03)----Filter: concat_equality_ordered.c = concat(concat_equality_ordered.a, concat_equality_ordered.b) 04)------TableScan: concat_equality_ordered projection=[c, a, b], partial_filters=[concat_equality_ordered.c = concat(concat_equality_ordered.a, concat_equality_ordered.b)] physical_plan -01)SortPreservingMergeExec: [a@0 ASC NULLS LAST, b@1 ASC NULLS LAST] -02)--SortExec: expr=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST], preserve_partitioning=[true] -03)----FilterExec: c@0 = concat(a@1, b@2), projection=[a@1, b@2] -04)------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, maintains_sort_order=true -05)--------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/monotonic_projection_test/concat_equality_ordered.parquet]]}, projection=[c, a, b], output_ordering=[c@0 ASC NULLS LAST, a@1 ASC NULLS LAST, b@2 ASC NULLS LAST], file_type=parquet, predicate=c@0 = concat(a@1, b@2) +01)SortExec: expr=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST], preserve_partitioning=[false] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/monotonic_projection_test/concat_equality_ordered.parquet]]}, projection=[a, b], file_type=parquet, predicate=c@0 = concat(a@1, b@2), sort_order_for_reorder=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST] query TT SELECT a, b diff --git a/datafusion/sqllogictest/test_files/parquet.slt b/datafusion/sqllogictest/test_files/parquet.slt index 9750038fed73f..7708370c0425a 100644 --- a/datafusion/sqllogictest/test_files/parquet.slt +++ b/datafusion/sqllogictest/test_files/parquet.slt @@ -457,10 +457,7 @@ EXPLAIN logical_plan 01)Filter: CAST(binary_as_string_default.binary_col AS Utf8View) LIKE Utf8View("%a%") AND CAST(binary_as_string_default.largebinary_col AS Utf8View) LIKE Utf8View("%a%") AND CAST(binary_as_string_default.binaryview_col AS Utf8View) LIKE Utf8View("%a%") 02)--TableScan: binary_as_string_default projection=[binary_col, largebinary_col, binaryview_col], partial_filters=[CAST(binary_as_string_default.binary_col AS Utf8View) LIKE Utf8View("%a%"), CAST(binary_as_string_default.largebinary_col AS Utf8View) LIKE Utf8View("%a%"), CAST(binary_as_string_default.binaryview_col AS Utf8View) LIKE Utf8View("%a%")] -physical_plan -01)FilterExec: CAST(binary_col@0 AS Utf8View) LIKE %a% AND CAST(largebinary_col@1 AS Utf8View) LIKE %a% AND CAST(binaryview_col@2 AS Utf8View) LIKE %a% -02)--RepartitionExec: partitioning=RoundRobinBatch(2), input_partitions=1 -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet/binary_as_string.parquet]]}, projection=[binary_col, largebinary_col, binaryview_col], file_type=parquet, predicate=CAST(binary_col@0 AS Utf8View) LIKE %a% AND CAST(largebinary_col@1 AS Utf8View) LIKE %a% AND CAST(binaryview_col@2 AS Utf8View) LIKE %a% +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet/binary_as_string.parquet]]}, projection=[binary_col, largebinary_col, binaryview_col], file_type=parquet, predicate=CAST(binary_col@0 AS Utf8View) LIKE %a% AND CAST(largebinary_col@1 AS Utf8View) LIKE %a% AND CAST(binaryview_col@2 AS Utf8View) LIKE %a% statement ok @@ -504,10 +501,7 @@ EXPLAIN logical_plan 01)Filter: binary_as_string_option.binary_col LIKE Utf8View("%a%") AND binary_as_string_option.largebinary_col LIKE Utf8View("%a%") AND binary_as_string_option.binaryview_col LIKE Utf8View("%a%") 02)--TableScan: binary_as_string_option projection=[binary_col, largebinary_col, binaryview_col], partial_filters=[binary_as_string_option.binary_col LIKE Utf8View("%a%"), binary_as_string_option.largebinary_col LIKE Utf8View("%a%"), binary_as_string_option.binaryview_col LIKE Utf8View("%a%")] -physical_plan -01)FilterExec: binary_col@0 LIKE %a% AND largebinary_col@1 LIKE %a% AND binaryview_col@2 LIKE %a% -02)--RepartitionExec: partitioning=RoundRobinBatch(2), input_partitions=1 -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet/binary_as_string.parquet]]}, projection=[binary_col, largebinary_col, binaryview_col], file_type=parquet, predicate=binary_col@0 LIKE %a% AND largebinary_col@1 LIKE %a% AND binaryview_col@2 LIKE %a% +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet/binary_as_string.parquet]]}, projection=[binary_col, largebinary_col, binaryview_col], file_type=parquet, predicate=binary_col@0 LIKE %a% AND largebinary_col@1 LIKE %a% AND binaryview_col@2 LIKE %a% statement ok @@ -554,10 +548,7 @@ EXPLAIN logical_plan 01)Filter: binary_as_string_both.binary_col LIKE Utf8View("%a%") AND binary_as_string_both.largebinary_col LIKE Utf8View("%a%") AND binary_as_string_both.binaryview_col LIKE Utf8View("%a%") 02)--TableScan: binary_as_string_both projection=[binary_col, largebinary_col, binaryview_col], partial_filters=[binary_as_string_both.binary_col LIKE Utf8View("%a%"), binary_as_string_both.largebinary_col LIKE Utf8View("%a%"), binary_as_string_both.binaryview_col LIKE Utf8View("%a%")] -physical_plan -01)FilterExec: binary_col@0 LIKE %a% AND largebinary_col@1 LIKE %a% AND binaryview_col@2 LIKE %a% -02)--RepartitionExec: partitioning=RoundRobinBatch(2), input_partitions=1 -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet/binary_as_string.parquet]]}, projection=[binary_col, largebinary_col, binaryview_col], file_type=parquet, predicate=binary_col@0 LIKE %a% AND largebinary_col@1 LIKE %a% AND binaryview_col@2 LIKE %a% +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet/binary_as_string.parquet]]}, projection=[binary_col, largebinary_col, binaryview_col], file_type=parquet, predicate=binary_col@0 LIKE %a% AND largebinary_col@1 LIKE %a% AND binaryview_col@2 LIKE %a% statement ok @@ -668,10 +659,7 @@ explain select * from foo where starts_with(column1, 'f'); logical_plan 01)Filter: foo.column1 LIKE Utf8View("f%") 02)--TableScan: foo projection=[column1], partial_filters=[foo.column1 LIKE Utf8View("f%")] -physical_plan -01)FilterExec: column1@0 LIKE f% -02)--RepartitionExec: partitioning=RoundRobinBatch(2), input_partitions=1 -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet/foo.parquet]]}, projection=[column1], file_type=parquet, predicate=column1@0 LIKE f%, pruning_predicate=column1_null_count@2 != row_count@3 AND column1_min@0 <= g AND f <= column1_max@1, required_guarantees=[] +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet/foo.parquet]]}, projection=[column1], file_type=parquet, predicate=column1@0 LIKE f%, pruning_predicate=column1_null_count@2 != row_count@3 AND column1_min@0 <= g AND f <= column1_max@1, required_guarantees=[] statement ok drop table foo diff --git a/datafusion/sqllogictest/test_files/parquet_filter_pushdown.slt b/datafusion/sqllogictest/test_files/parquet_filter_pushdown.slt index 2bdb99acfe3d3..224640cab0997 100644 --- a/datafusion/sqllogictest/test_files/parquet_filter_pushdown.slt +++ b/datafusion/sqllogictest/test_files/parquet_filter_pushdown.slt @@ -94,9 +94,7 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [a@0 ASC NULLS LAST] 02)--SortExec: expr=[a@0 ASC NULLS LAST], preserve_partitioning=[true] -03)----FilterExec: b@1 > 2, projection=[a@0] -04)------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=2 -05)--------DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_filter_pushdown/parquet_table/1.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_filter_pushdown/parquet_table/2.parquet]]}, projection=[a, b], file_type=parquet, predicate=b@1 > 2, pruning_predicate=b_null_count@1 != row_count@2 AND b_max@0 > 2, required_guarantees=[] +03)----DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_filter_pushdown/parquet_table/1.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_filter_pushdown/parquet_table/2.parquet]]}, projection=[a], file_type=parquet, predicate=b@1 > 2, sort_order_for_reorder=[a@0 ASC NULLS LAST], pruning_predicate=b_null_count@1 != row_count@2 AND b_max@0 > 2, required_guarantees=[] query TT EXPLAIN select a from t_pushdown where b > 2 ORDER BY a; @@ -131,9 +129,7 @@ logical_plan 04)------TableScan: t projection=[a, b], partial_filters=[t.b = Int32(2)] physical_plan 01)CoalescePartitionsExec -02)--FilterExec: b@1 = 2, projection=[a@0] -03)----RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=2 -04)------DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_filter_pushdown/parquet_table/1.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_filter_pushdown/parquet_table/2.parquet]]}, projection=[a, b], file_type=parquet, predicate=b@1 = 2, pruning_predicate=b_null_count@2 != row_count@3 AND b_min@0 <= 2 AND 2 <= b_max@1, required_guarantees=[b in (2)] +02)--DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_filter_pushdown/parquet_table/1.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_filter_pushdown/parquet_table/2.parquet]]}, projection=[a], file_type=parquet, predicate=b@1 = 2, pruning_predicate=b_null_count@2 != row_count@3 AND b_min@0 <= 2 AND 2 <= b_max@1, required_guarantees=[b in (2)] query TT EXPLAIN select a from t_pushdown where b = 2 ORDER BY b; @@ -262,9 +258,7 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [a@0 ASC NULLS LAST] 02)--SortExec: expr=[a@0 ASC NULLS LAST], preserve_partitioning=[true] -03)----FilterExec: b@1 > 2, projection=[a@0] -04)------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=2 -05)--------DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_filter_pushdown/parquet_table/1.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_filter_pushdown/parquet_table/2.parquet]]}, projection=[a, b], file_type=parquet, predicate=b@1 > 2, pruning_predicate=b_null_count@1 != row_count@2 AND b_max@0 > 2, required_guarantees=[] +03)----DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_filter_pushdown/parquet_table/1.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_filter_pushdown/parquet_table/2.parquet]]}, projection=[a], file_type=parquet, predicate=b@1 > 2, sort_order_for_reorder=[a@0 ASC NULLS LAST], pruning_predicate=b_null_count@1 != row_count@2 AND b_max@0 > 2, required_guarantees=[] query TT EXPLAIN select a from t_pushdown where b > 2 ORDER BY a; @@ -299,9 +293,7 @@ logical_plan 04)------TableScan: t projection=[a, b], partial_filters=[t.b = Int32(2)] physical_plan 01)CoalescePartitionsExec -02)--FilterExec: b@1 = 2, projection=[a@0] -03)----RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=2 -04)------DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_filter_pushdown/parquet_table/1.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_filter_pushdown/parquet_table/2.parquet]]}, projection=[a, b], file_type=parquet, predicate=b@1 = 2, pruning_predicate=b_null_count@2 != row_count@3 AND b_min@0 <= 2 AND 2 <= b_max@1, required_guarantees=[b in (2)] +02)--DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_filter_pushdown/parquet_table/1.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_filter_pushdown/parquet_table/2.parquet]]}, projection=[a], file_type=parquet, predicate=b@1 = 2, pruning_predicate=b_null_count@2 != row_count@3 AND b_min@0 <= 2 AND 2 <= b_max@1, required_guarantees=[b in (2)] query TT EXPLAIN select a from t_pushdown where b = 2 ORDER BY b; @@ -337,9 +329,7 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [a@0 ASC NULLS LAST] 02)--SortExec: expr=[a@0 ASC NULLS LAST], preserve_partitioning=[true] -03)----FilterExec: b@1 > 2, projection=[a@0] -04)------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=2 -05)--------DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_filter_pushdown/parquet_table/1.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_filter_pushdown/parquet_table/2.parquet]]}, projection=[a, b], file_type=parquet, predicate=b@1 > 2, pruning_predicate=b_null_count@1 != row_count@2 AND b_max@0 > 2, required_guarantees=[] +03)----DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_filter_pushdown/parquet_table/1.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_filter_pushdown/parquet_table/2.parquet]]}, projection=[a], file_type=parquet, predicate=b@1 > 2, sort_order_for_reorder=[a@0 ASC NULLS LAST], pruning_predicate=b_null_count@1 != row_count@2 AND b_max@0 > 2, required_guarantees=[] query T select a from t_pushdown where b = 2 ORDER BY b; diff --git a/datafusion/sqllogictest/test_files/parquet_max_row_group_bytes.slt b/datafusion/sqllogictest/test_files/parquet_max_row_group_bytes.slt index 21c5ef22c9c52..5223822e79f26 100644 --- a/datafusion/sqllogictest/test_files/parquet_max_row_group_bytes.slt +++ b/datafusion/sqllogictest/test_files/parquet_max_row_group_bytes.slt @@ -127,9 +127,7 @@ LOCATION 'test_files/scratch/parquet_max_row_group_bytes/rg_count_size_only/'; query TT EXPLAIN ANALYZE SELECT * FROM rg_count_size_only WHERE id >= 0; ---- -Plan with Metrics -01)FilterExec: id@0 >= 0 -02)--DataSourceExec: row_groups_pruned_statistics=5 total +Plan with Metrics DataSourceExec: row_groups_pruned_statistics=5 total # Both limits set: the byte limit also flushes the sub-1000-row remainders that # the row-count limit would otherwise carry into the next batch, so the file is @@ -148,9 +146,7 @@ LOCATION 'test_files/scratch/parquet_max_row_group_bytes/rg_count_size_and_bytes query TT EXPLAIN ANALYZE SELECT * FROM rg_count_size_and_bytes WHERE id >= 0; ---- -Plan with Metrics -01)FilterExec: id@0 >= 0 -02)--DataSourceExec: row_groups_pruned_statistics=8 total +Plan with Metrics DataSourceExec: row_groups_pruned_statistics=8 total # Byte limit drives alone: the row-count limit is far larger than the data, so # only the byte limit splits. Each 1024-row batch fills a fresh (empty) row @@ -169,9 +165,7 @@ LOCATION 'test_files/scratch/parquet_max_row_group_bytes/rg_count_bytes_only/'; query TT EXPLAIN ANALYZE SELECT * FROM rg_count_bytes_only WHERE id >= 0; ---- -Plan with Metrics -01)FilterExec: id@0 >= 0 -02)--DataSourceExec: row_groups_pruned_statistics=4 total +Plan with Metrics DataSourceExec: row_groups_pruned_statistics=4 total statement ok reset datafusion.execution.parquet.allow_single_file_parallelism; diff --git a/datafusion/sqllogictest/test_files/parquet_statistics.slt b/datafusion/sqllogictest/test_files/parquet_statistics.slt index 2c9a29c74b3c1..1eea640a17cb0 100644 --- a/datafusion/sqllogictest/test_files/parquet_statistics.slt +++ b/datafusion/sqllogictest/test_files/parquet_statistics.slt @@ -58,10 +58,7 @@ LOCATION 'test_files/scratch/parquet_statistics/test_table'; query TT EXPLAIN SELECT * FROM test_table WHERE column1 = 1; ---- -physical_plan -01)FilterExec: column1@0 = 1, statistics=[Rows=Inexact(2), Bytes=Inexact(10), [(Col[0]: Min=Exact(Int64(1)) Max=Exact(Int64(1)) Null=Exact(0) Distinct=Exact(1) ScanBytes=Inexact(10))]] -02)--RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=2, statistics=[Rows=Inexact(5), Bytes=Inexact(40), [(Col[0]: Min=Inexact(Int64(1)) Max=Inexact(Int64(4)) Null=Inexact(0) ScanBytes=Inexact(40))]] -03)----DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_statistics/test_table/0.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_statistics/test_table/1.parquet]]}, projection=[column1], file_type=parquet, predicate=column1@0 = 1, pruning_predicate=column1_null_count@2 != row_count@3 AND column1_min@0 <= 1 AND 1 <= column1_max@1, required_guarantees=[column1 in (1)], statistics=[Rows=Inexact(5), Bytes=Inexact(40), [(Col[0]: Min=Inexact(Int64(1)) Max=Inexact(Int64(4)) Null=Inexact(0) ScanBytes=Inexact(40))]] +physical_plan DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_statistics/test_table/0.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_statistics/test_table/1.parquet]]}, projection=[column1], file_type=parquet, predicate=column1@0 = 1, pruning_predicate=column1_null_count@2 != row_count@3 AND column1_min@0 <= 1 AND 1 <= column1_max@1, required_guarantees=[column1 in (1)], statistics=[Rows=Inexact(5), Bytes=Inexact(40), [(Col[0]: Min=Inexact(Int64(1)) Max=Inexact(Int64(4)) Null=Inexact(0) ScanBytes=Inexact(40))]] # cleanup statement ok @@ -83,10 +80,7 @@ LOCATION 'test_files/scratch/parquet_statistics/test_table'; query TT EXPLAIN SELECT * FROM test_table WHERE column1 = 1; ---- -physical_plan -01)FilterExec: column1@0 = 1, statistics=[Rows=Inexact(2), Bytes=Inexact(10), [(Col[0]: Min=Exact(Int64(1)) Max=Exact(Int64(1)) Null=Exact(0) Distinct=Exact(1) ScanBytes=Inexact(10))]] -02)--RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=2, statistics=[Rows=Inexact(5), Bytes=Inexact(40), [(Col[0]: Min=Inexact(Int64(1)) Max=Inexact(Int64(4)) Null=Inexact(0) ScanBytes=Inexact(40))]] -03)----DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_statistics/test_table/0.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_statistics/test_table/1.parquet]]}, projection=[column1], file_type=parquet, predicate=column1@0 = 1, pruning_predicate=column1_null_count@2 != row_count@3 AND column1_min@0 <= 1 AND 1 <= column1_max@1, required_guarantees=[column1 in (1)], statistics=[Rows=Inexact(5), Bytes=Inexact(40), [(Col[0]: Min=Inexact(Int64(1)) Max=Inexact(Int64(4)) Null=Inexact(0) ScanBytes=Inexact(40))]] +physical_plan DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_statistics/test_table/0.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_statistics/test_table/1.parquet]]}, projection=[column1], file_type=parquet, predicate=column1@0 = 1, pruning_predicate=column1_null_count@2 != row_count@3 AND column1_min@0 <= 1 AND 1 <= column1_max@1, required_guarantees=[column1 in (1)], statistics=[Rows=Inexact(5), Bytes=Inexact(40), [(Col[0]: Min=Inexact(Int64(1)) Max=Inexact(Int64(4)) Null=Inexact(0) ScanBytes=Inexact(40))]] # cleanup statement ok @@ -108,10 +102,7 @@ LOCATION 'test_files/scratch/parquet_statistics/test_table'; query TT EXPLAIN SELECT * FROM test_table WHERE column1 = 1; ---- -physical_plan -01)FilterExec: column1@0 = 1, statistics=[Rows=Absent, Bytes=Absent, [(Col[0]: Min=Exact(Int64(1)) Max=Exact(Int64(1)) Null=Exact(0) Distinct=Inexact(1))]] -02)--RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=2, statistics=[Rows=Absent, Bytes=Absent, [(Col[0]:)]] -03)----DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_statistics/test_table/0.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_statistics/test_table/1.parquet]]}, projection=[column1], file_type=parquet, predicate=column1@0 = 1, pruning_predicate=column1_null_count@2 != row_count@3 AND column1_min@0 <= 1 AND 1 <= column1_max@1, required_guarantees=[column1 in (1)], statistics=[Rows=Absent, Bytes=Absent, [(Col[0]:)]] +physical_plan DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_statistics/test_table/0.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_statistics/test_table/1.parquet]]}, projection=[column1], file_type=parquet, predicate=column1@0 = 1, pruning_predicate=column1_null_count@2 != row_count@3 AND column1_min@0 <= 1 AND 1 <= column1_max@1, required_guarantees=[column1 in (1)], statistics=[Rows=Absent, Bytes=Absent, [(Col[0]:)]] # cleanup statement ok @@ -151,37 +142,25 @@ LOCATION 'test_files/scratch/parquet_statistics/typed_table.parquet'; query TT EXPLAIN SELECT i8 FROM typed_table WHERE i8 = 2; ---- -physical_plan -01)FilterExec: i8@0 = 2, statistics=[Rows=Inexact(1), Bytes=Inexact(1), [(Col[0]: Min=Exact(Int8(2)) Max=Exact(Int8(2)) Null=Exact(0) Distinct=Exact(1) ScanBytes=Inexact(1))]] -02)--RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, statistics=[Rows=Inexact(5), Bytes=Inexact(5), [(Col[0]: Min=Inexact(Int8(1)) Max=Inexact(Int8(5)) Null=Inexact(0) ScanBytes=Inexact(5))]] -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_statistics/typed_table.parquet]]}, projection=[i8], file_type=parquet, predicate=i8@0 = 2, pruning_predicate=i8_null_count@2 != row_count@3 AND i8_min@0 <= 2 AND 2 <= i8_max@1, required_guarantees=[i8 in (2)], statistics=[Rows=Inexact(5), Bytes=Inexact(5), [(Col[0]: Min=Inexact(Int8(1)) Max=Inexact(Int8(5)) Null=Inexact(0) ScanBytes=Inexact(5))]] +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_statistics/typed_table.parquet]]}, projection=[i8], file_type=parquet, predicate=i8@0 = 2, pruning_predicate=i8_null_count@2 != row_count@3 AND i8_min@0 <= 2 AND 2 <= i8_max@1, required_guarantees=[i8 in (2)], statistics=[Rows=Inexact(5), Bytes=Inexact(5), [(Col[0]: Min=Inexact(Int8(1)) Max=Inexact(Int8(5)) Null=Inexact(0) ScanBytes=Inexact(5))]] # Int64 equality query TT EXPLAIN SELECT i64 FROM typed_table WHERE i64 = 2; ---- -physical_plan -01)FilterExec: i64@0 = 2, statistics=[Rows=Inexact(1), Bytes=Inexact(8), [(Col[0]: Min=Exact(Int64(2)) Max=Exact(Int64(2)) Null=Exact(0) Distinct=Exact(1) ScanBytes=Inexact(8))]] -02)--RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, statistics=[Rows=Inexact(5), Bytes=Inexact(40), [(Col[0]: Min=Inexact(Int64(1)) Max=Inexact(Int64(5)) Null=Inexact(0) ScanBytes=Inexact(40))]] -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_statistics/typed_table.parquet]]}, projection=[i64], file_type=parquet, predicate=i64@1 = 2, pruning_predicate=i64_null_count@2 != row_count@3 AND i64_min@0 <= 2 AND 2 <= i64_max@1, required_guarantees=[i64 in (2)], statistics=[Rows=Inexact(5), Bytes=Inexact(40), [(Col[0]: Min=Inexact(Int64(1)) Max=Inexact(Int64(5)) Null=Inexact(0) ScanBytes=Inexact(40))]] +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_statistics/typed_table.parquet]]}, projection=[i64], file_type=parquet, predicate=i64@1 = 2, pruning_predicate=i64_null_count@2 != row_count@3 AND i64_min@0 <= 2 AND 2 <= i64_max@1, required_guarantees=[i64 in (2)], statistics=[Rows=Inexact(5), Bytes=Inexact(40), [(Col[0]: Min=Inexact(Int64(1)) Max=Inexact(Int64(5)) Null=Inexact(0) ScanBytes=Inexact(40))]] # Float32 equality query TT EXPLAIN SELECT f32 FROM typed_table WHERE f32 = 2.5; ---- -physical_plan -01)FilterExec: CAST(f32@0 AS Float64) = 2.5, statistics=[Rows=Inexact(1), Bytes=Inexact(1), [(Col[0]: Min=Exact(Float32(2.5)) Max=Exact(Float32(2.5)) Null=Inexact(0) Distinct=Exact(1) ScanBytes=Inexact(1))]] -02)--RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, statistics=[Rows=Inexact(5), Bytes=Inexact(20), [(Col[0]: Min=Inexact(Float32(1.5)) Max=Inexact(Float32(5.5)) Null=Inexact(0) ScanBytes=Inexact(20))]] -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_statistics/typed_table.parquet]]}, projection=[f32], file_type=parquet, predicate=CAST(f32@2 AS Float64) = 2.5, pruning_predicate=f32_null_count@2 != row_count@3 AND CAST(f32_min@0 AS Float64) <= 2.5 AND 2.5 <= CAST(f32_max@1 AS Float64), required_guarantees=[], statistics=[Rows=Inexact(5), Bytes=Inexact(20), [(Col[0]: Min=Inexact(Float32(1.5)) Max=Inexact(Float32(5.5)) Null=Inexact(0) ScanBytes=Inexact(20))]] +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_statistics/typed_table.parquet]]}, projection=[f32], file_type=parquet, predicate=CAST(f32@2 AS Float64) = 2.5, pruning_predicate=f32_null_count@2 != row_count@3 AND CAST(f32_min@0 AS Float64) <= 2.5 AND 2.5 <= CAST(f32_max@1 AS Float64), required_guarantees=[], statistics=[Rows=Inexact(5), Bytes=Inexact(20), [(Col[0]: Min=Inexact(Float32(1.5)) Max=Inexact(Float32(5.5)) Null=Inexact(0) ScanBytes=Inexact(20))]] # Reversed operand order: literal = column (Float64) query TT EXPLAIN SELECT f64 FROM typed_table WHERE 2.5 = f64; ---- -physical_plan -01)FilterExec: f64@0 = 2.5, statistics=[Rows=Inexact(1), Bytes=Inexact(1), [(Col[0]: Min=Exact(Float64(2.5)) Max=Exact(Float64(2.5)) Null=Exact(0) Distinct=Exact(1) ScanBytes=Inexact(1))]] -02)--RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, statistics=[Rows=Inexact(5), Bytes=Inexact(40), [(Col[0]: Min=Inexact(Float64(1.5)) Max=Inexact(Float64(5.5)) Null=Inexact(0) ScanBytes=Inexact(40))]] -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_statistics/typed_table.parquet]]}, projection=[f64], file_type=parquet, predicate=f64@3 = 2.5, pruning_predicate=f64_null_count@2 != row_count@3 AND f64_min@0 <= 2.5 AND 2.5 <= f64_max@1, required_guarantees=[f64 in (2.5)], statistics=[Rows=Inexact(5), Bytes=Inexact(40), [(Col[0]: Min=Inexact(Float64(1.5)) Max=Inexact(Float64(5.5)) Null=Inexact(0) ScanBytes=Inexact(40))]] +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_statistics/typed_table.parquet]]}, projection=[f64], file_type=parquet, predicate=f64@3 = 2.5, pruning_predicate=f64_null_count@2 != row_count@3 AND f64_min@0 <= 2.5 AND 2.5 <= f64_max@1, required_guarantees=[f64 in (2.5)], statistics=[Rows=Inexact(5), Bytes=Inexact(40), [(Col[0]: Min=Inexact(Float64(1.5)) Max=Inexact(Float64(5.5)) Null=Inexact(0) ScanBytes=Inexact(40))]] statement ok DROP TABLE typed_table; diff --git a/datafusion/sqllogictest/test_files/preserve_file_partitioning.slt b/datafusion/sqllogictest/test_files/preserve_file_partitioning.slt index e2dd22cc82bba..2eaeca9827139 100644 --- a/datafusion/sqllogictest/test_files/preserve_file_partitioning.slt +++ b/datafusion/sqllogictest/test_files/preserve_file_partitioning.slt @@ -363,11 +363,8 @@ physical_plan 04)------RepartitionExec: partitioning=Hash([f_dkey@0], 3), input_partitions=3, preserve_order=true, sort_exprs=f_dkey@0 ASC NULLS LAST 05)--------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[max(d.env), max(d.service), count(Int64(1)), sum(f.value)], ordering_mode=Sorted 06)----------HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(d_dkey@0, f_dkey@1)], projection=[value@3, f_dkey@4, env@1, service@2] -07)------------CoalescePartitionsExec -08)--------------FilterExec: service@2 = log -09)----------------RepartitionExec: partitioning=RoundRobinBatch(3), input_partitions=1 -10)------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/dimension/data.parquet]]}, projection=[d_dkey, env, service], file_type=parquet, predicate=service@2 = log, pruning_predicate=service_null_count@2 != row_count@3 AND service_min@0 <= log AND log <= service_max@1, required_guarantees=[service in (log)] -11)------------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=C/data.parquet]]}, projection=[value, f_dkey], output_ordering=[f_dkey@1 ASC NULLS LAST], file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible +07)------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/dimension/data.parquet]]}, projection=[d_dkey, env, service], file_type=parquet, predicate=service@2 = log, pruning_predicate=service_null_count@2 != row_count@3 AND service_min@0 <= log AND log <= service_max@1, required_guarantees=[service in (log)] +08)------------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=C/data.parquet]]}, projection=[value, f_dkey], output_ordering=[f_dkey@1 ASC NULLS LAST], file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible # Verify results without optimization query TTTIR rowsort @@ -414,11 +411,8 @@ physical_plan 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, max(d.env)@1 as max(d.env), max(d.service)@2 as max(d.service), count(Int64(1))@3 as count(*), sum(f.value)@4 as sum(f.value)] 03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[max(d.env), max(d.service), count(Int64(1)), sum(f.value)], ordering_mode=Sorted 04)------HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(d_dkey@0, f_dkey@1)], projection=[value@3, f_dkey@4, env@1, service@2] -05)--------CoalescePartitionsExec -06)----------FilterExec: service@2 = log -07)------------RepartitionExec: partitioning=RoundRobinBatch(3), input_partitions=1 -08)--------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/dimension/data.parquet]]}, projection=[d_dkey, env, service], file_type=parquet, predicate=service@2 = log, pruning_predicate=service_null_count@2 != row_count@3 AND service_min@0 <= log AND log <= service_max@1, required_guarantees=[service in (log)] -09)--------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=C/data.parquet]]}, projection=[value, f_dkey], output_ordering=[f_dkey@1 ASC NULLS LAST], output_partitioning=Hash([f_dkey@1], 3), file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible +05)--------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/dimension/data.parquet]]}, projection=[d_dkey, env, service], file_type=parquet, predicate=service@2 = log, pruning_predicate=service_null_count@2 != row_count@3 AND service_min@0 <= log AND log <= service_max@1, required_guarantees=[service in (log)] +06)--------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=C/data.parquet]]}, projection=[value, f_dkey], output_ordering=[f_dkey@1 ASC NULLS LAST], output_partitioning=Hash([f_dkey@1], 3), file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible query TTTIR rowsort SELECT f.f_dkey, MAX(d.env), MAX(d.service), count(*), sum(f.value) diff --git a/datafusion/sqllogictest/test_files/projection.slt b/datafusion/sqllogictest/test_files/projection.slt index 5cfe4ba76bef3..f3844cc59f4a1 100644 --- a/datafusion/sqllogictest/test_files/projection.slt +++ b/datafusion/sqllogictest/test_files/projection.slt @@ -342,7 +342,4 @@ logical_plan 01)Projection: 02)--Filter: t1.a > Int64(1) 03)----TableScan: t1 projection=[a], partial_filters=[t1.a > Int64(1)] -physical_plan -01)FilterExec: a@0 > 1, projection=[] -02)--RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection/17513.parquet]]}, projection=[a], file_type=parquet, predicate=a@0 > 1, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 > 1, required_guarantees=[] +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection/17513.parquet]]}, file_type=parquet, predicate=a@0 > 1, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 > 1, required_guarantees=[] diff --git a/datafusion/sqllogictest/test_files/projection_pushdown.slt b/datafusion/sqllogictest/test_files/projection_pushdown.slt index c92c95fdfbc59..5b6af161adc24 100644 --- a/datafusion/sqllogictest/test_files/projection_pushdown.slt +++ b/datafusion/sqllogictest/test_files/projection_pushdown.slt @@ -239,10 +239,7 @@ logical_plan 02)--Filter: simple_struct.id > Int64(2) 03)----Projection: get_field(simple_struct.s, Utf8("value")) AS __datafusion_extracted_1, simple_struct.id 04)------TableScan: simple_struct projection=[id, s], partial_filters=[simple_struct.id > Int64(2)] -physical_plan -01)ProjectionExec: expr=[id@1 as id, __datafusion_extracted_1@0 as simple_struct.s[value]] -02)--FilterExec: id@1 > 2 -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, id], file_type=parquet, predicate=id@0 > 2, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 2, required_guarantees=[] +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[id, get_field(s@1, value) as simple_struct.s[value]], file_type=parquet, predicate=id@0 > 2, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 2, required_guarantees=[] # Verify correctness query II @@ -264,10 +261,7 @@ logical_plan 02)--Filter: simple_struct.id > Int64(2) 03)----Projection: get_field(simple_struct.s, Utf8("value")) AS __datafusion_extracted_1, simple_struct.id 04)------TableScan: simple_struct projection=[id, s], partial_filters=[simple_struct.id > Int64(2)] -physical_plan -01)ProjectionExec: expr=[id@1 as id, __datafusion_extracted_1@0 + 1 as simple_struct.s[value] + Int64(1)] -02)--FilterExec: id@1 > 2 -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, id], file_type=parquet, predicate=id@0 > 2, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 2, required_guarantees=[] +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[id, get_field(s@1, value) + 1 as simple_struct.s[value] + Int64(1)], file_type=parquet, predicate=id@0 > 2, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 2, required_guarantees=[] # Verify correctness query II @@ -289,10 +283,7 @@ logical_plan 02)--Filter: __datafusion_extracted_1 > Int64(150) 03)----Projection: get_field(simple_struct.s, Utf8("value")) AS __datafusion_extracted_1, simple_struct.id, get_field(simple_struct.s, Utf8("label")) AS __datafusion_extracted_2 04)------TableScan: simple_struct projection=[id, s], partial_filters=[get_field(simple_struct.s, Utf8("value")) > Int64(150)] -physical_plan -01)ProjectionExec: expr=[id@0 as id, __datafusion_extracted_2@1 as simple_struct.s[label]] -02)--FilterExec: __datafusion_extracted_1@0 > 150, projection=[id@1, __datafusion_extracted_2@2] -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, id, get_field(s@1, label) as __datafusion_extracted_2], file_type=parquet, predicate=get_field(s@1, value) > 150 +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[id, get_field(s@1, label) as simple_struct.s[label]], file_type=parquet, predicate=get_field(s@1, value) > 150 # Verify correctness query IT @@ -567,8 +558,7 @@ logical_plan physical_plan 01)ProjectionExec: expr=[id@1 as id, __datafusion_extracted_1@0 as simple_struct.s[value]] 02)--SortExec: expr=[__datafusion_extracted_1@0 ASC NULLS LAST], preserve_partitioning=[false] -03)----FilterExec: id@1 > 1 -04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, id], file_type=parquet, predicate=id@0 > 1, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 1, required_guarantees=[] +03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, id], file_type=parquet, predicate=id@0 > 1, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 1, required_guarantees=[] # Verify correctness query II @@ -595,8 +585,7 @@ logical_plan physical_plan 01)ProjectionExec: expr=[id@1 as id, __datafusion_extracted_1@0 as simple_struct.s[value]] 02)--SortExec: TopK(fetch=2), expr=[__datafusion_extracted_1@0 ASC NULLS LAST], preserve_partitioning=[false] -03)----FilterExec: id@1 > 1 -04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, id], file_type=parquet, predicate=id@0 > 1 AND DynamicFilter [ empty ], dynamic_rg_pruning=eligible, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 1, required_guarantees=[] +03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, id], file_type=parquet, predicate=id@0 > 1 AND DynamicFilter [ empty ], dynamic_rg_pruning=eligible, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 1, required_guarantees=[] # Verify correctness query II @@ -620,9 +609,7 @@ logical_plan 05)--------TableScan: simple_struct projection=[id, s], partial_filters=[simple_struct.id > Int64(1)] physical_plan 01)SortExec: TopK(fetch=2), expr=[id@0 ASC NULLS LAST], preserve_partitioning=[false] -02)--ProjectionExec: expr=[id@1 as id, __datafusion_extracted_1@0 + 1 as simple_struct.s[value] + Int64(1)] -03)----FilterExec: id@1 > 1 -04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, id], file_type=parquet, predicate=id@0 > 1 AND DynamicFilter [ empty ], dynamic_rg_pruning=eligible, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 1, required_guarantees=[] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[id, get_field(s@1, value) + 1 as simple_struct.s[value] + Int64(1)], file_type=parquet, predicate=id@0 > 1 AND DynamicFilter [ empty ], sort_order_for_reorder=[id@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 1, required_guarantees=[] # Verify correctness query II @@ -765,9 +752,7 @@ physical_plan 01)SortPreservingMergeExec: [id@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[id@1 as id, __datafusion_extracted_1@0 as multi_struct.s[value]] 03)----SortExec: expr=[id@1 ASC NULLS LAST], preserve_partitioning=[true] -04)------FilterExec: id@1 > 2 -05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=3 -06)----------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/multi/part1.parquet, WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/multi/part2.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/multi/part3.parquet, WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/multi/part4.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/multi/part5.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, id], file_type=parquet, predicate=id@0 > 2, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 2, required_guarantees=[] +04)------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/multi/part1.parquet, WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/multi/part2.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/multi/part3.parquet, WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/multi/part4.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/multi/part5.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, id], file_type=parquet, predicate=id@0 > 2, sort_order_for_reorder=[id@1 ASC NULLS LAST], pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 2, required_guarantees=[] # Verify correctness query II @@ -849,10 +834,7 @@ logical_plan 02)--Filter: __datafusion_extracted_1 IS NOT NULL 03)----Projection: get_field(nullable_struct.s, Utf8("value")) AS __datafusion_extracted_1, nullable_struct.id, get_field(nullable_struct.s, Utf8("label")) AS __datafusion_extracted_2 04)------TableScan: nullable_struct projection=[id, s], partial_filters=[get_field(nullable_struct.s, Utf8("value")) IS NOT NULL] -physical_plan -01)ProjectionExec: expr=[id@0 as id, __datafusion_extracted_2@1 as nullable_struct.s[label]] -02)--FilterExec: __datafusion_extracted_1@0 IS NOT NULL, projection=[id@1, __datafusion_extracted_2@2] -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/nullable.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, id, get_field(s@1, label) as __datafusion_extracted_2], file_type=parquet, predicate=get_field(s@1, value) IS NOT NULL +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/nullable.parquet]]}, projection=[id, get_field(s@1, label) as nullable_struct.s[label]], file_type=parquet, predicate=get_field(s@1, value) IS NOT NULL # Verify correctness query IT @@ -975,9 +957,7 @@ logical_plan 05)--------TableScan: simple_struct projection=[id, s], partial_filters=[simple_struct.id > Int64(2)] physical_plan 01)ProjectionExec: expr=[__common_expr_1@0 * __common_expr_1@0 as id_and_value] -02)--ProjectionExec: expr=[id@1 + __datafusion_extracted_2@0 as __common_expr_1] -03)----FilterExec: id@1 > 2 -04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_2, id], file_type=parquet, predicate=id@0 > 2, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 2, required_guarantees=[] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[id@0 + get_field(s@1, value) as __common_expr_1], file_type=parquet, predicate=id@0 > 2, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 2, required_guarantees=[] query TT @@ -988,10 +968,7 @@ logical_plan 02)--Filter: simple_struct.id > Int64(2) 03)----Projection: get_field(simple_struct.s, Utf8("value")) AS __datafusion_extracted_1, simple_struct.id 04)------TableScan: simple_struct projection=[id, s], partial_filters=[simple_struct.id > Int64(2)] -physical_plan -01)ProjectionExec: expr=[__datafusion_extracted_1@0 + __datafusion_extracted_1@0 as doubled] -02)--FilterExec: id@1 > 2, projection=[__datafusion_extracted_1@0] -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, id], file_type=parquet, predicate=id@0 > 2, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 2, required_guarantees=[] +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) + get_field(s@1, value) as doubled], file_type=parquet, predicate=id@0 > 2, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 2, required_guarantees=[] # Verify correctness query I @@ -1013,10 +990,7 @@ logical_plan 02)--Filter: simple_struct.id > Int64(2) 03)----Projection: get_field(simple_struct.s, Utf8("value")) AS __datafusion_extracted_1, get_field(simple_struct.s, Utf8("label")) AS __datafusion_extracted_2, simple_struct.id 04)------TableScan: simple_struct projection=[id, s], partial_filters=[simple_struct.id > Int64(2)] -physical_plan -01)ProjectionExec: expr=[__datafusion_extracted_1@0 as simple_struct.s[value], __datafusion_extracted_2@1 as simple_struct.s[label]] -02)--FilterExec: id@2 > 2, projection=[__datafusion_extracted_1@0, __datafusion_extracted_2@1] -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, get_field(s@1, label) as __datafusion_extracted_2, id], file_type=parquet, predicate=id@0 > 2, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 2, required_guarantees=[] +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as simple_struct.s[value], get_field(s@1, label) as simple_struct.s[label]], file_type=parquet, predicate=id@0 > 2, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 2, required_guarantees=[] # Verify correctness query IT @@ -1063,10 +1037,7 @@ logical_plan 02)--Filter: simple_struct.id > Int64(1) 03)----Projection: get_field(simple_struct.s, Utf8("value")) AS __datafusion_extracted_1, get_field(simple_struct.s, Utf8("label")) AS __datafusion_extracted_2, simple_struct.id 04)------TableScan: simple_struct projection=[id, s], partial_filters=[simple_struct.id > Int64(1)] -physical_plan -01)ProjectionExec: expr=[__datafusion_extracted_1@0 * 2 + CAST(character_length(__datafusion_extracted_2@1) AS Int64) as score] -02)--FilterExec: id@2 > 1, projection=[__datafusion_extracted_1@0, __datafusion_extracted_2@1] -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, get_field(s@1, label) as __datafusion_extracted_2, id], file_type=parquet, predicate=id@0 > 1, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 1, required_guarantees=[] +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) * 2 + CAST(character_length(get_field(s@1, label)) AS Int64) as score], file_type=parquet, predicate=id@0 > 1, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 1, required_guarantees=[] # Verify correctness query I @@ -1140,10 +1111,7 @@ logical_plan 02)--Filter: simple_struct.id > Int64(1) 03)----Projection: get_field(simple_struct.s, Utf8("value")) AS __datafusion_extracted_1, simple_struct.id 04)------TableScan: simple_struct projection=[id, s], partial_filters=[simple_struct.id > Int64(1)] -physical_plan -01)ProjectionExec: expr=[id@1 as id, __datafusion_extracted_1@0 as simple_struct.s[value]] -02)--FilterExec: id@1 > 1 -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, id], file_type=parquet, predicate=id@0 > 1, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 1, required_guarantees=[] +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[id, get_field(s@1, value) as simple_struct.s[value]], file_type=parquet, predicate=id@0 > 1, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 1, required_guarantees=[] # Verify correctness query II @@ -1160,10 +1128,7 @@ logical_plan 02)--Filter: simple_struct.id > Int64(1) AND (simple_struct.id < Int64(4) OR simple_struct.id = Int64(5)) 03)----Projection: get_field(simple_struct.s, Utf8("value")) AS __datafusion_extracted_1, simple_struct.id 04)------TableScan: simple_struct projection=[id, s], partial_filters=[simple_struct.id > Int64(1), simple_struct.id < Int64(4) OR simple_struct.id = Int64(5)] -physical_plan -01)ProjectionExec: expr=[__datafusion_extracted_1@0 as simple_struct.s[value]] -02)--FilterExec: id@1 > 1 AND (id@1 < 4 OR id@1 = 5), projection=[__datafusion_extracted_1@0] -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, id], file_type=parquet, predicate=id@0 > 1 AND (id@0 < 4 OR id@0 = 5), pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 1 AND (id_null_count@1 != row_count@2 AND id_min@3 < 4 OR id_null_count@1 != row_count@2 AND id_min@3 <= 5 AND 5 <= id_max@0), required_guarantees=[] +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as simple_struct.s[value]], file_type=parquet, predicate=id@0 > 1 AND (id@0 < 4 OR id@0 = 5), pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 1 AND (id_null_count@1 != row_count@2 AND id_min@3 < 4 OR id_null_count@1 != row_count@2 AND id_min@3 <= 5 AND 5 <= id_max@0), required_guarantees=[] # Verify correctness - should return rows where (id > 1) AND ((id < 4) OR (id = 5)) # That's: id=2,3 (1 Int64(1) AND simple_struct.id < Int64(5) 03)----Projection: get_field(simple_struct.s, Utf8("value")) AS __datafusion_extracted_1, simple_struct.id 04)------TableScan: simple_struct projection=[id, s], partial_filters=[simple_struct.id > Int64(1), simple_struct.id < Int64(5)] -physical_plan -01)ProjectionExec: expr=[__datafusion_extracted_1@0 as simple_struct.s[value]] -02)--FilterExec: id@1 > 1 AND id@1 < 5, projection=[__datafusion_extracted_1@0] -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, id], file_type=parquet, predicate=id@0 > 1 AND id@0 < 5, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 1 AND id_null_count@1 != row_count@2 AND id_min@3 < 5, required_guarantees=[] +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as simple_struct.s[value]], file_type=parquet, predicate=id@0 > 1 AND id@0 < 5, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 1 AND id_null_count@1 != row_count@2 AND id_min@3 < 5, required_guarantees=[] # Verify correctness - should return rows where 1 < id < 5 (id=2,3,4) query I @@ -1203,10 +1165,7 @@ logical_plan 02)--Filter: simple_struct.id > Int64(1) 03)----Projection: get_field(simple_struct.s, Utf8("value")) AS __datafusion_extracted_1, get_field(simple_struct.s, Utf8("label")) AS __datafusion_extracted_2, simple_struct.id 04)------TableScan: simple_struct projection=[id, s], partial_filters=[simple_struct.id > Int64(1)] -physical_plan -01)ProjectionExec: expr=[__datafusion_extracted_1@0 as simple_struct.s[value], __datafusion_extracted_2@1 as simple_struct.s[label], id@2 as id] -02)--FilterExec: id@2 > 1 -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, get_field(s@1, label) as __datafusion_extracted_2, id], file_type=parquet, predicate=id@0 > 1, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 1, required_guarantees=[] +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as simple_struct.s[value], get_field(s@1, label) as simple_struct.s[label], id], file_type=parquet, predicate=id@0 > 1, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 1, required_guarantees=[] # Verify correctness - note that id is now at index 2 in the augmented projection query ITI @@ -1224,10 +1183,7 @@ logical_plan 02)--Filter: character_length(__datafusion_extracted_1) > Int32(4) 03)----Projection: get_field(simple_struct.s, Utf8("label")) AS __datafusion_extracted_1, get_field(simple_struct.s, Utf8("value")) AS __datafusion_extracted_2 04)------TableScan: simple_struct projection=[s], partial_filters=[character_length(get_field(simple_struct.s, Utf8("label"))) > Int32(4)] -physical_plan -01)ProjectionExec: expr=[__datafusion_extracted_2@0 as simple_struct.s[value]] -02)--FilterExec: character_length(__datafusion_extracted_1@0) > 4, projection=[__datafusion_extracted_2@1] -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, label) as __datafusion_extracted_1, get_field(s@1, value) as __datafusion_extracted_2], file_type=parquet, predicate=character_length(get_field(s@1, label)) > 4 +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as simple_struct.s[value]], file_type=parquet, predicate=character_length(get_field(s@1, label)) > 4 # Verify correctness - filter on rows where label length > 4 (all have length 5, except 'one' has 3) # Wait, from the data: alpha(5), beta(4), gamma(5), delta(5), epsilon(7) @@ -1459,9 +1415,8 @@ logical_plan 06)--TableScan: join_right projection=[id] physical_plan 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(id@0, id@0)] -02)--FilterExec: __datafusion_extracted_1@0 > 150, projection=[id@1] -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, id], file_type=parquet, predicate=get_field(s@1, value) > 150 -04)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/join_right.parquet]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[id], file_type=parquet, predicate=get_field(s@1, value) > 150 +03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/join_right.parquet]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible # Verify correctness - id matches and value > 150 query II @@ -1498,10 +1453,8 @@ logical_plan 09)--------TableScan: join_right projection=[id, s], partial_filters=[get_field(join_right.s, Utf8("level")) > Int64(3)] physical_plan 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(id@0, id@0)] -02)--FilterExec: __datafusion_extracted_1@0 > 100, projection=[id@1] -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, id], file_type=parquet, predicate=get_field(s@1, value) > 100 -04)--FilterExec: __datafusion_extracted_2@0 > 3, projection=[id@1] -05)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/join_right.parquet]]}, projection=[get_field(s@1, level) as __datafusion_extracted_2, id], file_type=parquet, predicate=get_field(s@1, level) > 3 AND DynamicFilter [ empty ], dynamic_rg_pruning=eligible +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[id], file_type=parquet, predicate=get_field(s@1, value) > 100 +03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/join_right.parquet]]}, projection=[id], file_type=parquet, predicate=get_field(s@1, level) > 3 AND DynamicFilter [ empty ], dynamic_rg_pruning=eligible # Verify correctness - id matches, value > 100, and level > 3 # Matching ids where value > 100: 2(200), 3(150), 4(300), 5(250) @@ -1607,8 +1560,7 @@ physical_plan 01)ProjectionExec: expr=[id@0 as id, __datafusion_extracted_2@1 as simple_struct.s[value], __datafusion_extracted_3@2 as join_right.s[level]] 02)--HashJoinExec: mode=CollectLeft, join_type=Left, on=[(id@1, id@0)], projection=[id@1, __datafusion_extracted_2@0, __datafusion_extracted_3@3] 03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_2, id], file_type=parquet -04)----FilterExec: __datafusion_extracted_1@0 > 5, projection=[id@1, __datafusion_extracted_3@2] -05)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/join_right.parquet]]}, projection=[get_field(s@1, level) as __datafusion_extracted_1, id, get_field(s@1, level) as __datafusion_extracted_3], file_type=parquet, predicate=get_field(s@1, level) > 5 AND DynamicFilter [ empty ], dynamic_rg_pruning=eligible +04)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/join_right.parquet]]}, projection=[id, get_field(s@1, level) as __datafusion_extracted_3], file_type=parquet, predicate=get_field(s@1, level) > 5 AND DynamicFilter [ empty ], dynamic_rg_pruning=eligible # Verify correctness - left join with level > 5 condition # Only join_right rows with level > 5 are matched: id=1 (level=10), id=4 (level=8) @@ -1640,11 +1592,7 @@ logical_plan 02)--Filter: simple_struct.id > Int64(2) 03)----Projection: get_field(simple_struct.s, Utf8("value")) AS __datafusion_extracted_1, simple_struct.id 04)------TableScan: simple_struct projection=[id, s], partial_filters=[simple_struct.id > Int64(2)] -physical_plan -01)ProjectionExec: expr=[__datafusion_extracted_1@0 as simple_struct.s[value]] -02)--FilterExec: id@1 > 2, projection=[__datafusion_extracted_1@0] -03)----RepartitionExec: partitioning=RoundRobinBatch(32), input_partitions=1 -04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, id], file_type=parquet, predicate=id@0 > 2, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 2, required_guarantees=[] +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as simple_struct.s[value]], file_type=parquet, predicate=id@0 > 2, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 2, required_guarantees=[] ##################### # Section 14: SubqueryAlias tests @@ -1665,10 +1613,7 @@ logical_plan 04)------Filter: simple_struct.id > Int64(2) 05)--------Projection: get_field(simple_struct.s, Utf8("value")) AS __datafusion_extracted_1, simple_struct.id 06)----------TableScan: simple_struct projection=[id, s], partial_filters=[simple_struct.id > Int64(2)] -physical_plan -01)ProjectionExec: expr=[__datafusion_extracted_1@0 as t.s[value]] -02)--FilterExec: id@1 > 2, projection=[__datafusion_extracted_1@0] -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, id], file_type=parquet, predicate=id@0 > 2, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 2, required_guarantees=[] +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as t.s[value]], file_type=parquet, predicate=id@0 > 2, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 2, required_guarantees=[] # Verify correctness query I @@ -1715,10 +1660,7 @@ logical_plan 05)--------Filter: simple_struct.id > Int64(2) 06)----------Projection: get_field(simple_struct.s, Utf8("value")) AS __datafusion_extracted_1, simple_struct.id 07)------------TableScan: simple_struct projection=[id, s], partial_filters=[simple_struct.id > Int64(2)] -physical_plan -01)ProjectionExec: expr=[__datafusion_extracted_1@0 as u.s[value]] -02)--FilterExec: id@1 > 2, projection=[__datafusion_extracted_1@0] -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, id], file_type=parquet, predicate=id@0 > 2, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 2, required_guarantees=[] +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as u.s[value]], file_type=parquet, predicate=id@0 > 2, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 2, required_guarantees=[] # Verify correctness query I @@ -1738,9 +1680,7 @@ logical_plan 03)----Filter: __datafusion_extracted_1 > Int64(200) 04)------Projection: get_field(simple_struct.s, Utf8("value")) AS __datafusion_extracted_1, simple_struct.id 05)--------TableScan: simple_struct projection=[id, s], partial_filters=[get_field(simple_struct.s, Utf8("value")) > Int64(200)] -physical_plan -01)FilterExec: __datafusion_extracted_1@0 > 200, projection=[id@1] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, id], file_type=parquet, predicate=get_field(s@1, value) > 200 +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[id], file_type=parquet, predicate=get_field(s@1, value) > 200 # Verify correctness query I @@ -1776,10 +1716,8 @@ logical_plan physical_plan 01)ProjectionExec: expr=[__datafusion_extracted_1@0 as t.s[value]] 02)--UnionExec -03)----FilterExec: id@1 <= 3, projection=[__datafusion_extracted_1@0] -04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, id], file_type=parquet, predicate=id@0 <= 3, pruning_predicate=id_null_count@1 != row_count@2 AND id_min@0 <= 3, required_guarantees=[] -05)----FilterExec: id@1 > 3, projection=[__datafusion_extracted_1@0] -06)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, id], file_type=parquet, predicate=id@0 > 3, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 3, required_guarantees=[] +03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1], file_type=parquet, predicate=id@0 <= 3, pruning_predicate=id_null_count@1 != row_count@2 AND id_min@0 <= 3, required_guarantees=[] +04)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1], file_type=parquet, predicate=id@0 > 3, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 3, required_guarantees=[] # Verify correctness query I @@ -1821,11 +1759,9 @@ physical_plan 02)--ProjectionExec: expr=[__datafusion_extracted_1@0 as t.s[value], __datafusion_extracted_2@1 as t.s[label]] 03)----UnionExec 04)------SortExec: expr=[__datafusion_extracted_1@0 ASC NULLS LAST], preserve_partitioning=[false] -05)--------FilterExec: id@2 <= 3, projection=[__datafusion_extracted_1@0, __datafusion_extracted_2@1] -06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, get_field(s@1, label) as __datafusion_extracted_2, id], file_type=parquet, predicate=id@0 <= 3, pruning_predicate=id_null_count@1 != row_count@2 AND id_min@0 <= 3, required_guarantees=[] -07)------SortExec: expr=[__datafusion_extracted_1@0 ASC NULLS LAST], preserve_partitioning=[false] -08)--------FilterExec: id@2 > 3, projection=[__datafusion_extracted_1@0, __datafusion_extracted_2@1] -09)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, get_field(s@1, label) as __datafusion_extracted_2, id], file_type=parquet, predicate=id@0 > 3, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 3, required_guarantees=[] +05)--------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, get_field(s@1, label) as __datafusion_extracted_2], file_type=parquet, predicate=id@0 <= 3, pruning_predicate=id_null_count@1 != row_count@2 AND id_min@0 <= 3, required_guarantees=[] +06)------SortExec: expr=[__datafusion_extracted_1@0 ASC NULLS LAST], preserve_partitioning=[false] +07)--------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[get_field(s@1, value) as __datafusion_extracted_1, get_field(s@1, label) as __datafusion_extracted_2], file_type=parquet, predicate=id@0 > 3, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 > 3, required_guarantees=[] # Verify correctness query IT @@ -1985,9 +1921,7 @@ logical_plan 02)--Filter: CASE WHEN __datafusion_extracted_3 IS NOT NULL THEN __datafusion_extracted_3 ELSE __datafusion_extracted_4 END = Int64(1) 03)----Projection: get_field(t.s, Utf8("f1")) AS __datafusion_extracted_3, get_field(t.s, Utf8("f2")) AS __datafusion_extracted_4, get_field(t.s, Utf8("f1")) AS __datafusion_extracted_2 04)------TableScan: t projection=[s], partial_filters=[CASE WHEN get_field(t.s, Utf8("f1")) IS NOT NULL THEN get_field(t.s, Utf8("f1")) ELSE get_field(t.s, Utf8("f2")) END = Int64(1)] -physical_plan -01)FilterExec: CASE WHEN __datafusion_extracted_3@0 IS NOT NULL THEN __datafusion_extracted_3@0 ELSE __datafusion_extracted_4@1 END = 1, projection=[__datafusion_extracted_2@2] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/test.parquet]]}, projection=[get_field(s@0, f1) as __datafusion_extracted_3, get_field(s@0, f2) as __datafusion_extracted_4, get_field(s@0, f1) as __datafusion_extracted_2], file_type=parquet, predicate=CASE WHEN get_field(s@0, f1) IS NOT NULL THEN get_field(s@0, f1) ELSE get_field(s@0, f2) END = 1 +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/test.parquet]]}, projection=[get_field(s@0, f1) as __datafusion_extracted_2], file_type=parquet, predicate=CASE WHEN get_field(s@0, f1) IS NOT NULL THEN get_field(s@0, f1) ELSE get_field(s@0, f2) END = 1 query I SELECT diff --git a/datafusion/sqllogictest/test_files/repartition_scan.slt b/datafusion/sqllogictest/test_files/repartition_scan.slt index fca1223cb0b2b..ce82229929166 100644 --- a/datafusion/sqllogictest/test_files/repartition_scan.slt +++ b/datafusion/sqllogictest/test_files/repartition_scan.slt @@ -62,9 +62,7 @@ EXPLAIN SELECT column1 FROM parquet_table WHERE column1 <> 42; logical_plan 01)Filter: parquet_table.column1 != Int32(42) 02)--TableScan: parquet_table projection=[column1], partial_filters=[parquet_table.column1 != Int32(42)] -physical_plan -01)FilterExec: column1@0 != 42 -02)--DataSourceExec: file_groups={4 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/2.parquet:0..131], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/2.parquet:131..262], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/2.parquet:262..393], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/2.parquet:393..521]]}, projection=[column1], file_type=parquet, predicate=column1@0 != 42, pruning_predicate=column1_null_count@2 != row_count@3 AND (column1_min@0 != 42 OR 42 != column1_max@1), required_guarantees=[column1 not in (42)] +physical_plan DataSourceExec: file_groups={4 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/2.parquet:0..131], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/2.parquet:131..262], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/2.parquet:262..393], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/2.parquet:393..521]]}, projection=[column1], file_type=parquet, predicate=column1@0 != 42, pruning_predicate=column1_null_count@2 != row_count@3 AND (column1_min@0 != 42 OR 42 != column1_max@1), required_guarantees=[column1 not in (42)] # disable round robin repartitioning statement ok @@ -77,9 +75,7 @@ EXPLAIN SELECT column1 FROM parquet_table WHERE column1 <> 42; logical_plan 01)Filter: parquet_table.column1 != Int32(42) 02)--TableScan: parquet_table projection=[column1], partial_filters=[parquet_table.column1 != Int32(42)] -physical_plan -01)FilterExec: column1@0 != 42 -02)--DataSourceExec: file_groups={4 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/2.parquet:0..131], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/2.parquet:131..262], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/2.parquet:262..393], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/2.parquet:393..521]]}, projection=[column1], file_type=parquet, predicate=column1@0 != 42, pruning_predicate=column1_null_count@2 != row_count@3 AND (column1_min@0 != 42 OR 42 != column1_max@1), required_guarantees=[column1 not in (42)] +physical_plan DataSourceExec: file_groups={4 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/2.parquet:0..131], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/2.parquet:131..262], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/2.parquet:262..393], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/2.parquet:393..521]]}, projection=[column1], file_type=parquet, predicate=column1@0 != 42, pruning_predicate=column1_null_count@2 != row_count@3 AND (column1_min@0 != 42 OR 42 != column1_max@1), required_guarantees=[column1 not in (42)] # enable round robin repartitioning again statement ok @@ -102,8 +98,7 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [column1@0 ASC NULLS LAST] 02)--SortExec: expr=[column1@0 ASC NULLS LAST], preserve_partitioning=[true] -03)----FilterExec: column1@0 != 42 -04)------DataSourceExec: file_groups={4 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/1.parquet:0..258], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/1.parquet:258..510, WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/2.parquet:0..6], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/2.parquet:6..264], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/2.parquet:264..521]]}, projection=[column1], file_type=parquet, predicate=column1@0 != 42, pruning_predicate=column1_null_count@2 != row_count@3 AND (column1_min@0 != 42 OR 42 != column1_max@1), required_guarantees=[column1 not in (42)] +03)----DataSourceExec: file_groups={4 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/1.parquet:0..258], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/1.parquet:258..510, WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/2.parquet:0..6], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/2.parquet:6..264], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/2.parquet:264..521]]}, projection=[column1], file_type=parquet, predicate=column1@0 != 42, sort_order_for_reorder=[column1@0 ASC NULLS LAST], pruning_predicate=column1_null_count@2 != row_count@3 AND (column1_min@0 != 42 OR 42 != column1_max@1), required_guarantees=[column1 not in (42)] ## Read the files as though they are ordered @@ -137,8 +132,7 @@ logical_plan 03)----TableScan: parquet_table_with_order projection=[column1], partial_filters=[parquet_table_with_order.column1 != Int32(42)] physical_plan 01)SortPreservingMergeExec: [column1@0 ASC NULLS LAST] -02)--FilterExec: column1@0 != 42 -03)----DataSourceExec: file_groups={4 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/1.parquet:0..255], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/2.parquet:0..260], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/2.parquet:260..521], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/1.parquet:255..510]]}, projection=[column1], output_ordering=[column1@0 ASC NULLS LAST], file_type=parquet, predicate=column1@0 != 42, pruning_predicate=column1_null_count@2 != row_count@3 AND (column1_min@0 != 42 OR 42 != column1_max@1), required_guarantees=[column1 not in (42)] +02)--DataSourceExec: file_groups={4 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/1.parquet:0..255], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/2.parquet:0..260], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/2.parquet:260..521], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_scan/parquet_table/1.parquet:255..510]]}, projection=[column1], output_ordering=[column1@0 ASC NULLS LAST], file_type=parquet, predicate=column1@0 != 42, pruning_predicate=column1_null_count@2 != row_count@3 AND (column1_min@0 != 42 OR 42 != column1_max@1), required_guarantees=[column1 not in (42)] # Cleanup statement ok diff --git a/datafusion/sqllogictest/test_files/repartition_subset_satisfaction.slt b/datafusion/sqllogictest/test_files/repartition_subset_satisfaction.slt index 5371ca59beea1..6517fef40436b 100644 --- a/datafusion/sqllogictest/test_files/repartition_subset_satisfaction.slt +++ b/datafusion/sqllogictest/test_files/repartition_subset_satisfaction.slt @@ -378,9 +378,8 @@ physical_plan 10)------------------AggregateExec: mode=Partial, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@2) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp), env@1 as env], aggr=[max(j.value)], ordering_mode=PartiallySorted([0, 1]) 11)--------------------HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(d_dkey@1, f_dkey@2)], projection=[f_dkey@4, env@0, timestamp@2, value@3] 12)----------------------CoalescePartitionsExec -13)------------------------FilterExec: service@1 = log, projection=[env@0, d_dkey@2] -14)--------------------------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/dimension/d_dkey=A/data.parquet, WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/dimension/d_dkey=D/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/dimension/d_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/dimension/d_dkey=C/data.parquet]]}, projection=[env, service, d_dkey], output_partitioning=Hash([d_dkey@2], 3), file_type=parquet, predicate=service@1 = log, pruning_predicate=service_null_count@2 != row_count@3 AND service_min@0 <= log AND log <= service_max@1, required_guarantees=[service in (log)] -15)----------------------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=C/data.parquet]]}, projection=[timestamp, value, f_dkey], output_ordering=[f_dkey@2 ASC NULLS LAST, timestamp@0 ASC NULLS LAST], output_partitioning=Hash([f_dkey@2], 3), file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible +13)------------------------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/dimension/d_dkey=A/data.parquet, WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/dimension/d_dkey=D/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/dimension/d_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/dimension/d_dkey=C/data.parquet]]}, projection=[env, d_dkey], output_partitioning=Hash([d_dkey@1], 3), file_type=parquet, predicate=service@1 = log, pruning_predicate=service_null_count@2 != row_count@3 AND service_min@0 <= log AND log <= service_max@1, required_guarantees=[service in (log)] +14)----------------------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=C/data.parquet]]}, projection=[timestamp, value, f_dkey], output_ordering=[f_dkey@2 ASC NULLS LAST, timestamp@0 ASC NULLS LAST], output_partitioning=Hash([f_dkey@2], 3), file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible # Verify results without subset satisfaction query TPR rowsort @@ -473,9 +472,8 @@ physical_plan 08)--------------AggregateExec: mode=SinglePartitioned, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@2) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp), env@1 as env], aggr=[max(j.value)], ordering_mode=PartiallySorted([0, 1]) 09)----------------HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(d_dkey@1, f_dkey@2)], projection=[f_dkey@4, env@0, timestamp@2, value@3] 10)------------------CoalescePartitionsExec -11)--------------------FilterExec: service@1 = log, projection=[env@0, d_dkey@2] -12)----------------------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/dimension/d_dkey=A/data.parquet, WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/dimension/d_dkey=D/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/dimension/d_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/dimension/d_dkey=C/data.parquet]]}, projection=[env, service, d_dkey], output_partitioning=Hash([d_dkey@2], 3), file_type=parquet, predicate=service@1 = log, pruning_predicate=service_null_count@2 != row_count@3 AND service_min@0 <= log AND log <= service_max@1, required_guarantees=[service in (log)] -13)------------------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=C/data.parquet]]}, projection=[timestamp, value, f_dkey], output_ordering=[f_dkey@2 ASC NULLS LAST, timestamp@0 ASC NULLS LAST], output_partitioning=Hash([f_dkey@2], 3), file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible +11)--------------------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/dimension/d_dkey=A/data.parquet, WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/dimension/d_dkey=D/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/dimension/d_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/dimension/d_dkey=C/data.parquet]]}, projection=[env, d_dkey], output_partitioning=Hash([d_dkey@1], 3), file_type=parquet, predicate=service@1 = log, pruning_predicate=service_null_count@2 != row_count@3 AND service_min@0 <= log AND log <= service_max@1, required_guarantees=[service in (log)] +12)------------------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=C/data.parquet]]}, projection=[timestamp, value, f_dkey], output_ordering=[f_dkey@2 ASC NULLS LAST, timestamp@0 ASC NULLS LAST], output_partitioning=Hash([f_dkey@2], 3), file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible # Verify results match with subset satisfaction query TPR rowsort From 22ce8bfbef222c6dd69a0dcb678dae44826c8798 Mon Sep 17 00:00:00 2001 From: Adrian Garcia Badaracco <1755071+adriangb@users.noreply.github.com> Date: Wed, 23 Sep 2026 20:19:25 -0500 Subject: [PATCH 02/42] Update tests added on main for always-accepted parquet filters Tests and sqllogictest plans that landed on main after #22384 was opened still expect the old "scan only uses the predicate for pruning" behaviour. With the scan now applying every accepted filter: - Two opener tests (`test_prune_all_null_column_equality_from_file_statistics`, `test_no_prune_when_missing_column_collapses_mixed_predicate`) now expect only the matching rows. The missing-column test also checks `post_scan_rows_pruned` so it still proves the file was read, not pruned. - `string_in_list_pruning.rs` measured unpruned rows with the scan's `output_rows`. It now uses the post-scan matched + pruned counters, which count the rows that the scan decoded. - Regenerated plans in `dynamic_filter_pushdown_config.slt`, `filter_without_sort_exec.slt`, `push_down_filter_parquet.slt`, `range_partitioning.slt` and `range_sorted_time_bin_agg.slt`: the `FilterExec` above parquet scans is gone and the new `post_scan_rows_*` metrics appear. Co-Authored-By: Claude Opus 5.5 --- .../tests/parquet/string_in_list_pruning.rs | 28 +++++--- .../datasource-parquet/src/opener/mod.rs | 22 ++++-- .../dynamic_filter_pushdown_config.slt | 8 +-- .../test_files/filter_without_sort_exec.slt | 6 +- .../test_files/push_down_filter_parquet.slt | 68 +++++++++---------- .../test_files/range_partitioning.slt | 47 +++++-------- .../test_files/range_sorted_time_bin_agg.slt | 3 +- 7 files changed, 92 insertions(+), 90 deletions(-) diff --git a/datafusion/core/tests/parquet/string_in_list_pruning.rs b/datafusion/core/tests/parquet/string_in_list_pruning.rs index 1f9d6e31cbd92..1238238bb8a34 100644 --- a/datafusion/core/tests/parquet/string_in_list_pruning.rs +++ b/datafusion/core/tests/parquet/string_in_list_pruning.rs @@ -159,6 +159,16 @@ impl ScanOutput { .as_usize() } + /// Rows the scan decoded after row-group and page pruning. + /// + /// The scan runs with `pushdown_filters = false`, so it applies the whole + /// predicate in its post-scan filter: every decoded row is either matched + /// or pruned there. The scan's `output_rows` only counts the matched rows, + /// so it cannot show how much pruning skipped. + fn rows_decoded(&self) -> usize { + self.counter("post_scan_rows_matched") + self.counter("post_scan_rows_pruned") + } + fn pruned(&self, name: &str) -> usize { let value = self .metrics @@ -372,7 +382,7 @@ async fn check_string_in_list_pruning(page_pruning: bool) { assert!(!unpruned.plan.contains("IN_SET_INTERSECTS")); assert_eq!(unpruned.pruned("row_groups_pruned_statistics"), 0); assert_eq!(unpruned.pruned("page_index_rows_pruned"), 0); - assert_eq!(unpruned.counter("output_rows"), TOTAL_ROWS); + assert_eq!(unpruned.rows_decoded(), TOTAL_ROWS); let output = scan(&file, list_size, Some(list_size), page_pruning, false).await; output.assert_results(); @@ -400,7 +410,7 @@ async fn check_string_in_list_pruning(page_pruning: bool) { "list_size={list_size}, metrics={}", output.metrics ); - assert_eq!(output.counter("output_rows"), MATCHING_ROWS); + assert_eq!(output.rows_decoded(), MATCHING_ROWS); } // The default remains 20: enabling the compact representation must not @@ -410,7 +420,7 @@ async fn check_string_in_list_pruning(page_pruning: bool) { assert!(!default.plan.contains("IN_SET_INTERSECTS")); assert_eq!(default.pruned("row_groups_pruned_statistics"), 0); assert_eq!(default.pruned("page_index_rows_pruned"), 0); - assert_eq!(default.counter("output_rows"), TOTAL_ROWS); + assert_eq!(default.rows_decoded(), TOTAL_ROWS); } /// The compact `NOT IN` form must prune exactly the units whose single repeated @@ -424,7 +434,7 @@ async fn check_string_not_in_list_pruning(page_pruning: bool) { assert!(!unpruned.plan.contains("NOT_IN_SET_MAY_MATCH")); assert_eq!(unpruned.pruned("row_groups_pruned_statistics"), 0); assert_eq!(unpruned.pruned("page_index_rows_pruned"), 0); - assert_eq!(unpruned.counter("output_rows"), TOTAL_ROWS); + assert_eq!(unpruned.rows_decoded(), TOTAL_ROWS); let output = scan(&file, list_size, Some(list_size), page_pruning, true).await; output.assert_negated_results(); @@ -454,7 +464,7 @@ async fn check_string_not_in_list_pruning(page_pruning: bool) { "list_size={list_size}, metrics={}", output.metrics ); - assert_eq!(output.counter("output_rows"), MATCHING_ROWS); + assert_eq!(output.rows_decoded(), MATCHING_ROWS); } let default = scan(&file, 21, None, page_pruning, true).await; @@ -462,7 +472,7 @@ async fn check_string_not_in_list_pruning(page_pruning: bool) { assert!(!default.plan.contains("NOT_IN_SET_MAY_MATCH")); assert_eq!(default.pruned("row_groups_pruned_statistics"), 0); assert_eq!(default.pruned("page_index_rows_pruned"), 0); - assert_eq!(default.counter("output_rows"), TOTAL_ROWS); + assert_eq!(default.rows_decoded(), TOTAL_ROWS); // A mixed interval that overlaps a list member can still contain matching // rows. Only the adjacent single-valued unit can be excluded. @@ -481,7 +491,7 @@ async fn check_string_not_in_list_pruning(page_pruning: bool) { unpruned.assert_no_filter_interference(); assert_eq!(unpruned.pruned("row_groups_pruned_statistics"), 0); assert_eq!(unpruned.pruned("page_index_rows_pruned"), 0); - assert_eq!(unpruned.counter("output_rows"), ROWS_PER_UNIT * 2); + assert_eq!(unpruned.rows_decoded(), ROWS_PER_UNIT * 2); let output = scan(&mixed_file, 21, Some(21), page_pruning, true).await; assert_eq!( @@ -500,7 +510,7 @@ async fn check_string_not_in_list_pruning(page_pruning: bool) { output.pruned("page_index_rows_pruned"), if page_pruning { ROWS_PER_UNIT } else { 0 } ); - assert_eq!(output.counter("output_rows"), ROWS_PER_UNIT); + assert_eq!(output.rows_decoded(), ROWS_PER_UNIT); } async fn check_string_not_in_list_with_truncated_bounds(page_pruning: bool) { @@ -515,7 +525,7 @@ async fn check_string_not_in_list_with_truncated_bounds(page_pruning: bool) { assert!(output.plan.contains("NOT_IN_SET_MAY_MATCH")); assert_eq!(output.pruned("row_groups_pruned_statistics"), 0); assert_eq!(output.pruned("page_index_rows_pruned"), 0); - assert_eq!(output.counter("output_rows"), ROWS_PER_UNIT); + assert_eq!(output.rows_decoded(), ROWS_PER_UNIT); } #[tokio::test] diff --git a/datafusion/datasource-parquet/src/opener/mod.rs b/datafusion/datasource-parquet/src/opener/mod.rs index 4b97fcd66e657..9f28d1ef94e4a 100644 --- a/datafusion/datasource-parquet/src/opener/mod.rs +++ b/datafusion/datasource-parquet/src/opener/mod.rs @@ -3464,7 +3464,9 @@ mod test { ); // A predicate the statistics cannot disprove still reads the file, - // so the skip above is not simply "prune everything". + // so the skip above is not simply "prune everything". The scan + // accepts the filter and applies it post-scan, so only the matching + // row survives. let metrics = ExecutionPlanMetricsSet::new(); let expr = col("a").eq(lit(2)); let predicate = logical2physical(&expr, &table_schema); @@ -3472,7 +3474,7 @@ mod test { let stream = open_file(&opener, file).await.unwrap(); let (num_batches, num_rows) = count_batches_and_rows(stream).await; assert_eq!(num_batches, 1); - assert_eq!(num_rows, 3); + assert_eq!(num_rows, 1); assert_eq!(pruned_row_groups_statistics(&metrics), 0); } @@ -3590,11 +3592,19 @@ mod test { .build(); let stream = open_file(&opener, file).await.unwrap(); let (_, num_rows) = count_batches_and_rows(stream).await; - // The row group is read and its rows handed up for the filter above - // to apply, which is the missing-column path this test protects. + // The row group is read (not pruned), which is the missing-column + // path this test protects. The scan then applies the predicate + // post-scan: `b` is NULL in every row of this file, so `b = 2` rejects + // them all. assert_eq!( - num_rows, 3, - "the file must be scanned, not pruned, on the missing-column path" + num_rows, 0, + "the file must be scanned and filtered, not pruned, on the \ + missing-column path" + ); + assert_eq!( + counter_metric_value(&metrics, "post_scan_rows_pruned"), + 3, + "every row must be read and then rejected by the post-scan filter" ); assert_eq!( pruned_row_groups_statistics(&metrics), diff --git a/datafusion/sqllogictest/test_files/dynamic_filter_pushdown_config.slt b/datafusion/sqllogictest/test_files/dynamic_filter_pushdown_config.slt index 6a6bad99f0840..cf835b5dbd5de 100644 --- a/datafusion/sqllogictest/test_files/dynamic_filter_pushdown_config.slt +++ b/datafusion/sqllogictest/test_files/dynamic_filter_pushdown_config.slt @@ -102,9 +102,8 @@ Plan with Metrics 01)SortPreservingMergeExec: [v@1 DESC], fetch=3, metrics=[output_rows=3, ] 02)--SortExec: TopK(fetch=3), expr=[v@1 DESC], preserve_partitioning=[true], filter=[v@1 IS NULL OR v@1 > 800], metrics=[output_rows=3, ] 03)----ProjectionExec: expr=[id@0 as id, value@1 as v, value@1 + id@0 as name], metrics=[output_rows=10, ] -04)------FilterExec: value@1 > 3, metrics=[output_rows=10, , selectivity=100% (10/10)] -05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, metrics=[output_rows=10, ] -06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/dynamic_filter_pushdown_config/test_data.parquet]]}, projection=[id, value], file_type=parquet, predicate=value@1 > 3 AND DynamicFilter [ value@1 IS NULL OR value@1 > 800 ], dynamic_rg_pruning=eligible, pruning_predicate=value_null_count@1 != row_count@2 AND value_max@0 > 3 AND (value_null_count@1 > 0 OR value_null_count@1 != row_count@2 AND value_max@0 > 800), required_guarantees=[], metrics=[output_rows=10, elapsed_compute=, output_bytes=80.0 B, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched -> 1 fully matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, bytes_processed=1147.0 B, bytes_scanned=210.0 B, page_index_load_skipped=1, row_groups_pruned_dynamic_filter=0, metadata_load_time=, scan_efficiency_ratio=18.31% (210/1.15 K)] +04)------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, metrics=[output_rows=10, ] +05)--------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/dynamic_filter_pushdown_config/test_data.parquet]]}, projection=[id, value], file_type=parquet, predicate=value@1 > 3 AND DynamicFilter [ value@1 IS NULL OR value@1 > 800 ], sort_order_for_reorder=[value@1 DESC], reverse_row_groups=true, dynamic_rg_pruning=eligible, pruning_predicate=value_null_count@1 != row_count@2 AND value_max@0 > 3 AND (value_null_count@1 > 0 OR value_null_count@1 != row_count@2 AND value_max@0 > 800), required_guarantees=[], metrics=[output_rows=10, elapsed_compute=, output_bytes=64.0 KB, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched -> 1 fully matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_pages_skipped_by_fully_matched=1, limit_pruned_row_groups=0 total → 0 matched, bytes_processed=1147.0 B, bytes_scanned=210.0 B, page_index_load_skipped=1, row_groups_pruned_dynamic_filter=0, metadata_load_time=, scan_efficiency_ratio=18.31% (210/1.15 K)] statement ok set datafusion.explain.analyze_level = dev; @@ -337,8 +336,7 @@ physical_plan 01)HashJoinExec: mode=CollectLeft, join_type=RightSemi, on=[(id@0, id@0)] 02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/dynamic_filter_pushdown_config/join_right.parquet]]}, projection=[id], file_type=parquet 03)--SortExec: expr=[data@1 DESC], preserve_partitioning=[false] -04)----FilterExec: DynamicFilter [ empty ] -05)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/dynamic_filter_pushdown_config/join_left.parquet]]}, projection=[id, data], file_type=parquet, predicate=DynamicFilter [ empty ], sort_order_for_reorder=[data@1 DESC], reverse_row_groups=true, dynamic_rg_pruning=eligible +04)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/dynamic_filter_pushdown_config/join_left.parquet]]}, projection=[id, data], file_type=parquet, predicate=DynamicFilter [ empty ], sort_order_for_reorder=[data@1 DESC], reverse_row_groups=true, dynamic_rg_pruning=eligible statement count 0 SET datafusion.execution.parquet.pushdown_filters = true; diff --git a/datafusion/sqllogictest/test_files/filter_without_sort_exec.slt b/datafusion/sqllogictest/test_files/filter_without_sort_exec.slt index df23390c64dfd..f28d2b017f38d 100644 --- a/datafusion/sqllogictest/test_files/filter_without_sort_exec.slt +++ b/datafusion/sqllogictest/test_files/filter_without_sort_exec.slt @@ -184,11 +184,7 @@ logical_plan 01)Sort: cast_ordered.b ASC NULLS LAST 02)--Filter: cast_ordered.b > Int64(1) 03)----TableScan: cast_ordered projection=[b], partial_filters=[cast_ordered.b > Int64(1)] -physical_plan -01)SortPreservingMergeExec: [b@0 ASC NULLS LAST] -02)--FilterExec: b@0 > 1 -03)----RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, maintains_sort_order=true -04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/filter_without_sort_exec/cast_ordering.parquet]]}, projection=[b], output_ordering=[b@0 ASC NULLS LAST], file_type=parquet, predicate=b@0 > 1, pruning_predicate=b_null_count@1 != row_count@2 AND b_max@0 > 1, required_guarantees=[] +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/filter_without_sort_exec/cast_ordering.parquet]]}, projection=[b], output_ordering=[b@0 ASC NULLS LAST], file_type=parquet, predicate=b@0 > 1, pruning_predicate=b_null_count@1 != row_count@2 AND b_max@0 > 1, required_guarantees=[] statement ok SET datafusion.execution.target_partitions = 1; diff --git a/datafusion/sqllogictest/test_files/push_down_filter_parquet.slt b/datafusion/sqllogictest/test_files/push_down_filter_parquet.slt index 8dd763205f6c2..79c80cce37d2a 100644 --- a/datafusion/sqllogictest/test_files/push_down_filter_parquet.slt +++ b/datafusion/sqllogictest/test_files/push_down_filter_parquet.slt @@ -206,7 +206,7 @@ EXPLAIN ANALYZE SELECT t FROM topk_pushdown ORDER BY t * t LIMIT 10; ---- Plan with Metrics 01)SortExec: TopK(fetch=10), expr=[t@0 * t@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[t@0 * t@0 < 1884329474306198481], metrics=[output_rows=10, output_batches=1, row_replacements=10] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_pushdown.parquet]]}, projection=[t], output_ordering=[t@0 ASC NULLS LAST], file_type=parquet, predicate=DynamicFilter [ t@0 * t@0 < 1884329474306198481 ], dynamic_rg_pruning=eligible, metrics=[output_rows=128, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=782 total → 782 matched, row_groups_pruned_bloom_filter=782 total → 782 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=128, pushdown_rows_pruned=99.87 K, predicate_cache_inner_records=128, predicate_cache_records=128, scan_efficiency_ratio=64.87% (258.7 K/398.8 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_pushdown.parquet]]}, projection=[t], output_ordering=[t@0 ASC NULLS LAST], file_type=parquet, predicate=DynamicFilter [ t@0 * t@0 < 1884329474306198481 ], dynamic_rg_pruning=eligible, metrics=[output_rows=128, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=782 total → 782 matched, row_groups_pruned_bloom_filter=782 total → 782 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=128, pushdown_rows_pruned=99.87 K, predicate_cache_inner_records=128, predicate_cache_records=128, scan_efficiency_ratio=64.87% (258.7 K/398.8 K)] statement ok reset datafusion.explain.analyze_categories; @@ -268,7 +268,7 @@ EXPLAIN ANALYZE SELECT * FROM topk_single_col ORDER BY b DESC LIMIT 1; ---- Plan with Metrics 01)SortExec: TopK(fetch=1), expr=[b@1 DESC], preserve_partitioning=[false], filter=[b@1 IS NULL OR b@1 > bd], metrics=[output_rows=1, output_batches=1, row_replacements=1] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_single_col.parquet]]}, projection=[a, b, c], file_type=parquet, predicate=DynamicFilter [ b@1 IS NULL OR b@1 > bd ], sort_order_for_reorder=[b@1 DESC], reverse_row_groups=true, dynamic_rg_pruning=eligible, pruning_predicate=b_null_count@0 > 0 OR b_null_count@0 != row_count@2 AND b_max@1 > bd, required_guarantees=[], metrics=[output_rows=4, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=4, pushdown_rows_pruned=0, predicate_cache_inner_records=4, predicate_cache_records=4, scan_efficiency_ratio=21.94% (222/1.01 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_single_col.parquet]]}, projection=[a, b, c], file_type=parquet, predicate=DynamicFilter [ b@1 IS NULL OR b@1 > bd ], sort_order_for_reorder=[b@1 DESC], reverse_row_groups=true, dynamic_rg_pruning=eligible, pruning_predicate=b_null_count@0 > 0 OR b_null_count@0 != row_count@2 AND b_max@1 > bd, required_guarantees=[], metrics=[output_rows=4, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=4, pushdown_rows_pruned=0, predicate_cache_inner_records=4, predicate_cache_records=4, scan_efficiency_ratio=21.94% (222/1.01 K)] statement ok reset datafusion.explain.analyze_categories; @@ -319,7 +319,7 @@ EXPLAIN ANALYZE SELECT * FROM topk_multi_col ORDER BY b ASC NULLS LAST, a DESC L ---- Plan with Metrics 01)SortExec: TopK(fetch=2), expr=[b@1 ASC NULLS LAST, a@0 DESC], preserve_partitioning=[false], filter=[b@1 < bb OR b@1 = bb AND (a@0 IS NULL OR a@0 > ac)], metrics=[output_rows=2, output_batches=1, row_replacements=2] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_multi_col.parquet]]}, projection=[a, b, c], file_type=parquet, predicate=DynamicFilter [ b@1 < bb OR b@1 = bb AND (a@0 IS NULL OR a@0 > ac) ], sort_order_for_reorder=[b@1 ASC NULLS LAST, a@0 DESC], dynamic_rg_pruning=eligible, pruning_predicate=b_null_count@1 != row_count@2 AND b_min@0 < bb OR b_null_count@1 != row_count@2 AND b_min@0 <= bb AND bb <= b_max@3 AND (a_null_count@4 > 0 OR a_null_count@4 != row_count@2 AND a_max@5 > ac), required_guarantees=[], metrics=[output_rows=4, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=4, pushdown_rows_pruned=0, predicate_cache_inner_records=8, predicate_cache_records=8, scan_efficiency_ratio=21.94% (222/1.01 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_multi_col.parquet]]}, projection=[a, b, c], file_type=parquet, predicate=DynamicFilter [ b@1 < bb OR b@1 = bb AND (a@0 IS NULL OR a@0 > ac) ], sort_order_for_reorder=[b@1 ASC NULLS LAST, a@0 DESC], dynamic_rg_pruning=eligible, pruning_predicate=b_null_count@1 != row_count@2 AND b_min@0 < bb OR b_null_count@1 != row_count@2 AND b_min@0 <= bb AND bb <= b_max@3 AND (a_null_count@4 > 0 OR a_null_count@4 != row_count@2 AND a_max@5 > ac), required_guarantees=[], metrics=[output_rows=4, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=4, pushdown_rows_pruned=0, predicate_cache_inner_records=8, predicate_cache_records=8, scan_efficiency_ratio=21.94% (222/1.01 K)] statement ok reset datafusion.explain.analyze_categories; @@ -388,8 +388,8 @@ FROM join_probe p INNER JOIN join_build AS build ---- Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(a@0, a@0), (b@1, b@1)], projection=[a@3, b@4, c@2, e@5], metrics=[output_rows=2, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=1, input_rows=2, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/join_build.parquet]]}, projection=[a, b, c], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=19.88% (196/986)] -03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/join_probe.parquet]]}, projection=[a, b, e], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= ba AND b@1 <= bb AND struct(a@0, b@1) IN (SET) ([{c0:aa,c1:ba}, {c0:ab,c1:bb}]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= ba AND b_null_count@5 != row_count@2 AND b_min@6 <= bb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=22.37% (228/1.02 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/join_build.parquet]]}, projection=[a, b, c], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=19.88% (196/986)] +03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/join_probe.parquet]]}, projection=[a, b, e], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= ba AND b@1 <= bb AND struct(a@0, b@1) IN (SET) ([{c0:aa,c1:ba}, {c0:ab,c1:bb}]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= ba AND b_null_count@5 != row_count@2 AND b_min@6 <= bb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=22.37% (228/1.02 K)] statement ok reset datafusion.explain.analyze_categories; @@ -474,9 +474,9 @@ INNER JOIN nested_t3 ON nested_t2.c = nested_t3.d; Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(c@3, d@0)], metrics=[output_rows=2, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=1, input_rows=2, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] 02)--HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(a@0, b@0)], metrics=[output_rows=2, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=1, input_rows=2, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nested_t1.parquet]]}, projection=[a, x], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=17.72% (132/745)] -04)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nested_t2.parquet]]}, projection=[b, c, y], file_type=parquet, predicate=DynamicFilter [ b@0 >= aa AND b@0 <= ab AND b@0 IN (SET) ([aa, ab]) ], dynamic_rg_pruning=eligible, pruning_predicate=b_null_count@1 != row_count@2 AND b_max@0 >= aa AND b_null_count@1 != row_count@2 AND b_min@3 <= ab AND (b_null_count@1 != row_count@2 AND b_min@3 <= aa AND aa <= b_max@0 OR b_null_count@1 != row_count@2 AND b_min@3 <= ab AND ab <= b_max@0), required_guarantees=[b in (aa, ab)], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=5 total → 5 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=3, predicate_cache_inner_records=5, predicate_cache_records=2, scan_efficiency_ratio=22.78% (234/1.03 K)] -05)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nested_t3.parquet]]}, projection=[d, z], file_type=parquet, predicate=DynamicFilter [ d@0 >= ca AND d@0 <= cb AND hash_lookup ], dynamic_rg_pruning=eligible, pruning_predicate=d_null_count@1 != row_count@2 AND d_max@0 >= ca AND d_null_count@1 != row_count@2 AND d_min@3 <= cb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=8 total → 8 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=6, predicate_cache_inner_records=8, predicate_cache_records=2, scan_efficiency_ratio=21.86% (172/787)] +03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nested_t1.parquet]]}, projection=[a, x], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=17.72% (132/745)] +04)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nested_t2.parquet]]}, projection=[b, c, y], file_type=parquet, predicate=DynamicFilter [ b@0 >= aa AND b@0 <= ab AND b@0 IN (SET) ([aa, ab]) ], dynamic_rg_pruning=eligible, pruning_predicate=b_null_count@1 != row_count@2 AND b_max@0 >= aa AND b_null_count@1 != row_count@2 AND b_min@3 <= ab AND (b_null_count@1 != row_count@2 AND b_min@3 <= aa AND aa <= b_max@0 OR b_null_count@1 != row_count@2 AND b_min@3 <= ab AND ab <= b_max@0), required_guarantees=[b in (aa, ab)], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=5 total → 5 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=3, predicate_cache_inner_records=5, predicate_cache_records=2, scan_efficiency_ratio=22.78% (234/1.03 K)] +05)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nested_t3.parquet]]}, projection=[d, z], file_type=parquet, predicate=DynamicFilter [ d@0 >= ca AND d@0 <= cb AND hash_lookup ], dynamic_rg_pruning=eligible, pruning_predicate=d_null_count@1 != row_count@2 AND d_max@0 >= ca AND d_null_count@1 != row_count@2 AND d_min@3 <= cb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=8 total → 8 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=6, predicate_cache_inner_records=8, predicate_cache_records=2, scan_efficiency_ratio=21.86% (172/787)] statement ok reset datafusion.explain.analyze_categories; @@ -605,8 +605,8 @@ LIMIT 2; Plan with Metrics 01)SortExec: TopK(fetch=2), expr=[e@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[e@0 < bb], metrics=[output_rows=2, output_batches=1, row_replacements=2] 02)--HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(a@0, d@0)], projection=[e@2], metrics=[output_rows=2, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=1, input_rows=2, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_join_build.parquet]]}, projection=[a], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=6.49% (64/986)] -04)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_join_probe.parquet]]}, projection=[d, e], file_type=parquet, predicate=DynamicFilter [ d@0 >= aa AND d@0 <= ab AND d@0 IN (SET) ([aa, ab]) ] AND DynamicFilter [ e@1 < bb ], dynamic_rg_pruning=eligible, pruning_predicate=d_null_count@1 != row_count@2 AND d_max@0 >= aa AND d_null_count@1 != row_count@2 AND d_min@3 <= ab AND (d_null_count@1 != row_count@2 AND d_min@3 <= aa AND aa <= d_max@0 OR d_null_count@1 != row_count@2 AND d_min@3 <= ab AND ab <= d_max@0) AND e_null_count@5 != row_count@2 AND e_min@4 < bb, required_guarantees=[d in (aa, ab)], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=4 total → 4 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=15.11% (154/1.02 K)] +03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_join_build.parquet]]}, projection=[a], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=6.49% (64/986)] +04)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_join_probe.parquet]]}, projection=[d, e], file_type=parquet, predicate=DynamicFilter [ d@0 >= aa AND d@0 <= ab AND d@0 IN (SET) ([aa, ab]) ] AND DynamicFilter [ e@1 < bb ], dynamic_rg_pruning=eligible, pruning_predicate=d_null_count@1 != row_count@2 AND d_max@0 >= aa AND d_null_count@1 != row_count@2 AND d_min@3 <= ab AND (d_null_count@1 != row_count@2 AND d_min@3 <= aa AND aa <= d_max@0 OR d_null_count@1 != row_count@2 AND d_min@3 <= ab AND ab <= d_max@0) AND e_null_count@5 != row_count@2 AND e_min@4 < bb, required_guarantees=[d in (aa, ab)], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=4 total → 4 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=15.11% (154/1.02 K)] statement ok reset datafusion.explain.analyze_categories; @@ -656,7 +656,7 @@ EXPLAIN ANALYZE SELECT b, a FROM topk_proj ORDER BY a LIMIT 2; Plan with Metrics 01)ProjectionExec: expr=[b@1 as b, a@0 as a], metrics=[output_rows=2, output_batches=1] 02)--SortExec: TopK(fetch=2), expr=[a@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[a@0 < 2], metrics=[output_rows=2, output_batches=1, row_replacements=2] -03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_proj.parquet]]}, projection=[a, b], file_type=parquet, predicate=DynamicFilter [ a@0 < 2 ], sort_order_for_reorder=[a@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_min@0 < 2, required_guarantees=[], metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=3, pushdown_rows_pruned=0, predicate_cache_inner_records=3, predicate_cache_records=3, scan_efficiency_ratio=13.4% (141/1.05 K)] +03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_proj.parquet]]}, projection=[a, b], file_type=parquet, predicate=DynamicFilter [ a@0 < 2 ], sort_order_for_reorder=[a@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_min@0 < 2, required_guarantees=[], metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=3, pushdown_rows_pruned=0, predicate_cache_inner_records=3, predicate_cache_records=3, scan_efficiency_ratio=13.4% (141/1.05 K)] # Case 2: prune — `SELECT a` — filter stays as `a < 2` on the scan. query TT @@ -664,7 +664,7 @@ EXPLAIN ANALYZE SELECT a FROM topk_proj ORDER BY a LIMIT 2; ---- Plan with Metrics 01)SortExec: TopK(fetch=2), expr=[a@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[a@0 < 2], metrics=[output_rows=2, output_batches=1, row_replacements=2] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_proj.parquet]]}, projection=[a], file_type=parquet, predicate=DynamicFilter [ a@0 < 2 ], sort_order_for_reorder=[a@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_min@0 < 2, required_guarantees=[], metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=3, pushdown_rows_pruned=0, predicate_cache_inner_records=3, predicate_cache_records=3, scan_efficiency_ratio=6.94% (73/1.05 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_proj.parquet]]}, projection=[a], file_type=parquet, predicate=DynamicFilter [ a@0 < 2 ], sort_order_for_reorder=[a@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_min@0 < 2, required_guarantees=[], metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=3, pushdown_rows_pruned=0, predicate_cache_inner_records=3, predicate_cache_records=3, scan_efficiency_ratio=6.94% (73/1.05 K)] # Case 3: expression — `SELECT a+1 AS a_plus_1` — the TopK filter is on # `a_plus_1`, the scan predicate must read `a@0 + 1`. @@ -673,7 +673,7 @@ EXPLAIN ANALYZE SELECT a + 1 AS a_plus_1, b FROM topk_proj ORDER BY a_plus_1 LIM ---- Plan with Metrics 01)SortExec: TopK(fetch=2), expr=[a_plus_1@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[a_plus_1@0 < 3], metrics=[output_rows=2, output_batches=1, row_replacements=2] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_proj.parquet]]}, projection=[CAST(a@0 AS Int64) + 1 as a_plus_1, b], file_type=parquet, predicate=DynamicFilter [ CAST(a@0 AS Int64) + 1 < 3 ], dynamic_rg_pruning=eligible, metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=3, pushdown_rows_pruned=0, predicate_cache_inner_records=3, predicate_cache_records=3, scan_efficiency_ratio=13.4% (141/1.05 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_proj.parquet]]}, projection=[CAST(a@0 AS Int64) + 1 as a_plus_1, b], file_type=parquet, predicate=DynamicFilter [ CAST(a@0 AS Int64) + 1 < 3 ], dynamic_rg_pruning=eligible, metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=3, pushdown_rows_pruned=0, predicate_cache_inner_records=3, predicate_cache_records=3, scan_efficiency_ratio=13.4% (141/1.05 K)] # Case 4: alias shadowing — `SELECT a+1 AS a` — the projection renames # `a+1` to `a`, so the TopK's `a < 3` must still be rewritten to @@ -683,7 +683,7 @@ EXPLAIN ANALYZE SELECT a + 1 AS a, b FROM topk_proj ORDER BY a LIMIT 2; ---- Plan with Metrics 01)SortExec: TopK(fetch=2), expr=[a@0 ASC NULLS LAST], preserve_partitioning=[false], filter=[a@0 < 3], metrics=[output_rows=2, output_batches=1, row_replacements=2] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_proj.parquet]]}, projection=[CAST(a@0 AS Int64) + 1 as a, b], file_type=parquet, predicate=DynamicFilter [ CAST(a@0 AS Int64) + 1 < 3 ], sort_order_for_reorder=[a@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=3, pushdown_rows_pruned=0, predicate_cache_inner_records=3, predicate_cache_records=3, scan_efficiency_ratio=13.4% (141/1.05 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/topk_proj.parquet]]}, projection=[CAST(a@0 AS Int64) + 1 as a, b], file_type=parquet, predicate=DynamicFilter [ CAST(a@0 AS Int64) + 1 < 3 ], sort_order_for_reorder=[a@0 ASC NULLS LAST], dynamic_rg_pruning=eligible, metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=3, pushdown_rows_pruned=0, predicate_cache_inner_records=3, predicate_cache_records=3, scan_efficiency_ratio=13.4% (141/1.05 K)] statement ok reset datafusion.explain.analyze_categories; @@ -740,12 +740,12 @@ INNER JOIN ( ---- Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(a@0, a@0)], projection=[a@0, min_value@2], metrics=[output_rows=2, output_batches=2, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=2, input_rows=2, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/join_agg_build.parquet]]}, projection=[a], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=14.45% (64/443)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/join_agg_build.parquet]]}, projection=[a], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=14.45% (64/443)] 03)--ProjectionExec: expr=[a@0 as a, min(join_agg_probe.value)@1 as min_value], metrics=[output_rows=2, output_batches=2] 04)----AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[min(join_agg_probe.value)], metrics=[output_rows=2, output_batches=2, spill_count=0, spilled_rows=0] 05)------RepartitionExec: partitioning=Hash([a@0], 4), input_partitions=1, metrics=[output_rows=2, output_batches=2, spill_count=0, spilled_rows=0] 06)--------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[min(join_agg_probe.value)], metrics=[output_rows=2, output_batches=1, spill_count=0, spilled_rows=0, skipped_aggregation_rows=0, reduction_factor=100% (2/2)] -07)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/join_agg_probe.parquet]]}, projection=[a, value], file_type=parquet, predicate=DynamicFilter [ a@0 >= h1 AND a@0 <= h2 AND a@0 IN (SET) ([h1, h2]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= h1 AND a_null_count@1 != row_count@2 AND a_min@3 <= h2 AND (a_null_count@1 != row_count@2 AND a_min@3 <= h1 AND h1 <= a_max@0 OR a_null_count@1 != row_count@2 AND a_min@3 <= h2 AND h2 <= a_max@0), required_guarantees=[a in (h1, h2)], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=4 total → 4 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=4, predicate_cache_records=2, scan_efficiency_ratio=19.43% (151/777)] +07)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/join_agg_probe.parquet]]}, projection=[a, value], file_type=parquet, predicate=DynamicFilter [ a@0 >= h1 AND a@0 <= h2 AND a@0 IN (SET) ([h1, h2]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= h1 AND a_null_count@1 != row_count@2 AND a_min@3 <= h2 AND (a_null_count@1 != row_count@2 AND a_min@3 <= h1 AND h1 <= a_max@0 OR a_null_count@1 != row_count@2 AND a_min@3 <= h2 AND h2 <= a_max@0), required_guarantees=[a in (h1, h2)], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=4 total → 4 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=4, predicate_cache_records=2, scan_efficiency_ratio=19.43% (151/777)] statement ok reset datafusion.explain.analyze_categories; @@ -807,8 +807,8 @@ ON nulls_build.a = nulls_probe.a AND nulls_build.b = nulls_probe.b; ---- Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(a@0, a@0), (b@1, b@1)], metrics=[output_rows=1, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=3, input_batches=1, input_rows=1, avg_fanout=100% (1/1), probe_hit_rate=100% (1/1)] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nulls_build.parquet]]}, projection=[a, b], file_type=parquet, metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=18.6% (144/774)] -03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nulls_probe.parquet]]}, projection=[a, b, c], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= 1 AND b@1 <= 2 AND struct(a@0, b@1) IN (SET) ([{c0:aa,c1:1}, {c0:,c1:2}, {c0:ab,c1:}]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= 1 AND b_null_count@5 != row_count@2 AND b_min@6 <= 2, required_guarantees=[], metrics=[output_rows=1, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=1, pushdown_rows_pruned=3, predicate_cache_inner_records=8, predicate_cache_records=2, scan_efficiency_ratio=20.45% (225/1.10 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nulls_build.parquet]]}, projection=[a, b], file_type=parquet, metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=18.6% (144/774)] +03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nulls_probe.parquet]]}, projection=[a, b, c], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= 1 AND b@1 <= 2 AND struct(a@0, b@1) IN (SET) ([{c0:aa,c1:1}, {c0:,c1:2}, {c0:ab,c1:}]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= 1 AND b_null_count@5 != row_count@2 AND b_min@6 <= 2, required_guarantees=[], metrics=[output_rows=1, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=1, pushdown_rows_pruned=3, predicate_cache_inner_records=8, predicate_cache_records=2, scan_efficiency_ratio=20.45% (225/1.10 K)] statement ok reset datafusion.explain.analyze_categories; @@ -873,8 +873,8 @@ ON lj_build.a = lj_probe.a AND lj_build.b = lj_probe.b; ---- Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=Left, on=[(a@0, a@0), (b@1, b@1)], metrics=[output_rows=2, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=2, input_rows=2, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/lj_build.parquet]]}, projection=[a, b, c], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=19.88% (196/986)] -03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/lj_probe.parquet]]}, projection=[a, b, e], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= ba AND b@1 <= bb AND struct(a@0, b@1) IN (SET) ([{c0:aa,c1:ba}, {c0:ab,c1:bb}]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= ba AND b_null_count@5 != row_count@2 AND b_min@6 <= bb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=22.37% (228/1.02 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/lj_build.parquet]]}, projection=[a, b, c], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=19.88% (196/986)] +03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/lj_probe.parquet]]}, projection=[a, b, e], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= ba AND b@1 <= bb AND struct(a@0, b@1) IN (SET) ([{c0:aa,c1:ba}, {c0:ab,c1:bb}]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= ba AND b_null_count@5 != row_count@2 AND b_min@6 <= bb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=22.37% (228/1.02 K)] # LEFT SEMI JOIN: only matching build rows are returned; probe scan still # receives the dynamic filter. @@ -889,8 +889,8 @@ WHERE EXISTS ( ---- Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=LeftSemi, on=[(a@0, a@0), (b@1, b@1)], metrics=[output_rows=2, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=2, input_rows=4, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/lj_build.parquet]]}, projection=[a, b, c], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=19.88% (196/986)] -03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/lj_probe.parquet]]}, projection=[a, b], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= ba AND b@1 <= bb AND struct(a@0, b@1) IN (SET) ([{c0:aa,c1:ba}, {c0:ab,c1:bb}]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= ba AND b_null_count@5 != row_count@2 AND b_min@6 <= bb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=15.11% (154/1.02 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/lj_build.parquet]]}, projection=[a, b, c], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=19.88% (196/986)] +03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/lj_probe.parquet]]}, projection=[a, b], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= ba AND b@1 <= bb AND struct(a@0, b@1) IN (SET) ([{c0:aa,c1:ba}, {c0:ab,c1:bb}]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= ba AND b_null_count@5 != row_count@2 AND b_min@6 <= bb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=15.11% (154/1.02 K)] statement ok reset datafusion.explain.analyze_categories; @@ -959,8 +959,8 @@ FROM hl_probe p INNER JOIN hl_build AS build ---- Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(a@0, a@0), (b@1, b@1)], projection=[a@3, b@4, c@2, e@5], metrics=[output_rows=2, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=1, input_rows=2, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/hl_build.parquet]]}, projection=[a, b, c], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=19.88% (196/986)] -03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/hl_probe.parquet]]}, projection=[a, b, e], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= ba AND b@1 <= bb AND hash_lookup ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= ba AND b_null_count@5 != row_count@2 AND b_min@6 <= bb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=22.37% (228/1.02 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/hl_build.parquet]]}, projection=[a, b, c], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=19.88% (196/986)] +03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/hl_probe.parquet]]}, projection=[a, b, e], file_type=parquet, predicate=DynamicFilter [ a@0 >= aa AND a@0 <= ab AND b@1 >= ba AND b@1 <= bb AND hash_lookup ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@1 != row_count@2 AND a_max@0 >= aa AND a_null_count@1 != row_count@2 AND a_min@3 <= ab AND b_null_count@5 != row_count@2 AND b_max@4 >= ba AND b_null_count@5 != row_count@2 AND b_min@6 <= bb, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=22.37% (228/1.02 K)] statement ok drop table hl_build; @@ -1008,8 +1008,8 @@ FROM int_build b INNER JOIN int_probe p ---- Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(id1@0, id1@0), (id2@1, id2@1)], projection=[id1@0, id2@1, value@2, data@5], metrics=[output_rows=2, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=1, input_rows=2, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/int_build.parquet]]}, projection=[id1, id2, value], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=18.48% (204/1.10 K)] -03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/int_probe.parquet]]}, projection=[id1, id2, data], file_type=parquet, predicate=DynamicFilter [ id1@0 >= 1 AND id1@0 <= 2 AND id2@1 >= 10 AND id2@1 <= 20 AND hash_lookup ], dynamic_rg_pruning=eligible, pruning_predicate=id1_null_count@1 != row_count@2 AND id1_max@0 >= 1 AND id1_null_count@1 != row_count@2 AND id1_min@3 <= 2 AND id2_null_count@5 != row_count@2 AND id2_max@4 >= 10 AND id2_null_count@5 != row_count@2 AND id2_min@6 <= 20, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=20.67% (221/1.07 K)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/int_build.parquet]]}, projection=[id1, id2, value], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=18.48% (204/1.10 K)] +03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/int_probe.parquet]]}, projection=[id1, id2, data], file_type=parquet, predicate=DynamicFilter [ id1@0 >= 1 AND id1@0 <= 2 AND id2@1 >= 10 AND id2@1 <= 20 AND hash_lookup ], dynamic_rg_pruning=eligible, pruning_predicate=id1_null_count@1 != row_count@2 AND id1_max@0 >= 1 AND id1_null_count@1 != row_count@2 AND id1_min@3 <= 2 AND id2_null_count@5 != row_count@2 AND id2_max@4 >= 10 AND id2_null_count@5 != row_count@2 AND id2_min@6 <= 20, required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=2, predicate_cache_inner_records=8, predicate_cache_records=4, scan_efficiency_ratio=20.67% (221/1.07 K)] statement ok reset datafusion.explain.analyze_categories; @@ -1060,8 +1060,8 @@ EXPLAIN ANALYZE SELECT nej_build.id, nej_probe.id FROM nej_build JOIN nej_probe ---- Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(id@0, id@0)], NullsEqual: true, metrics=[output_rows=2, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=1, input_rows=2, avg_fanout=100% (2/2), probe_hit_rate=100% (2/2)] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nej_build.parquet]]}, projection=[id], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=12.92% (65/503)] -03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nej_probe.parquet]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ id@0 IS NULL OR id@0 >= 11 AND id@0 <= 11 AND id@0 IN (SET) ([11, NULL]) ], dynamic_rg_pruning=eligible, pruning_predicate=id_null_count@0 > 0 OR id_null_count@0 != row_count@2 AND id_max@1 >= 11 AND id_null_count@0 != row_count@2 AND id_min@3 <= 11 AND (id_null_count@0 != row_count@2 AND id_min@3 <= 11 AND 11 <= id_max@1 OR id_null_count@0 != row_count@2 AND id_min@3 <= NULL AND NULL <= id_max@1), required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=3 total → 3 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=1, predicate_cache_inner_records=3, predicate_cache_records=3, scan_efficiency_ratio=14.45% (74/512)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nej_build.parquet]]}, projection=[id], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=12.92% (65/503)] +03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nej_probe.parquet]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ id@0 IS NULL OR id@0 >= 11 AND id@0 <= 11 AND id@0 IN (SET) ([11, NULL]) ], dynamic_rg_pruning=eligible, pruning_predicate=id_null_count@0 > 0 OR id_null_count@0 != row_count@2 AND id_max@1 >= 11 AND id_null_count@0 != row_count@2 AND id_min@3 <= 11 AND (id_null_count@0 != row_count@2 AND id_min@3 <= 11 AND 11 <= id_max@1 OR id_null_count@0 != row_count@2 AND id_min@3 <= NULL AND NULL <= id_max@1), required_guarantees=[], metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=3 total → 3 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=2, pushdown_rows_pruned=1, predicate_cache_inner_records=3, predicate_cache_records=3, scan_efficiency_ratio=14.45% (74/512)] statement ok reset datafusion.explain.analyze_categories; @@ -1103,8 +1103,8 @@ EXPLAIN ANALYZE SELECT mnej_build.a, mnej_build.b, mnej_probe.a, mnej_probe.b FR ---- Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(a@0, a@0), (b@1, b@1)], NullsEqual: true, metrics=[output_rows=2, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=1, input_rows=3, avg_fanout=100% (2/2), probe_hit_rate=66.67% (2/3)] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/mnej_build.parquet]]}, projection=[a, b], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=16.42% (133/810)] -03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/mnej_probe.parquet]]}, projection=[a, b], file_type=parquet, predicate=DynamicFilter [ a@0 IS NULL OR b@1 IS NULL OR a@0 >= 1 AND a@0 <= 2 AND b@1 >= 10 AND b@1 <= 10 AND struct(a@0, b@1) IN (SET) ([{c0:1,c1:10}, {c0:2,c1:}]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@0 > 0 OR b_null_count@1 > 0 OR a_null_count@0 != row_count@3 AND a_max@2 >= 1 AND a_null_count@0 != row_count@3 AND a_min@4 <= 2 AND b_null_count@1 != row_count@3 AND b_max@5 >= 10 AND b_null_count@1 != row_count@3 AND b_min@6 <= 10, required_guarantees=[], metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=3, pushdown_rows_pruned=0, predicate_cache_inner_records=6, predicate_cache_records=6, scan_efficiency_ratio=18.16% (148/815)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/mnej_build.parquet]]}, projection=[a, b], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=16.42% (133/810)] +03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/mnej_probe.parquet]]}, projection=[a, b], file_type=parquet, predicate=DynamicFilter [ a@0 IS NULL OR b@1 IS NULL OR a@0 >= 1 AND a@0 <= 2 AND b@1 >= 10 AND b@1 <= 10 AND struct(a@0, b@1) IN (SET) ([{c0:1,c1:10}, {c0:2,c1:}]) ], dynamic_rg_pruning=eligible, pruning_predicate=a_null_count@0 > 0 OR b_null_count@1 > 0 OR a_null_count@0 != row_count@3 AND a_max@2 >= 1 AND a_null_count@0 != row_count@3 AND a_min@4 <= 2 AND b_null_count@1 != row_count@3 AND b_max@5 >= 10 AND b_null_count@1 != row_count@3 AND b_min@6 <= 10, required_guarantees=[], metrics=[output_rows=3, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=3, pushdown_rows_pruned=0, predicate_cache_inner_records=6, predicate_cache_records=6, scan_efficiency_ratio=18.16% (148/815)] statement ok reset datafusion.explain.analyze_categories; @@ -1144,8 +1144,8 @@ EXPLAIN ANALYZE SELECT nnb_build.id, nnb_probe.id FROM nnb_build JOIN nnb_probe ---- Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(id@0, id@0)], NullsEqual: true, metrics=[output_rows=1, output_batches=1, array_map_created_count=1, build_input_batches=1, build_input_rows=2, input_batches=1, input_rows=1, avg_fanout=100% (1/1), probe_hit_rate=100% (1/1)] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nnb_build.parquet]]}, projection=[id], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=13.71% (68/496)] -03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nnb_probe.parquet]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ id@0 >= 11 AND id@0 <= 22 AND id@0 IN (SET) ([11, 22]) ], dynamic_rg_pruning=eligible, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 >= 11 AND id_null_count@1 != row_count@2 AND id_min@3 <= 22 AND (id_null_count@1 != row_count@2 AND id_min@3 <= 11 AND 11 <= id_max@0 OR id_null_count@1 != row_count@2 AND id_min@3 <= 22 AND 22 <= id_max@0), required_guarantees=[id in (11, 22)], metrics=[output_rows=1, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=3 total → 3 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=1, pushdown_rows_pruned=2, predicate_cache_inner_records=3, predicate_cache_records=1, scan_efficiency_ratio=14.45% (74/512)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nnb_build.parquet]]}, projection=[id], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=13.71% (68/496)] +03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nnb_probe.parquet]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ id@0 >= 11 AND id@0 <= 22 AND id@0 IN (SET) ([11, 22]) ], dynamic_rg_pruning=eligible, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 >= 11 AND id_null_count@1 != row_count@2 AND id_min@3 <= 22 AND (id_null_count@1 != row_count@2 AND id_min@3 <= 11 AND 11 <= id_max@0 OR id_null_count@1 != row_count@2 AND id_min@3 <= 22 AND 22 <= id_max@0), required_guarantees=[id in (11, 22)], metrics=[output_rows=1, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=3 total → 3 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=1, pushdown_rows_pruned=2, predicate_cache_inner_records=3, predicate_cache_records=1, scan_efficiency_ratio=14.45% (74/512)] statement ok reset datafusion.explain.analyze_categories; @@ -1185,8 +1185,8 @@ EXPLAIN ANALYZE SELECT nnp_build.id, nnp_probe.id FROM nnp_build JOIN nnp_probe ---- Plan with Metrics 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(id@0, id@0)], NullsEqual: true, metrics=[output_rows=1, output_batches=1, array_map_created_count=0, build_input_batches=1, build_input_rows=2, input_batches=1, input_rows=1, avg_fanout=100% (1/1), probe_hit_rate=100% (1/1)] -02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nnp_build.parquet]]}, projection=[id], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=12.92% (65/503)] -03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nnp_probe.parquet]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ id@0 >= 11 AND id@0 <= 11 AND id@0 IN (SET) ([11, NULL]) ], dynamic_rg_pruning=eligible, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 >= 11 AND id_null_count@1 != row_count@2 AND id_min@3 <= 11 AND (id_null_count@1 != row_count@2 AND id_min@3 <= 11 AND 11 <= id_max@0 OR id_null_count@1 != row_count@2 AND id_min@3 <= NULL AND NULL <= id_max@0), required_guarantees=[id in (11, NULL)], metrics=[output_rows=1, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=2 total → 2 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, predicate_evaluation_errors=0, pushdown_rows_matched=1, pushdown_rows_pruned=1, predicate_cache_inner_records=2, predicate_cache_records=1, scan_efficiency_ratio=13.71% (68/496)] +02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nnp_build.parquet]]}, projection=[id], file_type=parquet, metrics=[output_rows=2, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=0 total → 0 matched, page_index_rows_pruned=0 total → 0 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=0, pushdown_rows_pruned=0, predicate_cache_inner_records=0, predicate_cache_records=0, scan_efficiency_ratio=12.92% (65/503)] +03)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/push_down_filter_parquet/nnp_probe.parquet]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ id@0 >= 11 AND id@0 <= 11 AND id@0 IN (SET) ([11, NULL]) ], dynamic_rg_pruning=eligible, pruning_predicate=id_null_count@1 != row_count@2 AND id_max@0 >= 11 AND id_null_count@1 != row_count@2 AND id_min@3 <= 11 AND (id_null_count@1 != row_count@2 AND id_min@3 <= 11 AND 11 <= id_max@0 OR id_null_count@1 != row_count@2 AND id_min@3 <= NULL AND NULL <= id_max@0), required_guarantees=[id in (11, NULL)], metrics=[output_rows=1, output_batches=1, files_ranges_pruned_statistics=1 total → 1 matched, row_groups_pruned_statistics=1 total → 1 matched, row_groups_pruned_bloom_filter=1 total → 1 matched, page_index_pages_pruned=1 total → 1 matched, page_index_rows_pruned=2 total → 2 matched, limit_pruned_row_groups=0 total → 0 matched, batches_split=0, file_open_errors=0, file_scan_errors=0, files_opened=1, files_processed=1, num_predicate_creation_errors=0, post_scan_rows_matched=0, post_scan_rows_pruned=0, predicate_evaluation_errors=0, pushdown_rows_matched=1, pushdown_rows_pruned=1, predicate_cache_inner_records=2, predicate_cache_records=1, scan_efficiency_ratio=13.71% (68/496)] statement ok reset datafusion.explain.analyze_categories; diff --git a/datafusion/sqllogictest/test_files/range_partitioning.slt b/datafusion/sqllogictest/test_files/range_partitioning.slt index ec374b3d62a28..82d64f7df5d5f 100644 --- a/datafusion/sqllogictest/test_files/range_partitioning.slt +++ b/datafusion/sqllogictest/test_files/range_partitioning.slt @@ -342,8 +342,7 @@ ON l.range_key = r.range_key; physical_plan 01)HashJoinExec: mode=Partitioned, join_type=Left, on=[(range_key@0, range_key@0)], projection=[range_key@0, value@1, value@3] 02)--DataSourceExec: file_groups=, projection=[range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet -03)--FilterExec: value@1 <= 150 -04)----DataSourceExec: file_groups=, projection=[range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet +03)--DataSourceExec: file_groups=, projection=[range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet, predicate=value@2 <= 150, pruning_predicate=value_null_count@1 != row_count@2 AND value_min@0 <= 150, required_guarantees=[] query III SELECT l.range_key, l.value, r.value @@ -370,8 +369,7 @@ ON l.range_key = r.range_key; physical_plan 01)HashJoinExec: mode=Partitioned, join_type=LeftSemi, on=[(range_key@0, range_key@0)] 02)--DataSourceExec: file_groups=, projection=[range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet -03)--FilterExec: value@1 <= 150, projection=[range_key@0] -04)----DataSourceExec: file_groups=, projection=[range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet +03)--DataSourceExec: file_groups=, projection=[range_key], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet, predicate=value@2 <= 150, pruning_predicate=value_null_count@1 != row_count@2 AND value_min@0 <= 150, required_guarantees=[] query II SELECT l.range_key, l.value @@ -394,8 +392,7 @@ ON l.range_key = r.range_key; physical_plan 01)HashJoinExec: mode=Partitioned, join_type=LeftAnti, on=[(range_key@0, range_key@0)] 02)--DataSourceExec: file_groups=, projection=[range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet -03)--FilterExec: value@1 <= 150, projection=[range_key@0] -04)----DataSourceExec: file_groups=, projection=[range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet +03)--DataSourceExec: file_groups=, projection=[range_key], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet, predicate=value@2 <= 150, pruning_predicate=value_null_count@1 != row_count@2 AND value_min@0 <= 150, required_guarantees=[] query II SELECT l.range_key, l.value @@ -428,8 +425,7 @@ physical_plan 02)--RepartitionExec: partitioning=Hash([range_key@0, non_range_key@1], 4), input_partitions=4 03)----DataSourceExec: file_groups=, projection=[range_key, non_range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet 04)--RepartitionExec: partitioning=Hash([range_key@0, non_range_key@1], 4), input_partitions=4 -05)----FilterExec: value@2 <= 150 -06)------DataSourceExec: file_groups=, projection=[range_key, non_range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet +05)----DataSourceExec: file_groups=, projection=[range_key, non_range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet, predicate=value@2 <= 150, pruning_predicate=value_null_count@1 != row_count@2 AND value_min@0 <= 150, required_guarantees=[] # Range([range_key]) does not satisfy a join keyed on non_range_key. query TT @@ -508,8 +504,7 @@ physical_plan 01)FilterExec: non_range_key@1 = 2 OR mark@3, projection=[range_key@0, value@2] 02)--HashJoinExec: mode=Partitioned, join_type=LeftMark, on=[(range_key@0, range_key@0)] 03)----DataSourceExec: file_groups=, projection=[range_key, non_range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet -04)----FilterExec: value@1 <= 150, projection=[range_key@0] -05)------DataSourceExec: file_groups=, projection=[range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet +04)----DataSourceExec: file_groups=, projection=[range_key], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet, predicate=value@2 <= 150, pruning_predicate=value_null_count@1 != row_count@2 AND value_min@0 <= 150, required_guarantees=[] query II SELECT l.range_key, l.value @@ -755,9 +750,8 @@ RIGHT JOIN range_partitioned r ON l.range_key = r.range_key; ---- physical_plan 01)HashJoinExec: mode=Partitioned, join_type=Right, on=[(range_key@0, range_key@0)], projection=[value@1, range_key@2, value@3] -02)--FilterExec: value@1 <= 150 -03)----DataSourceExec: file_groups=, projection=[range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet, predicate=value@2 <= 150, pruning_predicate=value_null_count@1 != row_count@2 AND value_min@0 <= 150, required_guarantees=[] -04)--DataSourceExec: file_groups=, projection=[range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet +02)--DataSourceExec: file_groups=, projection=[range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet, predicate=value@2 <= 150, pruning_predicate=value_null_count@1 != row_count@2 AND value_min@0 <= 150, required_guarantees=[] +03)--DataSourceExec: file_groups=, projection=[range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet query III SELECT l.value, r.range_key, r.value @@ -787,9 +781,8 @@ RIGHT SEMI JOIN range_partitioned r ON l.range_key = r.range_key; ---- physical_plan 01)HashJoinExec: mode=Partitioned, join_type=RightSemi, on=[(range_key@0, range_key@0)] -02)--FilterExec: value@1 <= 150, projection=[range_key@0] -03)----DataSourceExec: file_groups=, projection=[range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet, predicate=value@2 <= 150, pruning_predicate=value_null_count@1 != row_count@2 AND value_min@0 <= 150, required_guarantees=[] -04)--DataSourceExec: file_groups=, projection=[range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet +02)--DataSourceExec: file_groups=, projection=[range_key], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet, predicate=value@2 <= 150, pruning_predicate=value_null_count@1 != row_count@2 AND value_min@0 <= 150, required_guarantees=[] +03)--DataSourceExec: file_groups=, projection=[range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet query II SELECT r.range_key, r.value @@ -815,9 +808,8 @@ RIGHT ANTI JOIN range_partitioned r ON l.range_key = r.range_key; ---- physical_plan 01)HashJoinExec: mode=Partitioned, join_type=RightAnti, on=[(range_key@0, range_key@0)] -02)--FilterExec: value@1 <= 150, projection=[range_key@0] -03)----DataSourceExec: file_groups=, projection=[range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet, predicate=value@2 <= 150, pruning_predicate=value_null_count@1 != row_count@2 AND value_min@0 <= 150, required_guarantees=[] -04)--DataSourceExec: file_groups=, projection=[range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet +02)--DataSourceExec: file_groups=, projection=[range_key], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet, predicate=value@2 <= 150, pruning_predicate=value_null_count@1 != row_count@2 AND value_min@0 <= 150, required_guarantees=[] +03)--DataSourceExec: file_groups=, projection=[range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet query II SELECT r.range_key, r.value @@ -845,10 +837,9 @@ RIGHT JOIN range_partitioned_shifted r ON l.range_key = r.range_key; physical_plan 01)HashJoinExec: mode=Partitioned, join_type=Right, on=[(range_key@0, range_key@0)], projection=[value@1, range_key@2, value@3] 02)--RepartitionExec: partitioning=Hash([range_key@0], 4), input_partitions=4 -03)----FilterExec: value@1 <= 150 -04)------DataSourceExec: file_groups=, projection=[range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet, predicate=value@2 <= 150, pruning_predicate=value_null_count@1 != row_count@2 AND value_min@0 <= 150, required_guarantees=[] -05)--RepartitionExec: partitioning=Hash([range_key@0], 4), input_partitions=4 -06)----DataSourceExec: file_groups=, projection=[range_key, value], output_partitioning=Range([range_key@0 ASC], [(15), (20), (30)], 4), file_type=parquet +03)----DataSourceExec: file_groups=, projection=[range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet, predicate=value@2 <= 150, pruning_predicate=value_null_count@1 != row_count@2 AND value_min@0 <= 150, required_guarantees=[] +04)--RepartitionExec: partitioning=Hash([range_key@0], 4), input_partitions=4 +05)----DataSourceExec: file_groups=, projection=[range_key, value], output_partitioning=Range([range_key@0 ASC], [(15), (20), (30)], 4), file_type=parquet query III SELECT l.value, r.range_key, r.value @@ -952,10 +943,9 @@ RIGHT JOIN range_partitioned r ON l.non_range_key = r.non_range_key; physical_plan 01)HashJoinExec: mode=Partitioned, join_type=Right, on=[(non_range_key@0, non_range_key@1)], projection=[value@1, range_key@2, value@4] 02)--RepartitionExec: partitioning=Hash([non_range_key@0], 4), input_partitions=4 -03)----FilterExec: range_key@0 < 10, projection=[non_range_key@1, value@2] -04)------DataSourceExec: file_groups=, projection=[range_key, non_range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet, predicate=range_key@0 < 10, pruning_predicate=range_key_null_count@1 != row_count@2 AND range_key_min@0 < 10, required_guarantees=[] -05)--RepartitionExec: partitioning=Hash([non_range_key@1], 4), input_partitions=4 -06)----DataSourceExec: file_groups=, projection=[range_key, non_range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet +03)----DataSourceExec: file_groups=, projection=[non_range_key, value], output_partitioning=UnknownPartitioning(4), file_type=parquet, predicate=range_key@0 < 10, pruning_predicate=range_key_null_count@1 != row_count@2 AND range_key_min@0 < 10, required_guarantees=[] +04)--RepartitionExec: partitioning=Hash([non_range_key@1], 4), input_partitions=4 +05)----DataSourceExec: file_groups=, projection=[range_key, non_range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet query III SELECT l.value, r.range_key, r.value @@ -988,8 +978,7 @@ physical_plan 01)FilterExec: non_range_key@1 = 2 OR mark@3, projection=[range_key@0, value@2] 02)--HashJoinExec: mode=Partitioned, join_type=LeftMark, on=[(range_key@0, range_key@0)] 03)----DataSourceExec: file_groups=, projection=[range_key, non_range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet -04)----FilterExec: value@1 <= 150, projection=[range_key@0] -05)------DataSourceExec: file_groups=, projection=[range_key, value], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet +04)----DataSourceExec: file_groups=, projection=[range_key], output_partitioning=Range([range_key@0 ASC], [(10), (20), (30)], 4), file_type=parquet, predicate=value@2 <= 150, pruning_predicate=value_null_count@1 != row_count@2 AND value_min@0 <= 150, required_guarantees=[] # Matched rows have mark=true and are returned; unmatched rows have # mark=false and are only returned when non_range_key = 2. diff --git a/datafusion/sqllogictest/test_files/range_sorted_time_bin_agg.slt b/datafusion/sqllogictest/test_files/range_sorted_time_bin_agg.slt index 18123a492dbd6..0a3ea2da5f353 100644 --- a/datafusion/sqllogictest/test_files/range_sorted_time_bin_agg.slt +++ b/datafusion/sqllogictest/test_files/range_sorted_time_bin_agg.slt @@ -100,8 +100,7 @@ physical_plan 02)--AggregateExec: mode=FinalPartitioned, gby=[key@0 as key, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)], aggr=[sum(range_sorted_time_bin.value)], ordering_mode=Sorted 03)----RepartitionExec: partitioning=Hash([key@0, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1], 2), input_partitions=2, preserve_order=true, sort_exprs=key@0 ASC, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1 ASC 04)------AggregateExec: mode=Partial, gby=[key@0 as key, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }, timestamp@1) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)], aggr=[sum(range_sorted_time_bin.value)], ordering_mode=Sorted -05)--------FilterExec: col4@1 = a, projection=[key@0, timestamp@2, value@3] -06)----------DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch_range_partitioning/range_sorted_time_bin/part-0.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch_range_partitioning/range_sorted_time_bin/part-1.parquet]]}, projection=[key, col4, timestamp, value], output_ordering=[key@0 ASC, timestamp@2 ASC], output_partitioning=Range([timestamp@2 ASC], [(1704070800000000000)], 2), file_type=parquet, predicate=col4@4 = a, pruning_predicate=col4_null_count@2 != row_count@3 AND col4_min@0 <= a AND a <= col4_max@1, required_guarantees=[col4 in (a)] +05)--------DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch_range_partitioning/range_sorted_time_bin/part-0.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch_range_partitioning/range_sorted_time_bin/part-1.parquet]]}, projection=[key, timestamp, value], output_ordering=[key@0 ASC, timestamp@1 ASC], output_partitioning=Range([timestamp@1 ASC], [(1704070800000000000)], 2), file_type=parquet, predicate=col4@4 = a, pruning_predicate=col4_null_count@2 != row_count@3 AND col4_min@0 <= a AND a <= col4_max@1, required_guarantees=[col4 in (a)] query TPI SELECT key, date_bin(INTERVAL '60 seconds', timestamp) AS time_bin, sum(value) From 31199fcee93a3fb96117854ec5ad3beb554f3bad Mon Sep 17 00:00:00 2001 From: Adrian Garcia Badaracco <1755071+adriangb@users.noreply.github.com> Date: Thu, 24 Sep 2026 23:39:42 -0500 Subject: [PATCH 03/42] fix: stop the Parquet push decoder stream after the final flush When the TopK dynamic filter prunes every remaining row group at a row group boundary, `rebuild_decoder_at_boundary` returns `Ok(true)` and the stream calls `finish()`. With a batch coalescer (a post-scan filter is present, for example with the default `pushdown_filters = false`), `finish()` flushes the coalescer and returns a batch. The next poll then went back to the decoder, which still pointed at a row group that the plan had dropped, and `sync_rg_plan_to_decoder_frontier` failed with "push decoder frontier RG N is not in rg_plan; decoder and plan have diverged". ClickBench Q23, Q24 and Q26 fail with this error. After the flush, the stream now only drains the coalescer. The new sqllogictest in `dynamic_row_group_pruning.slt` fails without the fix. Co-Authored-By: Claude Opus 5.5 --- .../datasource-parquet/src/push_decoder.rs | 10 ++++ .../test_files/dynamic_row_group_pruning.slt | 50 +++++++++++++++++++ 2 files changed, 60 insertions(+) diff --git a/datafusion/datasource-parquet/src/push_decoder.rs b/datafusion/datasource-parquet/src/push_decoder.rs index a811212a3dfe3..e38cec26da65a 100644 --- a/datafusion/datasource-parquet/src/push_decoder.rs +++ b/datafusion/datasource-parquet/src/push_decoder.rs @@ -481,6 +481,16 @@ impl PushDecoderStreamState { // every return. let elapsed_compute = self.baseline_metrics.elapsed_compute().clone(); let mut timer = elapsed_compute.timer(); + // Once `finish` has flushed the coalescer, the stream only drains it. + // The decoder can still point at row groups that the plan dropped + // (for example, when a dynamic filter pruned every remaining row + // group at a boundary), so it must not be driven again. + if self.flushed { + if self.remaining_limit == Some(0) { + return None; + } + return self.emit_completed(); + } loop { // Hand out anything the coalescer has already assembled into a // full-size batch before doing more decoding work. diff --git a/datafusion/sqllogictest/test_files/dynamic_row_group_pruning.slt b/datafusion/sqllogictest/test_files/dynamic_row_group_pruning.slt index c674dede75706..fec9add854b3a 100644 --- a/datafusion/sqllogictest/test_files/dynamic_row_group_pruning.slt +++ b/datafusion/sqllogictest/test_files/dynamic_row_group_pruning.slt @@ -329,3 +329,53 @@ set datafusion.execution.target_partitions = 4; statement ok RESET datafusion.execution.parquet.pushdown_filters; + +# Regression test: the TopK dynamic filter prunes every remaining row group at a +# row group boundary while the in-scan post-scan filter (`pushdown_filters = +# false`, the default) holds rows in its batch coalescer. The stream flushes the +# coalescer and must then only drain it. Before the fix the next poll went back +# to the decoder, which still pointed at a row group that the plan had dropped, +# and failed with "decoder and plan have diverged". +# Layout: 10 row groups of 1000 rows, row group k holds `ts` k*1000..k*1000+999 +# in a scrambled order, so the first row group fills the TopK heap and its +# threshold prunes all other row groups. +statement ok +set datafusion.execution.target_partitions = 1; + +statement ok +COPY ( + SELECT (value / 1000) * 1000 + ((value * 7919) % 1000) AS ts, 'x' AS s + FROM generate_series(0, 9999) +) +TO 'test_files/scratch/dynamic_row_group_pruning/flush.parquet' +STORED AS PARQUET +OPTIONS ('format.max_row_group_size' '1000'); + +statement ok +CREATE EXTERNAL TABLE flush_t +STORED AS PARQUET +LOCATION 'test_files/scratch/dynamic_row_group_pruning/flush.parquet'; + +query IIII +SELECT count(*), count(DISTINCT ts), min(ts), max(ts) +FROM (SELECT * FROM flush_t WHERE s <> '' ORDER BY ts LIMIT 500); +---- +500 500 0 499 + +statement ok +set datafusion.execution.parquet.pushdown_filters = true; + +query IIII +SELECT count(*), count(DISTINCT ts), min(ts), max(ts) +FROM (SELECT * FROM flush_t WHERE s <> '' ORDER BY ts LIMIT 500); +---- +500 500 0 499 + +statement ok +drop table flush_t; + +statement ok +set datafusion.execution.target_partitions = 4; + +statement ok +RESET datafusion.execution.parquet.pushdown_filters; From c2d8f85980716908689922b349bce16bd9ea11bc Mon Sep 17 00:00:00 2001 From: Adrian Garcia Badaracco <1755071+adriangb@users.noreply.github.com> Date: Fri, 25 Sep 2026 20:04:41 -0500 Subject: [PATCH 04/42] refactor(physical-plan): share the round-robin row check of EnforceDistribution Move the check "a round-robin repartition of this input is useful for its row count" from `EnforceDistribution` into `repartition::round_robin_beneficial_for_rows`. The behavior does not change. The next commit uses the same check in the file scan, so that the scan and the optimizer make the same decision. PR: #22384 Co-Authored-By: Claude Opus 5.5 --- .../enforce_distribution.rs | 34 +++++++------------ .../physical-plan/src/repartition/mod.rs | 24 +++++++++++++ 2 files changed, 36 insertions(+), 22 deletions(-) diff --git a/datafusion/physical-optimizer/src/ensure_requirements/enforce_distribution.rs b/datafusion/physical-optimizer/src/ensure_requirements/enforce_distribution.rs index 776e3c3bf14bd..fb62ea0e1b8b3 100644 --- a/datafusion/physical-optimizer/src/ensure_requirements/enforce_distribution.rs +++ b/datafusion/physical-optimizer/src/ensure_requirements/enforce_distribution.rs @@ -41,7 +41,6 @@ use crate::utils::{ use arrow::compute::SortOptions; use datafusion_common::config::ConfigOptions; use datafusion_common::error::Result; -use datafusion_common::stats::Precision; use datafusion_common::tree_node::Transformed; use datafusion_expr::logical_plan::{Aggregate, JoinType}; use datafusion_physical_expr::expressions::{Column, NoOp}; @@ -62,7 +61,9 @@ use datafusion_physical_plan::joins::{ CrossJoinExec, HashJoinExec, PartitionMode, SortMergeJoinExec, }; use datafusion_physical_plan::projection::{ProjectionExec, ProjectionExpr}; -use datafusion_physical_plan::repartition::RepartitionExec; +use datafusion_physical_plan::repartition::{ + RepartitionExec, round_robin_beneficial_for_rows, +}; use datafusion_physical_plan::sorts::sort::SortExec; use datafusion_physical_plan::sorts::sort_preserving_merge::SortPreservingMergeExec; use datafusion_physical_plan::statistics::{StatisticsArgs, StatisticsContext}; @@ -995,8 +996,7 @@ struct DistributionChildState { )] fn get_repartition_requirement_status( plan: &Arc, - batch_size: usize, - should_use_estimates: bool, + config: &ConfigOptions, stats_ctx: &StatisticsContext, ) -> Result> { let mut needs_alignment = false; @@ -1009,14 +1009,12 @@ fn get_repartition_requirement_status( { // Decide whether adding a round robin is beneficial depending on // the statistical information we have on the number of rows: - let roundrobin_beneficial_stats = match stats_ctx - .compute(child.as_ref(), &StatisticsArgs::new())? - .num_rows - { - Precision::Exact(n_rows) => n_rows > batch_size, - Precision::Inexact(n_rows) => !should_use_estimates || (n_rows > batch_size), - Precision::Absent => true, - }; + let roundrobin_beneficial_stats = round_robin_beneficial_for_rows( + &stats_ctx + .compute(child.as_ref(), &StatisticsArgs::new())? + .num_rows, + config, + ); let is_hash = matches!( requirement, Distribution::HashPartitioned(_) | Distribution::KeyPartitioned(_) @@ -1375,10 +1373,6 @@ pub fn ensure_distribution_with_stats( // When `false`, round robin repartition will not be added to increase parallelism let enable_round_robin = config.optimizer.enable_round_robin_repartition; let repartition_file_scans = config.optimizer.repartition_file_scans; - let batch_size = config.execution.batch_size.get(); - let should_use_estimates = config - .execution - .use_row_number_estimates_to_optimize_partitioning; let subset_satisfaction_threshold = config.optimizer.subset_repartition_threshold; let unbounded_and_pipeline_friendly = dist_context.plan.boundedness().is_unbounded() && matches!( @@ -1455,12 +1449,8 @@ pub fn ensure_distribution_with_stats( || plan.is::(); let input_distributions = plan.input_distribution_requirements(); - let repartition_status_flags = get_repartition_requirement_status( - &plan, - batch_size, - should_use_estimates, - stats_ctx, - )?; + let repartition_status_flags = + get_repartition_requirement_status(&plan, config, stats_ctx)?; // This loop iterates over all the children to: // - Increase parallelism for every child if it is beneficial. // - Satisfy the distribution requirements of every child, if it is not diff --git a/datafusion/physical-plan/src/repartition/mod.rs b/datafusion/physical-plan/src/repartition/mod.rs index 794888b4ea827..773fe6869eac1 100644 --- a/datafusion/physical-plan/src/repartition/mod.rs +++ b/datafusion/physical-plan/src/repartition/mod.rs @@ -1384,6 +1384,30 @@ impl BatchPartitioner { } } +/// Returns `true` if the row count of an input makes a round-robin +/// repartition of that input useful: the input has more rows than one batch, +/// or its row count is unknown. +/// +/// The optimizer uses an inexact row count only when +/// `datafusion.execution.use_row_number_estimates_to_optimize_partitioning` +/// is set. Otherwise it treats an inexact row count as unknown. +pub fn round_robin_beneficial_for_rows( + num_rows: &Precision, + config: &ConfigOptions, +) -> bool { + let batch_size = config.execution.batch_size.get(); + match num_rows { + Precision::Exact(n_rows) => *n_rows > batch_size, + Precision::Inexact(n_rows) => { + !config + .execution + .use_row_number_estimates_to_optimize_partitioning + || *n_rows > batch_size + } + Precision::Absent => true, + } +} + /// Maps `N` input partitions to `M` output partitions based on a /// [`Partitioning`] scheme. /// From 9ff1146909c23ff7fec54fc00fff73ceefe40cab Mon Sep 17 00:00:00 2001 From: Adrian Garcia Badaracco <1755071+adriangb@users.noreply.github.com> Date: Fri, 25 Sep 2026 21:38:26 -0500 Subject: [PATCH 05/42] fix(datasource): keep a filter above a scan that cannot give the target partitions The Parquet scan accepts all pushable filters, thus `FilterPushdown` removes the `FilterExec`. For a scan of one small file (one partition, too small to split into byte ranges), main puts a round-robin `RepartitionExec` between the scan and the `FilterExec`, and a `CoalescePartitionsExec` above them. Without the `FilterExec`, the optimizer adds neither. The filter then runs in one partition, and the scans of the build sides of the hash joins run one after the other in the task of the probe side, not in parallel tasks. On TPC-DS SF1 this made short queries 5% to 30% slower than main (for example Q37 1.26x). `FileScanConfig::try_pushdown_filters` now makes the same decision as `EnforceDistribution`: if the scan has fewer than `target_partitions` partitions, `repartitioned` cannot give more, and a round-robin repartition is useful for the rows that the scan reads, the filters stay above the scan (`PushedDown::No`). The scan still gets them, through the new `FileSource::try_pushdown_pruning_filters`, and uses them only to prune files, row groups and pages. This is what main does with all filters when `pushdown_filters` is false. The plan is then the plan of main for these scans. - Only the filters of a `FilterExec` stay above the scan. A dynamic filter (of a join, a TopK or an aggregate) has no `FilterExec` above the scan, thus the scan applies it as before. - The default of `try_pushdown_pruning_filters` returns `None`: other file sources get their filters as before. - The Parquet scan applies all conjuncts of its predicate or none of them. A scan with a pruning-only predicate uses later filters only to prune too. - `ParquetScanExecNode` gets `pruning_only_predicate`, thus a decoded scan does not apply its predicate again. - An exact row count of at most one batch keeps the filter in the scan: a round-robin repartition cannot split one batch. Tests: - unit tests for the decision in `file_scan_config` and for the pruning-only predicate of `ParquetSource`; - a proto round trip of the pruning-only predicate; - a sqllogictest plan pin in `parquet_filter_pushdown.slt`; - `parquet_statistics.slt` (no statistics, thus unknown rows): the plan is the plan of main again; - two Parquet integration tests that check the filter in the scan use one target partition. PR: #22384 Co-Authored-By: Claude Opus 5.5 --- datafusion/core/tests/parquet/expr_adapter.rs | 6 +- .../core/tests/parquet/filter_pushdown.rs | 4 + .../datasource-parquet/src/opener/mod.rs | 14 +- datafusion/datasource-parquet/src/source.rs | 115 ++++++- datafusion/datasource/src/file.rs | 20 ++ .../datasource/src/file_scan_config/mod.rs | 312 +++++++++++++++++- .../proto-models/proto/datafusion.proto | 4 + .../proto-models/src/generated/pbjson.rs | 18 + .../proto-models/src/generated/prost.rs | 4 + datafusion/proto/tests/cases/plans/sources.rs | 43 +++ .../test_files/parquet_filter_pushdown.slt | 69 ++++ .../test_files/parquet_statistics.slt | 5 +- 12 files changed, 608 insertions(+), 6 deletions(-) diff --git a/datafusion/core/tests/parquet/expr_adapter.rs b/datafusion/core/tests/parquet/expr_adapter.rs index dfed2eb5bda74..c2668a0734d8a 100644 --- a/datafusion/core/tests/parquet/expr_adapter.rs +++ b/datafusion/core/tests/parquet/expr_adapter.rs @@ -875,10 +875,14 @@ async fn test_all_null_struct_decimal_cast_filter_pushdown() -> Result<()> { write_parquet(batch, Arc::clone(&store), "null_decimal/data.parquet").await; for pushdown_filters in [false, true] { + // One target partition: with more, the scan of one file keeps the + // filter in a `FilterExec` above it, which runs in more + // partitions. This test checks the filter in the scan. let mut config = SessionConfig::new() .with_collect_statistics(false) .with_parquet_pruning(false) - .with_parquet_page_index_pruning(false); + .with_parquet_page_index_pruning(false) + .with_target_partitions(1); config.options_mut().execution.parquet.pushdown_filters = pushdown_filters; let ctx = SessionContext::new_with_config(config); register_memory_listing_table( diff --git a/datafusion/core/tests/parquet/filter_pushdown.rs b/datafusion/core/tests/parquet/filter_pushdown.rs index d337979e5fd00..a450d62d1cc87 100644 --- a/datafusion/core/tests/parquet/filter_pushdown.rs +++ b/datafusion/core/tests/parquet/filter_pushdown.rs @@ -655,6 +655,10 @@ async fn predicate_cache_stats_issue_19561() -> datafusion_common::Result<()> { // force to get multiple batches to trigger repeated metric compound bug config.options_mut().execution.batch_size = datafusion_common::config::ConfigNonZeroUsize::try_new(1)?; + // One target partition: with more, the scan of one file keeps the + // filter in a `FilterExec` above it (8 rows are 8 batches), and this + // test checks the row filter of the scan. + config.options_mut().execution.target_partitions = 1; let ctx = SessionContext::new_with_config(config); // The cache is on by default, and used when filter pushdown is enabled PredicateCacheTest { diff --git a/datafusion/datasource-parquet/src/opener/mod.rs b/datafusion/datasource-parquet/src/opener/mod.rs index 9f28d1ef94e4a..5108ff93890f8 100644 --- a/datafusion/datasource-parquet/src/opener/mod.rs +++ b/datafusion/datasource-parquet/src/opener/mod.rs @@ -252,6 +252,9 @@ pub(super) struct ParquetMorselizer { pub preserve_order: bool, /// Optional predicate to apply during the scan pub predicate: Option>, + /// If true, the scan uses `predicate` only to prune: a `FilterExec` + /// above the scan applies it. + pub pruning_only_predicate: bool, /// Table schema, including partition columns. pub table_schema: TableSchema, /// Optional hint for how large the initial request to read parquet metadata @@ -455,6 +458,7 @@ struct PreparedParquetOpen { output_schema: SchemaRef, projection: ProjectionExprs, predicate: Option>, + pruning_only_predicate: bool, /// Per-scan virtual-column state, Arc-cloned from [`ParquetMorselizer`] so /// each file shares validated fields, precomputed null replacements, and /// the logical-with-virtual schema. `None` when no virtual columns were @@ -556,8 +560,14 @@ impl DecoderReadPlans { // // Either way every conjunct is applied; nothing is silently dropped. // --------------------------------------------------------------- + // A pruning-only predicate is applied by a `FilterExec` above the + // scan: the scan uses it only to prune. + let applied_predicate = prepared + .predicate + .as_ref() + .filter(|_| !prepared.pruning_only_predicate); let (row_filter_context, post_scan_conjuncts) = - match (prepared.pushdown_filters, prepared.predicate.as_ref()) { + match (prepared.pushdown_filters, applied_predicate) { // Pushdown enabled: precompute the candidate list once per file. // Both the initial `RowFilter` and any per-RG rebuilds (via // `RowFilterContext::build_row_filter`) reuse it, so tree walks @@ -1005,6 +1015,7 @@ impl ParquetMorselizer { output_schema, projection, predicate, + pruning_only_predicate: self.pruning_only_predicate, virtual_state: self.virtual_state.as_ref().map(Arc::clone), reorder_predicates: self.reorder_filters, pushdown_filters: self.pushdown_filters, @@ -2613,6 +2624,7 @@ mod test { limit: self.limit, preserve_order: self.preserve_order, predicate: self.predicate, + pruning_only_predicate: false, table_schema, metadata_size_hint: self.metadata_size_hint, metrics: self.metrics, diff --git a/datafusion/datasource-parquet/src/source.rs b/datafusion/datasource-parquet/src/source.rs index c1e7e47b459e3..686ca4f1ad5a4 100644 --- a/datafusion/datasource-parquet/src/source.rs +++ b/datafusion/datasource-parquet/src/source.rs @@ -304,6 +304,10 @@ pub struct ParquetSource { pub(crate) table_schema: TableSchema, /// Optional predicate for row filtering during parquet scan pub(crate) predicate: Option>, + /// If true, the scan uses [`Self::predicate`] only to prune: a + /// `FilterExec` above the scan applies it (see + /// [`FileSource::try_pushdown_pruning_filters`]). + pub(crate) pruning_only_predicate: bool, /// Optional user defined parquet file reader factory pub(crate) parquet_file_reader_factory: Option>, /// Optional policy for deriving each file's Arrow schema after footer loading. @@ -342,6 +346,7 @@ impl ParquetSource { table_parquet_options: TableParquetOptions::default(), metrics: ExecutionPlanMetricsSet::new(), predicate: None, + pruning_only_predicate: false, parquet_file_reader_factory: None, schema_provider: None, batch_size: None, @@ -644,7 +649,7 @@ impl FileSource for ParquetSource { self.table_schema.virtual_columns(), self.table_schema.file_schema(), self.predicate.as_ref(), - self.pushdown_filters(), + self.pushdown_filters() && !self.pruning_only_predicate, )?; Ok(Box::new(ParquetMorselizer { @@ -656,6 +661,7 @@ impl FileSource for ParquetSource { limit: base_config.limit, preserve_order: base_config.preserve_order, predicate: self.predicate.clone(), + pruning_only_predicate: self.pruning_only_predicate, table_schema: self.table_schema.clone(), metadata_size_hint: self.metadata_size_hint, metrics: self.metrics().clone(), @@ -894,6 +900,16 @@ impl FileSource for ParquetSource { None => conjunction(allowed_filters), }; source.predicate = Some(predicate); + if self.pruning_only_predicate { + // The scan applies all conjuncts of its predicate or none of + // them. It uses its predicate only to prune (a `FilterExec` + // above the scan applies the filters that it got before), thus it + // uses the new filters only to prune too. + return Ok(FilterPushdownPropagation::with_parent_pushdown_result( + vec![PushedDown::No; filters.len()], + ) + .with_updated_node(Arc::new(source) as _)); + } source = source.with_pushdown_filters(pushdown_filters); let source = Arc::new(source); // The parquet scan always accepts pushable filters: report each @@ -910,6 +926,36 @@ impl FileSource for ParquetSource { .with_updated_node(source)) } + /// Takes the pushable `filters` for pruning only, as the scan does with + /// all filters when `pushdown_filters` is false on main. The scan applies + /// either all conjuncts of its predicate or none of them, thus it refuses + /// (`None`) when it already has a predicate that it applies. + fn try_pushdown_pruning_filters( + &self, + filters: &[Arc], + _config: &ConfigOptions, + ) -> datafusion_common::Result>> { + if self.predicate.is_some() && !self.pruning_only_predicate { + return Ok(None); + } + let pushable_schema = self.table_schema.schema_without_virtual_columns(); + let pushable = filters + .iter() + .filter(|filter| { + can_expr_be_pushed_down_with_schemas(filter, pushable_schema) + }) + .cloned() + .collect_vec(); + if pushable.is_empty() { + return Ok(None); + } + let mut source = self.clone(); + source.predicate = + Some(conjunction(self.predicate.iter().cloned().chain(pushable))); + source.pruning_only_predicate = true; + Ok(Some(Arc::new(source))) + } + /// Try to optimize the scan to produce data in the requested sort order. /// /// Inputs: @@ -1112,6 +1158,7 @@ impl FileSource for ParquetSource { // Carried by `base`. table_schema: _, predicate, + pruning_only_predicate, // Rebuilt from the decode context. parquet_file_reader_factory: _, // Requires a custom codec for serialization. @@ -1158,6 +1205,7 @@ impl FileSource for ParquetSource { sort_order_for_reorder, reverse_row_groups: *reverse_row_groups, metadata_size_hint, + pruning_only_predicate: *pruning_only_predicate, }; Ok(Some(protobuf::PhysicalPlanNode { physical_plan_type: Some(PhysicalPlanType::ParquetScan(node)), @@ -1199,6 +1247,7 @@ impl ParquetSource { sort_order_for_reorder, reverse_row_groups, metadata_size_hint, + pruning_only_predicate, } = scan; let base_conf = base_conf.as_ref().ok_or_else(|| { @@ -1283,6 +1332,7 @@ impl ParquetSource { if let Some(predicate) = predicate { source = source.with_predicate(predicate); } + source.pruning_only_predicate = *pruning_only_predicate; let base_config = FileScanConfig::try_from_proto(base_conf, ctx, Arc::new(source))?; Ok(DataSourceExec::from_data_source(base_config)) @@ -2108,6 +2158,69 @@ mod tests { } } + /// Filters for pruning only stay above the scan. The scan applies all + /// conjuncts of its predicate or none of them. + #[test] + fn pruning_filters_stay_above_the_scan() { + use arrow::datatypes::{DataType, Field, Schema}; + use datafusion_common::config::ConfigOptions; + use datafusion_expr::{col, lit as logical_lit}; + use datafusion_physical_expr::planner::logical2physical; + use datafusion_physical_plan::filter_pushdown::PushedDown; + + let schema = Arc::new(Schema::new(vec![Field::new( + "value", + DataType::Int64, + false, + )])); + let filter = |v: i64| logical2physical(&col("value").gt(logical_lit(v)), &schema); + let config = ConfigOptions::default(); + let downcast = |source: &Arc| { + source.downcast_ref::().unwrap().clone() + }; + + // A source without a predicate takes filters for pruning only. + let source = ParquetSource::new(Arc::clone(&schema)); + let pruning = downcast( + &source + .try_pushdown_pruning_filters(&[filter(1)], &config) + .unwrap() + .expect("the source takes filters for pruning only"), + ); + assert!(pruning.pruning_only_predicate); + assert_eq!( + pruning.predicate.as_ref().unwrap().to_string(), + "value@0 > 1" + ); + + // It uses later filters only to prune too. + let prop = pruning + .try_pushdown_filters(vec![filter(2)], &config) + .unwrap(); + assert!(matches!(prop.filters[..], [PushedDown::No])); + let later = downcast(&prop.updated_node.unwrap()); + assert!(later.pruning_only_predicate); + assert_eq!( + later.predicate.as_ref().unwrap().to_string(), + "value@0 > 1 AND value@0 > 2" + ); + + // A source that applies its predicate does not take filters for + // pruning only. + let prop = source + .try_pushdown_filters(vec![filter(1)], &config) + .unwrap(); + assert!(matches!(prop.filters[..], [PushedDown::Yes])); + let applied = downcast(&prop.updated_node.unwrap()); + assert!(!applied.pruning_only_predicate); + assert!( + applied + .try_pushdown_pruning_filters(&[filter(2)], &config) + .unwrap() + .is_none() + ); + } + #[test] fn test_try_pushdown_filters_rejects_virtual_column_refs() { // Virtual columns are produced by the reader and cannot be referenced diff --git a/datafusion/datasource/src/file.rs b/datafusion/datasource/src/file.rs index f1a94f2e12363..88c98ad058476 100644 --- a/datafusion/datasource/src/file.rs +++ b/datafusion/datasource/src/file.rs @@ -208,6 +208,26 @@ pub trait FileSource: Any + Send + Sync { )) } + /// Try to push down filters that stay above the scan, for pruning only. + /// + /// A `FilterExec` above the scan applies `filters`, thus the source does + /// not need to apply them to the rows. It can use them to prune, for + /// example files, row groups and pages. `filters` are in terms of the + /// unprojected table schema, as for [`Self::try_pushdown_filters`]. + /// + /// `FileScanConfig` calls this method instead of + /// [`Self::try_pushdown_filters`] when a `FilterExec` above the scan runs + /// in more partitions than the scan. Returns the new source, or `None` + /// (the default) if the source does not support it: then + /// `FileScanConfig` calls [`Self::try_pushdown_filters`]. + fn try_pushdown_pruning_filters( + &self, + _filters: &[Arc], + _config: &ConfigOptions, + ) -> Result>> { + Ok(None) + } + /// Try to create a new FileSource that can produce data in the specified sort order. /// /// This method attempts to optimize data retrieval to match the requested ordering. diff --git a/datafusion/datasource/src/file_scan_config/mod.rs b/datafusion/datasource/src/file_scan_config/mod.rs index 61b41fd1e95a9..740f715592303 100644 --- a/datafusion/datasource/src/file_scan_config/mod.rs +++ b/datafusion/datasource/src/file_scan_config/mod.rs @@ -52,7 +52,9 @@ use datafusion_common::stats::{Precision, is_known_empty}; use datafusion_physical_expr::expressions::{BinaryExpr, Column}; use datafusion_physical_expr::projection::{ProjectionExprs, ProjectionMapping}; use datafusion_physical_expr::utils::reassign_expr_columns; -use datafusion_physical_expr::{EquivalenceProperties, Partitioning, split_conjunction}; +use datafusion_physical_expr::{ + DynamicFilterTracking, EquivalenceProperties, Partitioning, split_conjunction, +}; use datafusion_physical_expr_adapter::PhysicalExprAdapterFactory; use datafusion_physical_expr_common::physical_expr::{PhysicalExpr, is_volatile}; use datafusion_physical_expr_common::sort_expr::{LexOrdering, PhysicalSortExpr}; @@ -62,8 +64,9 @@ use datafusion_physical_plan::execution_plan::SchedulingType; use datafusion_physical_plan::{ DisplayAs, DisplayFormatType, display::{ProjectSchemaDisplay, display_orderings}, - filter_pushdown::FilterPushdownPropagation, + filter_pushdown::{FilterPushdownPropagation, PushedDown}, metrics::ExecutionPlanMetricsSet, + repartition::round_robin_beneficial_for_rows, }; use log::{debug, warn}; use std::any::Any; @@ -1055,6 +1058,28 @@ impl DataSource for FileScanConfig { .map(|filter| reassign_expr_columns(filter, table_schema)) .collect::>>()?; + // A filter that the scan applies runs in the partitions of the scan. + // If the optimizer would run a `FilterExec` above this scan in more + // partitions than the scan has, the filters stay above the scan + // (`PushedDown::No`), if the file source can use them for pruning + // only. This is only for the filters of a `FilterExec`: a dynamic + // filter (of a join, a TopK or an aggregate) has no `FilterExec` + // above the scan. + if !remapped_filters.iter().any(|filter| { + DynamicFilterTracking::classify(filter).contains_dynamic_filter() + }) && self.filters_run_in_more_partitions_above(config)? + && let Some(file_source) = self + .file_source + .try_pushdown_pruning_filters(&remapped_filters, config)? + { + let mut new_file_scan_config = self.clone(); + new_file_scan_config.file_source = file_source; + return Ok(FilterPushdownPropagation { + filters: vec![PushedDown::No; remapped_filters.len()], + updated_node: Some(Arc::new(new_file_scan_config) as _), + }); + } + let result = self .file_source .try_pushdown_filters(remapped_filters, config)?; @@ -1306,6 +1331,49 @@ impl FileScanConfig { ) } + /// Returns `true` if a filter above this scan runs in more partitions than + /// a filter in this scan. + /// + /// A filter that the scan applies runs in the partitions of the scan. A + /// `FilterExec` above the scan runs in the partitions of its input. If + /// the scan cannot give `target_partitions` partitions, the + /// `EnforceDistribution` rule puts a round-robin `RepartitionExec` + /// between the scan and the `FilterExec`, when the input can have more + /// rows than one batch. The `FilterExec` then runs in + /// `target_partitions` partitions. + /// + /// This function uses the same checks as `EnforceDistribution`: the + /// partitions that [`DataSource::repartitioned`] gives, and + /// [`round_robin_beneficial_for_rows`] on the rows that the scan reads. + /// The rows before the filter are the input of that round-robin + /// repartition. An exact count is a limit: when it is at most one batch, + /// a round-robin repartition cannot split the work. + fn filters_run_in_more_partitions_above( + &self, + config: &ConfigOptions, + ) -> Result { + let target_partitions = config.execution.target_partitions; + let partitions = self.output_partitioning().partition_count(); + if !config.optimizer.enable_round_robin_repartition + || partitions >= target_partitions + || !round_robin_beneficial_for_rows(&self.statistics.num_rows, config) + { + return Ok(false); + } + if !config.optimizer.repartition_file_scans { + return Ok(true); + } + let repartitioned = self.repartitioned( + target_partitions, + config.optimizer.repartition_file_min_size, + self.eq_properties().output_ordering(), + )?; + let partitions = repartitioned + .map(|source| source.output_partitioning().partition_count()) + .unwrap_or(partitions); + Ok(partitions < target_partitions) + } + /// Get the file schema (schema of the files without partition columns) pub fn file_schema(&self) -> &SchemaRef { self.file_source.table_schema().file_schema() @@ -4127,4 +4195,244 @@ mod tests { assert!(!would_duplicate_costly_exprs(&inner, &outer)); } + + /// Tests for the filters that stay above a scan that cannot give the + /// parallelism of a filter above it. + mod filters_above_scan { + use super::*; + use datafusion_physical_expr::expressions::{binary, lit}; + + /// A file source that applies all filters that it gets. If + /// `pruning_filters` is true, it also takes filters for pruning only. + #[derive(Clone)] + struct FilteringSource { + metrics: ExecutionPlanMetricsSet, + table_schema: TableSchema, + pruning_filters: bool, + filter: Option>, + pruning_only: bool, + } + + impl FileSource for FilteringSource { + fn create_file_opener( + &self, + _object_store: Arc, + _base_config: &FileScanConfig, + _partition: usize, + ) -> Result> { + unimplemented!() + } + + fn table_schema(&self) -> &TableSchema { + &self.table_schema + } + + fn with_batch_size(&self, _batch_size: usize) -> Arc { + Arc::new(self.clone()) + } + + fn metrics(&self) -> &ExecutionPlanMetricsSet { + &self.metrics + } + + fn file_type(&self) -> &str { + "filtering" + } + + fn filter(&self) -> Option> { + self.filter.clone() + } + + fn try_pushdown_filters( + &self, + filters: Vec>, + _config: &ConfigOptions, + ) -> Result>> { + let pushed_down = vec![PushedDown::Yes; filters.len()]; + let source = Self { + filter: Some(datafusion_physical_expr::conjunction(filters)), + pruning_only: false, + ..self.clone() + }; + Ok( + FilterPushdownPropagation::with_parent_pushdown_result(pushed_down) + .with_updated_node(Arc::new(source) as _), + ) + } + + fn try_pushdown_pruning_filters( + &self, + filters: &[Arc], + _config: &ConfigOptions, + ) -> Result>> { + Ok(self.pruning_filters.then(|| { + Arc::new(Self { + filter: Some(datafusion_physical_expr::conjunction( + filters.iter().cloned(), + )), + pruning_only: true, + ..self.clone() + }) as _ + })) + } + + fn apply_expressions( + &self, + _f: &mut dyn FnMut(&Arc) -> Result, + ) -> Result { + Ok(TreeNodeRecursion::Continue) + } + } + + fn schema() -> SchemaRef { + Arc::new(Schema::new(vec![Field::new("a", DataType::Int32, false)])) + } + + /// A scan of `files` files of `file_size` bytes each, one file in each + /// partition, with `num_rows` rows in total. + fn scan_with( + files: usize, + file_size: u64, + num_rows: usize, + pruning_filters: bool, + ) -> FileScanConfig { + let schema = schema(); + let source = FilteringSource { + metrics: ExecutionPlanMetricsSet::new(), + table_schema: TableSchema::from(&schema), + pruning_filters, + filter: None, + pruning_only: false, + }; + let files = (0..files) + .map(|i| { + FileGroup::new(vec![PartitionedFile::new( + format!("f{i}.parquet"), + file_size, + )]) + }) + .collect::>(); + FileScanConfigBuilder::new( + ObjectStoreUrl::local_filesystem(), + Arc::new(source), + ) + .with_file_groups(files) + .with_statistics( + Statistics::new_unknown(&schema) + .with_num_rows(Precision::Exact(num_rows)), + ) + .build() + } + + fn scan(files: usize, file_size: u64, num_rows: usize) -> FileScanConfig { + scan_with(files, file_size, num_rows, true) + } + + fn config(target_partitions: usize) -> ConfigOptions { + let mut config = ConfigOptions::default(); + config.execution.target_partitions = target_partitions; + config + } + + fn a_gt_5() -> Arc { + binary( + col("a", &schema()).unwrap(), + Operator::Gt, + lit(5), + &schema(), + ) + .unwrap() + } + + /// Pushes `a > 5` into `scan`. Returns if the scan applies it, and + /// if the new file source uses it for pruning only. + fn push(scan: &FileScanConfig, config: &ConfigOptions) -> (PushedDown, bool) { + push_filter(scan, a_gt_5(), config) + } + + fn push_filter( + scan: &FileScanConfig, + filter: Arc, + config: &ConfigOptions, + ) -> (PushedDown, bool) { + let expected = filter.to_string(); + let result = scan.try_pushdown_filters(vec![filter], config).unwrap(); + let node = result.updated_node.expect("the source takes the filter"); + let node = node.downcast_ref::().unwrap(); + let source = node + .file_source + .as_ref() + .downcast_ref::() + .unwrap(); + assert_eq!(source.filter.as_ref().unwrap().to_string(), expected); + (result.filters[0], source.pruning_only) + } + + #[test] + fn filter_stays_above_a_scan_that_cannot_split() { + // One small file: the scan has one partition and cannot split + // the file, thus a filter above the scan runs in 4 partitions. + let (pushed_down, pruning_only) = push(&scan(1, 1024, 100_000), &config(4)); + assert!(matches!(pushed_down, PushedDown::No)); + assert!(pruning_only); + } + + #[test] + fn scan_applies_the_filter_when_it_has_the_partitions() { + // One file for each target partition. + let (pushed_down, pruning_only) = push(&scan(4, 1024, 100_000), &config(4)); + assert!(matches!(pushed_down, PushedDown::Yes)); + assert!(!pruning_only); + + // One large file that the scan splits into 4 byte ranges. + let (pushed_down, _) = push(&scan(1, 1 << 30, 100_000), &config(4)); + assert!(matches!(pushed_down, PushedDown::Yes)); + + // One target partition. + let (pushed_down, _) = push(&scan(1, 1024, 100_000), &config(1)); + assert!(matches!(pushed_down, PushedDown::Yes)); + } + + #[test] + fn scan_applies_the_filter_when_a_round_robin_does_not_help() { + // The scan reads at most one batch: a round-robin repartition + // cannot split the work. + let (pushed_down, _) = push(&scan(1, 1024, 8192), &config(4)); + assert!(matches!(pushed_down, PushedDown::Yes)); + let (pushed_down, _) = push(&scan(1, 1024, 0), &config(4)); + assert!(matches!(pushed_down, PushedDown::Yes)); + + // Round-robin repartitions are off. + let mut config = config(4); + config.optimizer.enable_round_robin_repartition = false; + let (pushed_down, _) = push(&scan(1, 1024, 100_000), &config); + assert!(matches!(pushed_down, PushedDown::Yes)); + } + + #[test] + fn dynamic_filter_goes_into_the_scan() { + // A dynamic filter has no `FilterExec` above the scan, thus the + // scan applies it. + use datafusion_physical_expr::expressions::DynamicFilterPhysicalExpr; + let dynamic: Arc = + Arc::new(DynamicFilterPhysicalExpr::new( + vec![col("a", &schema()).unwrap()], + a_gt_5(), + )); + let (pushed_down, pruning_only) = + push_filter(&scan(1, 1024, 100_000), dynamic, &config(4)); + assert!(matches!(pushed_down, PushedDown::Yes)); + assert!(!pruning_only); + } + + #[test] + fn source_without_pruning_filters_applies_the_filter() { + // The source does not take filters for pruning only: the scan + // pushes them as before. + let (pushed_down, pruning_only) = + push(&scan_with(1, 1024, 100_000, false), &config(4)); + assert!(matches!(pushed_down, PushedDown::Yes)); + assert!(!pruning_only); + } + } } diff --git a/datafusion/proto-models/proto/datafusion.proto b/datafusion/proto-models/proto/datafusion.proto index 10acc53ccb2b3..ccd830ef74eea 100644 --- a/datafusion/proto-models/proto/datafusion.proto +++ b/datafusion/proto-models/proto/datafusion.proto @@ -1331,6 +1331,10 @@ message ParquetScanExecNode { // Source-specific footer prefetch size. Absent means no hint. optional uint64 metadata_size_hint = 7; + + // If true, the scan uses `predicate` only to prune: a filter above the + // scan applies it. + bool pruning_only_predicate = 8; } message CsvScanExecNode { diff --git a/datafusion/proto-models/src/generated/pbjson.rs b/datafusion/proto-models/src/generated/pbjson.rs index e6e465cb0aa3c..4779760866fc6 100644 --- a/datafusion/proto-models/src/generated/pbjson.rs +++ b/datafusion/proto-models/src/generated/pbjson.rs @@ -16785,6 +16785,9 @@ impl serde::Serialize for ParquetScanExecNode { if self.metadata_size_hint.is_some() { len += 1; } + if self.pruning_only_predicate { + len += 1; + } let mut struct_ser = serializer.serialize_struct("datafusion.ParquetScanExecNode", len)?; if let Some(v) = self.base_conf.as_ref() { struct_ser.serialize_field("baseConf", v)?; @@ -16806,6 +16809,9 @@ impl serde::Serialize for ParquetScanExecNode { #[allow(clippy::needless_borrows_for_generic_args)] struct_ser.serialize_field("metadataSizeHint", ToString::to_string(&v).as_str())?; } + if self.pruning_only_predicate { + struct_ser.serialize_field("pruningOnlyPredicate", &self.pruning_only_predicate)?; + } struct_ser.end() } } @@ -16827,6 +16833,8 @@ impl<'de> serde::Deserialize<'de> for ParquetScanExecNode { "reverseRowGroups", "metadata_size_hint", "metadataSizeHint", + "pruning_only_predicate", + "pruningOnlyPredicate", ]; #[allow(clippy::enum_variant_names)] @@ -16837,6 +16845,7 @@ impl<'de> serde::Deserialize<'de> for ParquetScanExecNode { SortOrderForReorder, ReverseRowGroups, MetadataSizeHint, + PruningOnlyPredicate, } impl<'de> serde::Deserialize<'de> for GeneratedField { fn deserialize(deserializer: D) -> std::result::Result @@ -16864,6 +16873,7 @@ impl<'de> serde::Deserialize<'de> for ParquetScanExecNode { "sortOrderForReorder" | "sort_order_for_reorder" => Ok(GeneratedField::SortOrderForReorder), "reverseRowGroups" | "reverse_row_groups" => Ok(GeneratedField::ReverseRowGroups), "metadataSizeHint" | "metadata_size_hint" => Ok(GeneratedField::MetadataSizeHint), + "pruningOnlyPredicate" | "pruning_only_predicate" => Ok(GeneratedField::PruningOnlyPredicate), _ => Err(serde::de::Error::unknown_field(value, FIELDS)), } } @@ -16889,6 +16899,7 @@ impl<'de> serde::Deserialize<'de> for ParquetScanExecNode { let mut sort_order_for_reorder__ = None; let mut reverse_row_groups__ = None; let mut metadata_size_hint__ = None; + let mut pruning_only_predicate__ = None; while let Some(k) = map_.next_key()? { match k { GeneratedField::BaseConf => { @@ -16929,6 +16940,12 @@ impl<'de> serde::Deserialize<'de> for ParquetScanExecNode { map_.next_value::<::std::option::Option<::pbjson::private::NumberDeserialize<_>>>()?.map(|x| x.0) ; } + GeneratedField::PruningOnlyPredicate => { + if pruning_only_predicate__.is_some() { + return Err(serde::de::Error::duplicate_field("pruningOnlyPredicate")); + } + pruning_only_predicate__ = Some(map_.next_value()?); + } } } Ok(ParquetScanExecNode { @@ -16938,6 +16955,7 @@ impl<'de> serde::Deserialize<'de> for ParquetScanExecNode { sort_order_for_reorder: sort_order_for_reorder__, reverse_row_groups: reverse_row_groups__.unwrap_or_default(), metadata_size_hint: metadata_size_hint__, + pruning_only_predicate: pruning_only_predicate__.unwrap_or_default(), }) } } diff --git a/datafusion/proto-models/src/generated/prost.rs b/datafusion/proto-models/src/generated/prost.rs index 2480d26e46a7f..5c4e4ce4de200 100644 --- a/datafusion/proto-models/src/generated/prost.rs +++ b/datafusion/proto-models/src/generated/prost.rs @@ -2035,6 +2035,10 @@ pub struct ParquetScanExecNode { /// Source-specific footer prefetch size. Absent means no hint. #[prost(uint64, optional, tag = "7")] pub metadata_size_hint: ::core::option::Option, + /// If true, the scan uses `predicate` only to prune: a filter above the + /// scan applies it. + #[prost(bool, tag = "8")] + pub pruning_only_predicate: bool, } #[derive(Clone, PartialEq, ::prost::Message)] pub struct CsvScanExecNode { diff --git a/datafusion/proto/tests/cases/plans/sources.rs b/datafusion/proto/tests/cases/plans/sources.rs index c4897357ee00b..d96f688c27941 100644 --- a/datafusion/proto/tests/cases/plans/sources.rs +++ b/datafusion/proto/tests/cases/plans/sources.rs @@ -166,6 +166,49 @@ fn roundtrip_parquet_exec_with_pruning_predicate() -> Result<()> { Ok(()) } +/// A predicate that the scan uses only to prune stays pruning-only after a +/// round trip: the decoded scan does not apply it. +#[test] +fn roundtrip_parquet_exec_with_pruning_only_predicate() -> Result<()> { + use datafusion::datasource::physical_plan::FileSource; + + let file_schema = + Arc::new(Schema::new(vec![Field::new("col", DataType::Utf8, false)])); + let predicate: Arc = Arc::new(BinaryExpr::new( + Arc::new(Column::new("col", 0)), + Operator::Eq, + lit("1"), + )); + let file_source = ParquetSource::new(Arc::clone(&file_schema)) + .try_pushdown_pruning_filters(&[predicate], &Default::default())? + .expect("the source takes filters for pruning only"); + let scan_config = + FileScanConfigBuilder::new(ObjectStoreUrl::local_filesystem(), file_source) + .with_file_groups(vec![FileGroup::new(vec![PartitionedFile::new( + "/path/to/file.parquet".to_string(), + 1024, + )])]) + .build(); + + let ctx = SessionContext::new(); + let codec = DefaultPhysicalExtensionCodec {}; + let roundtripped = roundtrip_test_and_return( + DataSourceExec::from_data_source(scan_config), + &ctx, + &codec, + &DefaultPhysicalProtoConverter {}, + )?; + let node = PhysicalPlanNode::try_from_physical_plan(roundtripped, &codec)?; + let Some(protobuf::physical_plan_node::PhysicalPlanType::ParquetScan(scan)) = + node.physical_plan_type + else { + return internal_err!("Expected ParquetScan node"); + }; + assert!(scan.predicate.is_some()); + assert!(scan.pruning_only_predicate); + Ok(()) +} + #[tokio::test] async fn roundtrip_parquet_exec_with_sort_pushdown() -> Result<()> { let ctx = all_types_context().await?; diff --git a/datafusion/sqllogictest/test_files/parquet_filter_pushdown.slt b/datafusion/sqllogictest/test_files/parquet_filter_pushdown.slt index 224640cab0997..ba2163dc06710 100644 --- a/datafusion/sqllogictest/test_files/parquet_filter_pushdown.slt +++ b/datafusion/sqllogictest/test_files/parquet_filter_pushdown.slt @@ -1042,3 +1042,72 @@ set datafusion.execution.parquet.reorder_filters = false; statement ok DROP TABLE dict_filter_bug; + + +### +# A filter stays above a scan that cannot give the target partitions. +# +# The filter in a scan runs in the partitions of the scan. One small file +# gives one partition. A `FilterExec` above a round-robin repartition runs in +# `target_partitions` partitions, thus the filter stays above the scan. The +# scan uses it only for pruning. +### + +statement ok +set datafusion.execution.parquet.pushdown_filters = true; + +statement ok +set datafusion.execution.target_partitions = 4; + +statement ok +COPY (SELECT value AS a, value % 7 AS b FROM generate_series(1, 20000)) +TO 'test_files/scratch/parquet_filter_pushdown/one_file/data.parquet'; + +statement ok +CREATE EXTERNAL TABLE one_file STORED AS PARQUET +LOCATION 'test_files/scratch/parquet_filter_pushdown/one_file/'; + +query TT +EXPLAIN SELECT a FROM one_file WHERE b = 3; +---- +logical_plan +01)Projection: one_file.a +02)--Filter: one_file.b = Int64(3) +03)----TableScan: one_file projection=[a, b], partial_filters=[one_file.b = Int64(3)] +physical_plan +01)FilterExec: b@1 = 3, projection=[a@0] +02)--RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, maintains_sort_order=true +03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_filter_pushdown/one_file/data.parquet]]}, projection=[a, b], output_ordering=[a@0 ASC NULLS LAST], file_type=parquet, predicate=b@1 = 3, pruning_predicate=b_null_count@2 != row_count@3 AND b_min@0 <= 3 AND 3 <= b_max@1, required_guarantees=[b in (3)] + +query I +SELECT count(*) FROM one_file WHERE b = 3; +---- +2857 + +# With one target partition, the filter above cannot run in more partitions: +# the scan applies it. +statement ok +set datafusion.execution.target_partitions = 1; + +query TT +EXPLAIN SELECT a FROM one_file WHERE b = 3; +---- +logical_plan +01)Projection: one_file.a +02)--Filter: one_file.b = Int64(3) +03)----TableScan: one_file projection=[a, b], partial_filters=[one_file.b = Int64(3)] +physical_plan DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_filter_pushdown/one_file/data.parquet]]}, projection=[a], output_ordering=[a@0 ASC NULLS LAST], file_type=parquet, predicate=b@1 = 3, pruning_predicate=b_null_count@2 != row_count@3 AND b_min@0 <= 3 AND 3 <= b_max@1, required_guarantees=[b in (3)] + +query I +SELECT count(*) FROM one_file WHERE b = 3; +---- +2857 + +statement ok +DROP TABLE one_file; + +statement ok +set datafusion.execution.target_partitions = 4; + +statement ok +set datafusion.execution.parquet.pushdown_filters = false; diff --git a/datafusion/sqllogictest/test_files/parquet_statistics.slt b/datafusion/sqllogictest/test_files/parquet_statistics.slt index 1eea640a17cb0..2d1c0ba6b4932 100644 --- a/datafusion/sqllogictest/test_files/parquet_statistics.slt +++ b/datafusion/sqllogictest/test_files/parquet_statistics.slt @@ -102,7 +102,10 @@ LOCATION 'test_files/scratch/parquet_statistics/test_table'; query TT EXPLAIN SELECT * FROM test_table WHERE column1 = 1; ---- -physical_plan DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_statistics/test_table/0.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_statistics/test_table/1.parquet]]}, projection=[column1], file_type=parquet, predicate=column1@0 = 1, pruning_predicate=column1_null_count@2 != row_count@3 AND column1_min@0 <= 1 AND 1 <= column1_max@1, required_guarantees=[column1 in (1)], statistics=[Rows=Absent, Bytes=Absent, [(Col[0]:)]] +physical_plan +01)FilterExec: column1@0 = 1, statistics=[Rows=Absent, Bytes=Absent, [(Col[0]: Min=Exact(Int64(1)) Max=Exact(Int64(1)) Null=Exact(0) Distinct=Inexact(1))]] +02)--RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=2, statistics=[Rows=Absent, Bytes=Absent, [(Col[0]:)]] +03)----DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_statistics/test_table/0.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/parquet_statistics/test_table/1.parquet]]}, projection=[column1], file_type=parquet, predicate=column1@0 = 1, pruning_predicate=column1_null_count@2 != row_count@3 AND column1_min@0 <= 1 AND 1 <= column1_max@1, required_guarantees=[column1 in (1)], statistics=[Rows=Absent, Bytes=Absent, [(Col[0]:)]] # cleanup statement ok From fe6aa04149bbcc3dbf9bd5d433e09e8a9092e2ae Mon Sep 17 00:00:00 2001 From: Adrian Garcia Badaracco <1755071+adriangb@users.noreply.github.com> Date: Wed, 23 Sep 2026 20:05:21 -0500 Subject: [PATCH 06/42] feat: add OptionalFilterPhysicalExpr to mark filters not needed for correctness Add `OptionalFilterPhysicalExpr`, a transparent wrapper that marks a filter as optional: a consumer can skip it without changing the query result. A consumer can skip it only when the wrapper is a direct conjunct of the root AND chain of its predicate. In all other positions the wrapper is transparent, because `evaluate()` always evaluates the inner expression. `snapshot()` returns the inner expression, so pruning sees through it. Also add: - `split_optional` and `is_optional_filter` helpers in `physical_expr::utils` for consumers - `PhysicalOptionalFilterNode` proto message (field 29 in `PhysicalExprNode`) with self-encoding `try_to_proto`/`try_from_proto` No producer uses the wrapper yet, so there is no behavior change. Co-Authored-By: Claude Opus 5.5 --- .../physical-expr/src/expressions/mod.rs | 2 + .../src/expressions/optional_filter.rs | 357 ++++++++++++++++++ .../physical-expr/src/simplifier/mod.rs | 32 +- datafusion/physical-expr/src/utils/mod.rs | 136 ++++++- .../proto-models/proto/datafusion.proto | 7 + .../proto-models/src/generated/pbjson.rs | 105 ++++++ .../proto-models/src/generated/prost.rs | 10 +- .../proto/src/physical_plan/from_proto.rs | 7 +- .../tests/cases/plans/dynamic_filters.rs | 65 +++- datafusion/pruning/src/pruning_predicate.rs | 41 ++ 10 files changed, 757 insertions(+), 5 deletions(-) create mode 100644 datafusion/physical-expr/src/expressions/optional_filter.rs diff --git a/datafusion/physical-expr/src/expressions/mod.rs b/datafusion/physical-expr/src/expressions/mod.rs index f2f9285de560a..d36f734afef39 100644 --- a/datafusion/physical-expr/src/expressions/mod.rs +++ b/datafusion/physical-expr/src/expressions/mod.rs @@ -33,6 +33,7 @@ mod literal; mod negative; mod no_op; mod not; +mod optional_filter; mod similar_to_pattern; mod try_cast; mod unknown_column; @@ -59,6 +60,7 @@ pub use literal::{Literal, lit}; pub use negative::{NegativeExpr, negative}; pub use no_op::NoOp; pub use not::{NotExpr, not}; +pub use optional_filter::OptionalFilterPhysicalExpr; pub(crate) use similar_to_pattern::translate_scalar; pub use similar_to_pattern::{SqlSimilarToPattern, sql_similar_to_regex}; pub use try_cast::{TryCastExpr, try_cast, try_cast_with_target_field}; diff --git a/datafusion/physical-expr/src/expressions/optional_filter.rs b/datafusion/physical-expr/src/expressions/optional_filter.rs new file mode 100644 index 0000000000000..5561f90e89345 --- /dev/null +++ b/datafusion/physical-expr/src/expressions/optional_filter.rs @@ -0,0 +1,357 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! [`OptionalFilterPhysicalExpr`]: a marker for filters that are not needed +//! for correctness. See the type documentation for the contract that +//! producers, consumers and rewriters must obey. + +use std::fmt; +use std::hash::Hash; +use std::sync::Arc; + +use crate::PhysicalExpr; + +use arrow::array::BooleanArray; +use arrow::datatypes::{DataType, FieldRef, Schema}; +use arrow::record_batch::RecordBatch; +use datafusion_common::{Result, assert_eq_or_internal_err}; +use datafusion_expr::ColumnarValue; +use datafusion_expr::interval_arithmetic::Interval; +use datafusion_expr::sort_properties::ExprProperties; + +/// Marks the inner filter as *optional*: it is not needed for correctness. +/// +/// Some filters are only performance hints. For example, a hash join can push +/// a dynamic filter into the probe side scan, but the join itself still +/// removes the rows that do not match. Such a filter is *optional*: a +/// consumer can skip it (for example, when the filter does not remove enough +/// rows to be worth its cost) and the query result stays the same. +/// +/// # Contract +/// +/// * **Skip only on the root AND chain.** A consumer can skip an optional +/// filter only when the `Optional` node is a direct conjunct of the root +/// `AND` chain of its predicate. For example, in `a AND Optional(b)` the +/// consumer can skip `b`. Use [`split_optional`] to find these conjuncts. +/// * **Transparent everywhere else.** [`PhysicalExpr::evaluate`] always +/// evaluates the inner expression. Thus an `Optional` in a different +/// position (for example under `NOT`, `IS NULL`, `CASE` or `OR`) can make a +/// query slower, but it cannot make the result incorrect. +/// * **Rewriters must not move nodes across the wrapper.** A rewrite must not +/// move an expression into or out of an `Optional`. For example, +/// `NOT(Optional(x))` must not become `Optional(NOT(x))`, because that +/// would make a required filter optional. +/// * **Pruning sees through the wrapper.** [`PhysicalExpr::snapshot`] +/// returns the inner expression, so [`snapshot_physical_expr`] removes the +/// wrapper. Thus statistics pruning uses an optional filter the same as a +/// required filter. +/// +/// [`split_optional`]: crate::utils::split_optional +/// [`snapshot_physical_expr`]: datafusion_physical_expr_common::physical_expr::snapshot_physical_expr +#[derive(Debug, Eq)] +pub struct OptionalFilterPhysicalExpr { + inner: Arc, +} + +// Manually derive PartialEq and Hash to work around https://github.com/rust-lang/rust/issues/78808 +impl PartialEq for OptionalFilterPhysicalExpr { + fn eq(&self, other: &Self) -> bool { + self.inner.eq(&other.inner) + } +} + +impl Hash for OptionalFilterPhysicalExpr { + fn hash(&self, state: &mut H) { + self.inner.hash(state); + } +} + +impl OptionalFilterPhysicalExpr { + /// Create a new optional filter that wraps `inner`. + pub fn new(inner: Arc) -> Self { + Self { inner } + } + + /// Get the wrapped filter expression. + pub fn inner(&self) -> &Arc { + &self.inner + } +} + +impl fmt::Display for OptionalFilterPhysicalExpr { + fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result { + write!(f, "Optional({})", self.inner) + } +} + +impl PhysicalExpr for OptionalFilterPhysicalExpr { + fn data_type(&self, input_schema: &Schema) -> Result { + self.inner.data_type(input_schema) + } + + fn nullable(&self, input_schema: &Schema) -> Result { + self.inner.nullable(input_schema) + } + + fn evaluate(&self, batch: &RecordBatch) -> Result { + self.inner.evaluate(batch) + } + + fn return_field(&self, input_schema: &Schema) -> Result { + self.inner.return_field(input_schema) + } + + fn evaluate_selection( + &self, + batch: &RecordBatch, + selection: &BooleanArray, + ) -> Result { + self.inner.evaluate_selection(batch, selection) + } + + fn children(&self) -> Vec<&Arc> { + vec![&self.inner] + } + + fn with_new_children( + self: Arc, + children: Vec>, + ) -> Result> { + assert_eq_or_internal_err!( + children.len(), + 1, + "OptionalFilterPhysicalExpr: expected 1 child" + ); + Ok(Arc::new(Self::new(Arc::clone(&children[0])))) + } + + // The wrapper is the identity function, so the bounds and properties of + // the child are also the bounds and properties of the wrapper. + fn evaluate_bounds(&self, children: &[&Interval]) -> Result { + Ok(children[0].clone()) + } + + fn propagate_constraints( + &self, + interval: &Interval, + children: &[&Interval], + ) -> Result>> { + Ok(children[0].intersect(interval)?.map(|result| vec![result])) + } + + fn get_properties(&self, children: &[ExprProperties]) -> Result { + Ok(children[0].clone()) + } + + fn fmt_sql(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + self.inner.fmt_sql(f) + } + + /// Returns the inner expression, so that snapshot consumers (for example + /// pruning) see through the wrapper. + /// + /// [`snapshot_physical_expr`] transforms the tree bottom up, so the inner + /// expression is already a snapshot when this method is called. For + /// example, `Optional(DynamicFilter)` becomes the current expression of + /// the dynamic filter. + /// + /// [`snapshot_physical_expr`]: datafusion_physical_expr_common::physical_expr::snapshot_physical_expr + fn snapshot(&self) -> Result>> { + Ok(Some(Arc::clone(&self.inner))) + } + + fn snapshot_generation(&self) -> u64 { + // The wrapper is not dynamic. `snapshot_generation(expr)` walks the + // tree and adds the generation of the inner expression. + 0 + } + + #[cfg(feature = "proto")] + fn try_to_proto( + &self, + ctx: &datafusion_physical_expr_common::physical_expr::proto_encode::PhysicalExprEncodeCtx<'_>, + ) -> Result> { + use datafusion_proto_models::protobuf; + + Ok(Some(protobuf::PhysicalExprNode { + expr_id: None, + expr_type: Some(protobuf::physical_expr_node::ExprType::OptionalFilter( + Box::new(protobuf::PhysicalOptionalFilterNode { + inner: Some(Box::new(ctx.encode_child(&self.inner)?)), + }), + )), + })) + } +} + +#[cfg(feature = "proto")] +impl OptionalFilterPhysicalExpr { + /// Reconstruct an [`OptionalFilterPhysicalExpr`] from its protobuf + /// representation. + pub fn try_from_proto( + node: &datafusion_proto_models::protobuf::PhysicalExprNode, + ctx: &datafusion_physical_expr_common::physical_expr::proto_decode::PhysicalExprDecodeCtx<'_>, + ) -> Result> { + use datafusion_physical_expr_common::expect_expr_variant; + use datafusion_proto_models::protobuf; + + let optional = expect_expr_variant!( + node, + protobuf::physical_expr_node::ExprType::OptionalFilter, + "OptionalFilter", + ); + let inner = ctx.decode_required_expression( + optional.inner.as_deref(), + "OptionalFilter", + "inner", + )?; + + Ok(Arc::new(Self::new(inner))) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::expressions::{BinaryExpr, DynamicFilterPhysicalExpr, col, lit, not}; + + use arrow::array::{ArrayRef, Int32Array}; + use arrow::datatypes::Field; + use datafusion_common::cast::as_boolean_array; + use datafusion_expr::Operator; + use datafusion_physical_expr_common::physical_expr::{ + fmt_sql, snapshot_generation, snapshot_physical_expr, + }; + + fn schema() -> Arc { + Arc::new(Schema::new(vec![Field::new("a", DataType::Int32, true)])) + } + + /// `a > 2` + fn a_gt_2(schema: &Schema) -> Arc { + Arc::new(BinaryExpr::new( + col("a", schema).unwrap(), + Operator::Gt, + lit(2i32), + )) + } + + fn optional(inner: Arc) -> Arc { + Arc::new(OptionalFilterPhysicalExpr::new(inner)) + } + + #[test] + fn evaluate_equals_inner() -> Result<()> { + let schema = schema(); + let values: ArrayRef = Arc::new(Int32Array::from(vec![Some(1), None, Some(3)])); + let batch = RecordBatch::try_new(Arc::clone(&schema), vec![values])?; + + let inner = a_gt_2(&schema); + let wrapped = optional(Arc::clone(&inner)); + + let expected = inner.evaluate(&batch)?.into_array(batch.num_rows())?; + let actual = wrapped.evaluate(&batch)?.into_array(batch.num_rows())?; + assert_eq!(as_boolean_array(&actual)?, as_boolean_array(&expected)?); + + let selection = BooleanArray::from(vec![true, false, true]); + let expected = inner + .evaluate_selection(&batch, &selection)? + .into_array(batch.num_rows())?; + let actual = wrapped + .evaluate_selection(&batch, &selection)? + .into_array(batch.num_rows())?; + assert_eq!(as_boolean_array(&actual)?, as_boolean_array(&expected)?); + + assert_eq!(wrapped.data_type(&schema)?, DataType::Boolean); + assert!(wrapped.nullable(&schema)?); + Ok(()) + } + + #[test] + fn display_and_fmt_sql() { + let schema = schema(); + let wrapped = optional(a_gt_2(&schema)); + assert_eq!(wrapped.to_string(), "Optional(a@0 > 2)"); + assert_eq!(fmt_sql(wrapped.as_ref()).to_string(), "a > 2"); + + let negated = not(Arc::clone(&wrapped)).unwrap(); + assert_eq!(negated.to_string(), "NOT Optional(a@0 > 2)"); + } + + #[test] + fn children_and_with_new_children() -> Result<()> { + let schema = schema(); + let wrapped = optional(a_gt_2(&schema)); + assert_eq!(wrapped.children().len(), 1); + + let new_inner = lit(true); + let rewrapped = Arc::clone(&wrapped).with_new_children(vec![new_inner])?; + let rewrapped = rewrapped + .downcast_ref::() + .expect("wrapper is kept"); + assert_eq!(rewrapped.inner().to_string(), "true"); + + assert!(wrapped.with_new_children(vec![]).is_err()); + Ok(()) + } + + #[test] + fn eq_and_hash_use_inner() { + use std::collections::HashSet; + + let schema = schema(); + let a = optional(a_gt_2(&schema)); + let b = optional(a_gt_2(&schema)); + let c = optional(lit(true)); + assert_eq!(&a, &b); + assert_ne!(&a, &c); + // The wrapper is not equal to the inner expression. + assert_ne!(&a, &a_gt_2(&schema)); + + let set: HashSet<_> = [a, b, c].into_iter().collect(); + assert_eq!(set.len(), 2); + } + + #[test] + fn snapshot_sees_through_dynamic_filter() -> Result<()> { + let schema = schema(); + let dynamic = Arc::new(DynamicFilterPhysicalExpr::new( + vec![col("a", &schema)?], + lit(true), + )); + let wrapped = optional(Arc::clone(&dynamic) as Arc); + + assert_eq!( + snapshot_physical_expr(Arc::clone(&wrapped))?.to_string(), + "true" + ); + + let generation = snapshot_generation(&wrapped); + dynamic.update(a_gt_2(&schema))?; + assert_ne!(snapshot_generation(&wrapped), generation); + assert_eq!( + snapshot_physical_expr(Arc::clone(&wrapped))?.to_string(), + "a@0 > 2" + ); + + // A static inner expression is also unwrapped. + let wrapped = optional(a_gt_2(&schema)); + assert_eq!(wrapped.snapshot_generation(), 0); + assert_eq!(snapshot_physical_expr(wrapped)?.to_string(), "a@0 > 2"); + Ok(()) + } +} diff --git a/datafusion/physical-expr/src/simplifier/mod.rs b/datafusion/physical-expr/src/simplifier/mod.rs index af87ce7f61d35..18fd29709510e 100644 --- a/datafusion/physical-expr/src/simplifier/mod.rs +++ b/datafusion/physical-expr/src/simplifier/mod.rs @@ -96,7 +96,8 @@ mod tests { use super::*; use crate::ScalarFunctionExpr; use crate::expressions::{ - BinaryExpr, CastExpr, Literal, NotExpr, TryCastExpr, col, in_list, lit, + BinaryExpr, CastExpr, Literal, NotExpr, OptionalFilterPhysicalExpr, TryCastExpr, + col, in_list, lit, }; use arrow::datatypes::{DataType, Field}; use datafusion_common::ScalarValue; @@ -238,6 +239,35 @@ mod tests { Ok(()) } + #[test] + fn test_not_optional_filter_unchanged() -> Result<()> { + let schema = not_test_schema(); + let simplifier = PhysicalExprSimplifier::new(&schema); + let optional = |e: Arc| -> Arc { + Arc::new(OptionalFilterPhysicalExpr::new(e)) + }; + + // NOT(Optional(c > 5)) is unchanged: the NOT must not move into the + // Optional, because NOT(c > 5) is a required filter. + let c_gt_5: Arc = Arc::new(BinaryExpr::new( + col("c", &schema)?, + Operator::Gt, + lit(ScalarValue::Int32(Some(5))), + )); + let expr: Arc = + Arc::new(NotExpr::new(optional(Arc::clone(&c_gt_5)))); + assert_not_simplify(&simplifier, Arc::clone(&expr), expr); + + // NOT(Optional(NOT(a))) is unchanged: no double negation elimination + // across the Optional. + let expr: Arc = Arc::new(NotExpr::new(optional(Arc::new( + NotExpr::new(col("a", &schema)?), + )))); + assert_not_simplify(&simplifier, Arc::clone(&expr), expr); + + Ok(()) + } + #[test] fn test_not_literal() -> Result<()> { let schema = not_test_schema(); diff --git a/datafusion/physical-expr/src/utils/mod.rs b/datafusion/physical-expr/src/utils/mod.rs index 1be57c9192626..88d483a92d8c5 100644 --- a/datafusion/physical-expr/src/utils/mod.rs +++ b/datafusion/physical-expr/src/utils/mod.rs @@ -21,7 +21,7 @@ pub use guarantee::{Guarantee, LiteralGuarantee}; use std::borrow::Borrow; use std::sync::Arc; -use crate::expressions::{BinaryExpr, Column, Literal}; +use crate::expressions::{BinaryExpr, Column, Literal, OptionalFilterPhysicalExpr}; use crate::tree_node::ExprContext; use crate::{ AcrossPartitions, ConstExpr, EquivalenceProperties, PhysicalExpr, PhysicalSortExpr, @@ -46,6 +46,42 @@ pub fn split_conjunction( split_impl(Operator::And, predicate, vec![]) } +/// Split the root `AND` chain of `predicate` into required and optional +/// filters. +/// +/// Returns `(required, optional)`. `required` holds the conjuncts that must be +/// applied. `optional` holds the *inner* expressions of the conjuncts that are +/// an [`OptionalFilterPhysicalExpr`], which a consumer can skip without +/// affecting correctness. +/// +/// Only the root `AND` chain is examined (the same conjuncts as +/// [`split_conjunction`]). An `Optional` in any other position, for example +/// `NOT(Optional(x))` or `Optional(x) OR y`, stays inside a required conjunct. +/// +/// For example, `a AND Optional(b) AND (c AND Optional(d))` gives +/// `([a, c], [b, d])`. +#[expect(clippy::type_complexity)] +pub fn split_optional( + predicate: &Arc, +) -> (Vec>, Vec>) { + let mut required = vec![]; + let mut optional = vec![]; + for conjunct in split_conjunction(predicate) { + match conjunct.downcast_ref::() { + Some(opt) => optional.push(Arc::clone(opt.inner())), + None => required.push(Arc::clone(conjunct)), + } + } + (required, optional) +} + +/// Returns `true` if `expr` itself is an [`OptionalFilterPhysicalExpr`]. +/// +/// This does not examine the children of `expr`. +pub fn is_optional_filter(expr: &Arc) -> bool { + expr.is::() +} + impl ConstExpr { /// Collects predicate-derived constants from equality conjunctions. /// @@ -644,4 +680,102 @@ pub(crate) mod tests { Ok(()) } + + fn optional_test_schema() -> Schema { + Schema::new(vec![ + Field::new("a", DataType::Boolean, true), + Field::new("b", DataType::Boolean, true), + Field::new("c", DataType::Boolean, true), + Field::new("d", DataType::Boolean, true), + ]) + } + + fn optional(inner: Arc) -> Arc { + Arc::new(OptionalFilterPhysicalExpr::new(inner)) + } + + fn and( + left: Arc, + right: Arc, + ) -> Arc { + Arc::new(BinaryExpr::new(left, Operator::And, right)) + } + + fn or( + left: Arc, + right: Arc, + ) -> Arc { + Arc::new(BinaryExpr::new(left, Operator::Or, right)) + } + + fn to_strings(exprs: &[Arc]) -> Vec { + exprs.iter().map(|e| e.to_string()).collect() + } + + #[test] + fn test_split_optional_root_chain() -> Result<()> { + let schema = optional_test_schema(); + let [a, b, c, d] = ["a", "b", "c", "d"].map(|n| col(n, &schema).unwrap()); + + // a AND Optional(b) AND (c AND Optional(d)) + let predicate = and( + and(Arc::clone(&a), optional(Arc::clone(&b))), + and(Arc::clone(&c), optional(Arc::clone(&d))), + ); + let (required, optional_filters) = split_optional(&predicate); + assert_eq!(to_strings(&required), vec!["a@0", "c@2"]); + assert_eq!(to_strings(&optional_filters), vec!["b@1", "d@3"]); + + // A single Optional at the root is optional. + let predicate = optional(Arc::clone(&a)); + let (required, optional_filters) = split_optional(&predicate); + assert!(required.is_empty()); + assert_eq!(to_strings(&optional_filters), vec!["a@0"]); + + // An Optional that wraps an AND is one optional filter. + let predicate = optional(and(Arc::clone(&a), Arc::clone(&b))); + let (required, optional_filters) = split_optional(&predicate); + assert!(required.is_empty()); + assert_eq!(to_strings(&optional_filters), vec!["a@0 AND b@1"]); + + // No Optional: everything is required. + let predicate = and(Arc::clone(&a), Arc::clone(&b)); + let (required, optional_filters) = split_optional(&predicate); + assert_eq!(to_strings(&required), vec!["a@0", "b@1"]); + assert!(optional_filters.is_empty()); + Ok(()) + } + + #[test] + fn test_split_optional_not_on_root_chain() -> Result<()> { + let schema = optional_test_schema(); + let [x, y] = ["a", "b"].map(|n| col(n, &schema).unwrap()); + + // NOT(Optional(x)) is required + let predicate = crate::expressions::not(optional(Arc::clone(&x)))?; + let (required, optional_filters) = split_optional(&predicate); + assert_eq!(required, vec![Arc::clone(&predicate)]); + assert!(optional_filters.is_empty()); + + // Optional(x) OR y is required + let predicate = or(optional(Arc::clone(&x)), Arc::clone(&y)); + let (required, optional_filters) = split_optional(&predicate); + assert_eq!(required, vec![Arc::clone(&predicate)]); + assert!(optional_filters.is_empty()); + Ok(()) + } + + #[test] + fn test_is_optional_filter() -> Result<()> { + let schema = optional_test_schema(); + let a = col("a", &schema)?; + + assert!(is_optional_filter(&optional(Arc::clone(&a)))); + assert!(!is_optional_filter(&a)); + // Only the node itself is examined. + assert!(!is_optional_filter(&crate::expressions::not(optional( + Arc::clone(&a) + ))?)); + Ok(()) + } } diff --git a/datafusion/proto-models/proto/datafusion.proto b/datafusion/proto-models/proto/datafusion.proto index ccd830ef74eea..7fd9529a2ee25 100644 --- a/datafusion/proto-models/proto/datafusion.proto +++ b/datafusion/proto-models/proto/datafusion.proto @@ -1093,6 +1093,8 @@ message PhysicalExprNode { PhysicalLambdaVariableExprNode lambda_variable = 26; PhysicalRangeExprNode range_expr = 27; PhysicalSqlSimilarToPatternNode sql_similar_to_pattern = 28; + + PhysicalOptionalFilterNode optional_filter = 29; } } @@ -1104,6 +1106,11 @@ message PhysicalDynamicFilterNode { bool is_complete = 5; } +// Marks the wrapped filter as optional: it is not needed for correctness. +message PhysicalOptionalFilterNode { + PhysicalExprNode inner = 1; +} + message PhysicalSqlSimilarToPatternNode { PhysicalExprNode expr = 1; } diff --git a/datafusion/proto-models/src/generated/pbjson.rs b/datafusion/proto-models/src/generated/pbjson.rs index 4779760866fc6..ef8b74442c71e 100644 --- a/datafusion/proto-models/src/generated/pbjson.rs +++ b/datafusion/proto-models/src/generated/pbjson.rs @@ -19567,6 +19567,9 @@ impl serde::Serialize for PhysicalExprNode { physical_expr_node::ExprType::SqlSimilarToPattern(v) => { struct_ser.serialize_field("sqlSimilarToPattern", v)?; } + physical_expr_node::ExprType::OptionalFilter(v) => { + struct_ser.serialize_field("optionalFilter", v)?; + } } } struct_ser.end() @@ -19626,6 +19629,8 @@ impl<'de> serde::Deserialize<'de> for PhysicalExprNode { "rangeExpr", "sql_similar_to_pattern", "sqlSimilarToPattern", + "optional_filter", + "optionalFilter", ]; #[allow(clippy::enum_variant_names)] @@ -19657,6 +19662,7 @@ impl<'de> serde::Deserialize<'de> for PhysicalExprNode { LambdaVariable, RangeExpr, SqlSimilarToPattern, + OptionalFilter, } impl<'de> serde::Deserialize<'de> for GeneratedField { fn deserialize(deserializer: D) -> std::result::Result @@ -19705,6 +19711,7 @@ impl<'de> serde::Deserialize<'de> for PhysicalExprNode { "lambdaVariable" | "lambda_variable" => Ok(GeneratedField::LambdaVariable), "rangeExpr" | "range_expr" => Ok(GeneratedField::RangeExpr), "sqlSimilarToPattern" | "sql_similar_to_pattern" => Ok(GeneratedField::SqlSimilarToPattern), + "optionalFilter" | "optional_filter" => Ok(GeneratedField::OptionalFilter), _ => Err(serde::de::Error::unknown_field(value, FIELDS)), } } @@ -19916,6 +19923,13 @@ impl<'de> serde::Deserialize<'de> for PhysicalExprNode { return Err(serde::de::Error::duplicate_field("sqlSimilarToPattern")); } expr_type__ = map_.next_value::<::std::option::Option<_>>()?.map(physical_expr_node::ExprType::SqlSimilarToPattern) +; + } + GeneratedField::OptionalFilter => { + if expr_type__.is_some() { + return Err(serde::de::Error::duplicate_field("optionalFilter")); + } + expr_type__ = map_.next_value::<::std::option::Option<_>>()?.map(physical_expr_node::ExprType::OptionalFilter) ; } } @@ -21377,6 +21391,97 @@ impl<'de> serde::Deserialize<'de> for PhysicalNot { deserializer.deserialize_struct("datafusion.PhysicalNot", FIELDS, GeneratedVisitor) } } +impl serde::Serialize for PhysicalOptionalFilterNode { + #[allow(deprecated)] + fn serialize(&self, serializer: S) -> std::result::Result + where + S: serde::Serializer, + { + use serde::ser::SerializeStruct; + let mut len = 0; + if self.inner.is_some() { + len += 1; + } + let mut struct_ser = serializer.serialize_struct("datafusion.PhysicalOptionalFilterNode", len)?; + if let Some(v) = self.inner.as_ref() { + struct_ser.serialize_field("inner", v)?; + } + struct_ser.end() + } +} +impl<'de> serde::Deserialize<'de> for PhysicalOptionalFilterNode { + #[allow(deprecated)] + fn deserialize(deserializer: D) -> std::result::Result + where + D: serde::Deserializer<'de>, + { + const FIELDS: &[&str] = &[ + "inner", + ]; + + #[allow(clippy::enum_variant_names)] + enum GeneratedField { + Inner, + } + impl<'de> serde::Deserialize<'de> for GeneratedField { + fn deserialize(deserializer: D) -> std::result::Result + where + D: serde::Deserializer<'de>, + { + struct GeneratedVisitor; + + impl serde::de::Visitor<'_> for GeneratedVisitor { + type Value = GeneratedField; + + fn expecting(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(formatter, "expected one of: {:?}", &FIELDS) + } + + #[allow(unused_variables)] + fn visit_str(self, value: &str) -> std::result::Result + where + E: serde::de::Error, + { + match value { + "inner" => Ok(GeneratedField::Inner), + _ => Err(serde::de::Error::unknown_field(value, FIELDS)), + } + } + } + deserializer.deserialize_identifier(GeneratedVisitor) + } + } + struct GeneratedVisitor; + impl<'de> serde::de::Visitor<'de> for GeneratedVisitor { + type Value = PhysicalOptionalFilterNode; + + fn expecting(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + formatter.write_str("struct datafusion.PhysicalOptionalFilterNode") + } + + fn visit_map(self, mut map_: V) -> std::result::Result + where + V: serde::de::MapAccess<'de>, + { + let mut inner__ = None; + while let Some(k) = map_.next_key()? { + match k { + GeneratedField::Inner => { + if inner__.is_some() { + return Err(serde::de::Error::duplicate_field("inner")); + } + inner__ = map_.next_value()?; + } + } + } + Ok(PhysicalOptionalFilterNode { + inner: inner__, + }) + } + } + deserializer.deserialize_struct("datafusion.PhysicalOptionalFilterNode", FIELDS, GeneratedVisitor) + } +} impl serde::Serialize for PhysicalPlanNode { #[allow(deprecated)] fn serialize(&self, serializer: S) -> std::result::Result diff --git a/datafusion/proto-models/src/generated/prost.rs b/datafusion/proto-models/src/generated/prost.rs index 5c4e4ce4de200..f2e0d8a56d5f5 100644 --- a/datafusion/proto-models/src/generated/prost.rs +++ b/datafusion/proto-models/src/generated/prost.rs @@ -1613,7 +1613,7 @@ pub struct PhysicalExprNode { pub expr_id: ::core::option::Option, #[prost( oneof = "physical_expr_node::ExprType", - tags = "1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 14, 15, 16, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28" + tags = "1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 14, 15, 16, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29" )] pub expr_type: ::core::option::Option, } @@ -1682,6 +1682,8 @@ pub mod physical_expr_node { SqlSimilarToPattern( ::prost::alloc::boxed::Box, ), + #[prost(message, tag = "29")] + OptionalFilter(::prost::alloc::boxed::Box), } } #[derive(Clone, PartialEq, ::prost::Message)] @@ -1697,6 +1699,12 @@ pub struct PhysicalDynamicFilterNode { #[prost(bool, tag = "5")] pub is_complete: bool, } +/// Marks the wrapped filter as optional: it is not needed for correctness. +#[derive(Clone, PartialEq, ::prost::Message)] +pub struct PhysicalOptionalFilterNode { + #[prost(message, optional, boxed, tag = "1")] + pub inner: ::core::option::Option<::prost::alloc::boxed::Box>, +} #[derive(Clone, PartialEq, ::prost::Message)] pub struct PhysicalSqlSimilarToPatternNode { #[prost(message, optional, boxed, tag = "1")] diff --git a/datafusion/proto/src/physical_plan/from_proto.rs b/datafusion/proto/src/physical_plan/from_proto.rs index bc443149df413..91ec5180c22ed 100644 --- a/datafusion/proto/src/physical_plan/from_proto.rs +++ b/datafusion/proto/src/physical_plan/from_proto.rs @@ -52,7 +52,9 @@ use super::{ }; use crate::protobuf::physical_expr_node::ExprType; use crate::{convert_required, protobuf}; -use datafusion_physical_expr::expressions::DynamicFilterPhysicalExpr; +use datafusion_physical_expr::expressions::{ + DynamicFilterPhysicalExpr, OptionalFilterPhysicalExpr, +}; /// Parses a physical sort expression from a protobuf. /// @@ -362,6 +364,9 @@ pub fn parse_physical_expr_with_converter( ExprType::DynamicFilter(_) => { DynamicFilterPhysicalExpr::try_from_proto(proto, &decode_ctx)? } + ExprType::OptionalFilter(_) => { + OptionalFilterPhysicalExpr::try_from_proto(proto, &decode_ctx)? + } ExprType::SqlSimilarToPattern(_) => { SqlSimilarToPattern::try_from_proto(proto, &decode_ctx)? } diff --git a/datafusion/proto/tests/cases/plans/dynamic_filters.rs b/datafusion/proto/tests/cases/plans/dynamic_filters.rs index 7892ffd9a1ab0..7d07e59233411 100644 --- a/datafusion/proto/tests/cases/plans/dynamic_filters.rs +++ b/datafusion/proto/tests/cases/plans/dynamic_filters.rs @@ -40,7 +40,8 @@ use datafusion::physical_plan::aggregates::{ }; use datafusion::physical_plan::empty::EmptyExec; use datafusion::physical_plan::expressions::{ - BinaryExpr, Column, DynamicFilterPhysicalExpr, PhysicalSortExpr, lit, + BinaryExpr, Column, DynamicFilterPhysicalExpr, OptionalFilterPhysicalExpr, + PhysicalSortExpr, lit, }; use datafusion::physical_plan::filter::FilterExec; use datafusion::physical_plan::joins::{HashJoinExec, PartitionMode}; @@ -326,6 +327,68 @@ fn test_dynamic_filter_roundtrip_dedupe() -> Result<()> { Ok(()) } +// An `Optional` wrapper survives the roundtrip, and a dynamic filter inside it +// is deduped with an unwrapped clone of the same dynamic filter. +#[test] +fn test_optional_dynamic_filter_roundtrip_dedupe() -> Result<()> { + let schema = Arc::new(Schema::new(vec![Field::new("a", DataType::Int64, false)])); + let dynamic_filter = make_dynamic_filter(); + let optional_filter = + Arc::new(OptionalFilterPhysicalExpr::new(Arc::clone(&dynamic_filter))) + as Arc; + + let (optional_after_roundtrip, dynamic_after_roundtrip) = + roundtrip_dynamic_filter_expr_pair( + Arc::clone(&optional_filter), + Arc::clone(&dynamic_filter), + schema, + )?; + + assert_eq!( + optional_filter.to_string(), + optional_after_roundtrip.to_string() + ); + let inner_after_roundtrip = optional_after_roundtrip + .downcast_ref::() + .expect("Expected OptionalFilterPhysicalExpr") + .inner(); + assert_dynamic_filters_equal(&dynamic_filter, inner_after_roundtrip); + assert_dynamic_filters_equal(&dynamic_filter, &dynamic_after_roundtrip); + + // Assert referential integrity through the wrapper. + assert_dynamic_filter_update_is_visible( + inner_after_roundtrip, + &dynamic_after_roundtrip, + )?; + + Ok(()) +} + +// An `Optional` wrapper around a static expression survives the roundtrip. +#[test] +fn test_optional_filter_roundtrip() -> Result<()> { + let schema = Arc::new(Schema::new(vec![Field::new("a", DataType::Int64, false)])); + let expr = Arc::new(OptionalFilterPhysicalExpr::new(Arc::new(BinaryExpr::new( + Arc::new(Column::new("a", 0)), + Operator::Gt, + lit(5_i64), + )))) as Arc; + + let codec = DefaultPhysicalExtensionCodec {}; + let converter = DefaultPhysicalProtoConverter {}; + let proto = converter.physical_expr_to_proto(&expr, &codec)?; + let ctx = SessionContext::new(); + let task_ctx = ctx.task_ctx(); + let decode_ctx = PhysicalPlanDecodeContext::new(task_ctx.as_ref(), &codec); + let roundtrip = converter.proto_to_physical_expr(&proto, &schema, &decode_ctx)?; + + assert!(roundtrip.is::()); + assert_eq!(&expr, &roundtrip); + assert_eq!(roundtrip.to_string(), "Optional(a@0 > 5)"); + + Ok(()) +} + /// Roundtrip test for an execution plan where there are multiple instances of a dynamic filter /// with different children. #[test] diff --git a/datafusion/pruning/src/pruning_predicate.rs b/datafusion/pruning/src/pruning_predicate.rs index b49b72058e0cd..0332ff903b005 100644 --- a/datafusion/pruning/src/pruning_predicate.rs +++ b/datafusion/pruning/src/pruning_predicate.rs @@ -3448,6 +3448,47 @@ mod tests { assert_eq!(result, expected); } + /// Pruning sees through `OptionalFilterPhysicalExpr`: an optional filter + /// prunes the same as the inner filter. + #[test] + fn prune_optional_filter_same_as_inner() { + let (schema, statistics) = int32_setup(); + let expected = &[true, true, false, true, true]; + + let required = logical2physical(&col("i").gt(lit(0)), &schema); + let optional = Arc::new(phys_expr::OptionalFilterPhysicalExpr::new(Arc::clone( + &required, + ))) as Arc; + let dynamic = Arc::new(DynamicFilterPhysicalExpr::new( + collect_columns(&required) + .into_iter() + .map(|c| Arc::new(c) as Arc) + .collect(), + Arc::clone(&required), + )) as Arc; + let optional_dynamic = + Arc::new(phys_expr::OptionalFilterPhysicalExpr::new(dynamic)) + as Arc; + + let build = |expr: Arc| { + PruningPredicateBuilder::new() + .with_file_schema(Arc::clone(&schema)) + .try_build(expr) + .unwrap() + }; + let baseline = build(required); + assert_eq!(baseline.prune(&statistics).unwrap(), expected); + + for expr in [optional, optional_dynamic] { + let p = build(expr); + assert_eq!( + p.predicate_expr().to_string(), + baseline.predicate_expr().to_string() + ); + assert_eq!(p.prune(&statistics).unwrap(), expected); + } + } + #[test] fn row_group_predicate_lt_bool() -> Result<()> { let schema = Schema::new(vec![Field::new("c1", DataType::Boolean, false)]); From 2246afc4cb3c6d92fa7d816fecd4482c1c8b7967 Mon Sep 17 00:00:00 2001 From: Adrian Garcia Badaracco <1755071+adriangb@users.noreply.github.com> Date: Thu, 24 Sep 2026 18:19:41 -0500 Subject: [PATCH 07/42] feat: keep optional filters pruning-only in the Parquet post-scan path With `pushdown_filters = false`, the scan evaluates all accepted filters for each row after the decode. This includes the optional filters (hash join, TopK and aggregate dynamic filters, which the producers wrap in `Optional(...)`). The join dynamic filter evaluated for each row in the scan caused a 1.16x TPC-H regression. Optional filters are not needed for correctness. Thus the post-scan filter now never gets an optional conjunct (a root `AND` conjunct found with `split_optional`): - `pushdown_filters = false`: only the required conjuncts run post-scan. Optional conjuncts are used only for statistics, page index, bloom filter and file pruning. - `pushdown_filters = true`: an optional conjunct that the row filter rejects for a file is not used for that file. A required conjunct that is rejected still runs post-scan. - A whole-file row filter build error sends only the required conjuncts to the post-scan filter. Accepted optional conjuncts with `pushdown_filters = true` stay row filter predicates, as before. Co-Authored-By: Claude Opus 5.5 --- .../datasource-parquet/src/opener/mod.rs | 146 +++++++++++++----- .../datasource-parquet/src/push_decoder.rs | 14 +- .../datasource-parquet/src/row_filter.rs | 34 +++- 3 files changed, 143 insertions(+), 51 deletions(-) diff --git a/datafusion/datasource-parquet/src/opener/mod.rs b/datafusion/datasource-parquet/src/opener/mod.rs index 5108ff93890f8..a2d5c7d495345 100644 --- a/datafusion/datasource-parquet/src/opener/mod.rs +++ b/datafusion/datasource-parquet/src/opener/mod.rs @@ -61,7 +61,7 @@ use datafusion_common::{ use datafusion_datasource::{PartitionedFile, TableSchema}; use datafusion_physical_expr::expressions::{Column, DynamicFilterTracking, Literal}; use datafusion_physical_expr::simplifier::PhysicalExprSimplifier; -use datafusion_physical_expr::utils::collect_columns; +use datafusion_physical_expr::utils::{collect_columns, split_optional}; use datafusion_physical_expr_adapter::PhysicalExprAdapterFactory; use datafusion_physical_expr_common::physical_expr::PhysicalExpr; use datafusion_physical_expr_common::sort_expr::LexOrdering; @@ -558,7 +558,10 @@ impl DecoderReadPlans { // `RowFilter` machinery cannot evaluate on this file — the rejected // conjuncts returned by `RowFilterContext::try_new`). // - // Either way every conjunct is applied; nothing is silently dropped. + // Either way every required conjunct is applied; nothing is silently + // dropped. Optional conjuncts (see `split_optional`) are not needed + // for correctness. They never go to the post-scan filter: they are + // row filter predicates or they are used only for statistics pruning. // --------------------------------------------------------------- // A pruning-only predicate is applied by a `FilterExec` above the // scan: the scan uses it only to prune. @@ -582,15 +585,13 @@ impl DecoderReadPlans { prepared.file_metrics.clone(), prepared.max_predicate_cache_size, ), - // Pushdown disabled: the whole predicate runs post-scan (in-scan - // equivalent of a `FilterExec`). - (false, Some(predicate)) => ( - None, - datafusion_physical_expr::split_conjunction(predicate) - .into_iter() - .cloned() - .collect(), - ), + // Pushdown disabled: the required conjuncts run post-scan + // (in-scan equivalent of a `FilterExec`). Optional conjuncts + // (for example hash join dynamic filters) are not needed for + // correctness and are expensive to evaluate for each row, + // thus they are used only for statistics pruning, as before + // the scan accepted the filters. + (false, Some(predicate)) => (None, split_optional(predicate).0), (_, None) => (None, Vec::new()), }; @@ -5396,13 +5397,101 @@ mod test { /// post-scan filter, so only the rows with a non-null struct survive. #[tokio::test] async fn rejected_struct_conjunct_runs_post_scan_not_dropped() { + let store = Arc::new(InMemory::new()) as Arc; + let (schema, file) = write_struct_file(&store).await; + + // `s IS NOT NULL` references a whole struct, which `PushdownChecker` + // flags as non-primitive — `FilterCandidateBuilder::build` returns + // `Ok(None)` and the conjunct lands in `rejected`. + let predicate = logical2physical(&col("s").is_not_null(), &schema); + + let morselizer = ParquetMorselizerBuilder::new() + .with_store(Arc::clone(&store)) + .with_schema(Arc::clone(&schema)) + .with_predicate(predicate) + // The RowFilter path: emulates the post-`try_pushdown_filters` + // state where the parent `FilterExec` has already been removed + // and the scan owns the conjunct. + .with_pushdown_filters(true) + .build(); + + let stream = open_file(&morselizer, file).await.unwrap(); + let (_, rows) = count_batches_and_rows(stream).await; + + // 2 rows have a non-null struct. Before the fix this returned 3 + // (the conjunct was silently dropped). + assert_eq!( + rows, 2, + "expected 2 rows with non-null struct; the rejected conjunct must \ + be applied post-scan, not silently dropped" + ); + } + + /// An optional conjunct is not needed for correctness. The scan never + /// evaluates it after the decode: not when `pushdown_filters` is false, + /// and not when the `RowFilter` rejects it for the file. A required + /// conjunct in the same situations is evaluated after the decode. + #[tokio::test] + async fn optional_conjunct_is_never_evaluated_post_scan() { + let store = Arc::new(InMemory::new()) as Arc; + let (schema, file) = write_struct_file(&store).await; + + // `s IS NOT NULL` is rejected by the `RowFilter` (whole struct). + // `id > 1` can be a `RowFilter` predicate. + let rejected = logical2physical(&col("s").is_not_null(), &schema); + let pushable = logical2physical(&col("id").gt(lit(1)), &schema); + let optional = |expr: &Arc| -> Arc { + Arc::new( + datafusion_physical_expr::expressions::OptionalFilterPhysicalExpr::new( + Arc::clone(expr), + ), + ) + }; + + // (predicate, pushdown_filters, expected rows, expected post-scan rows) + let cases: Vec<(Arc, bool, usize, usize)> = vec![ + // Required conjuncts: applied post-scan (#22384). + (Arc::clone(&rejected), false, 2, 3), + (Arc::clone(&rejected), true, 2, 3), + (Arc::clone(&pushable), false, 2, 3), + // Optional conjuncts: never evaluated post-scan. + (optional(&rejected), false, 3, 0), + (optional(&rejected), true, 3, 0), + (optional(&pushable), false, 3, 0), + // An optional conjunct that the `RowFilter` accepts is a row + // filter predicate. + (optional(&pushable), true, 2, 0), + ]; + for (predicate, pushdown, expected_rows, expected_post_scan_rows) in cases { + let metrics = ExecutionPlanMetricsSet::new(); + let morselizer = ParquetMorselizerBuilder::new() + .with_store(Arc::clone(&store)) + .with_schema(Arc::clone(&schema)) + .with_predicate(Arc::clone(&predicate)) + .with_pushdown_filters(pushdown) + .with_metrics(metrics.clone()) + .build(); + let stream = open_file(&morselizer, file.clone()).await.unwrap(); + let (_, rows) = count_batches_and_rows(stream).await; + let post_scan_rows = counter_metric_value(&metrics, "post_scan_rows_pruned") + + counter_metric_value(&metrics, "post_scan_rows_matched"); + assert_eq!( + (rows, post_scan_rows), + (expected_rows, expected_post_scan_rows), + "predicate {predicate}, pushdown_filters {pushdown}" + ); + } + } + + /// Writes a file with the columns `id` (Int32: 1, 2, 3) and `s` + /// (Struct{value: Int32, label: Utf8}; row 1 is null). + async fn write_struct_file( + store: &Arc, + ) -> (SchemaRef, PartitionedFile) { use arrow::array::{Int32Array, StringArray, StructArray}; use arrow::buffer::NullBuffer; use arrow::datatypes::Fields; - let store = Arc::new(InMemory::new()) as Arc; - - // Schema: id (Int32), s (Struct{value: Int32, label: Utf8}). let struct_fields: Fields = vec![ Arc::new(Field::new("value", DataType::Int32, true)), Arc::new(Field::new("label", DataType::Utf8, true)), @@ -5432,7 +5521,7 @@ mod test { .unwrap(); let data_size = write_parquet_batches( - Arc::clone(&store), + Arc::clone(store), "rejected.parquet", vec![batch], None, @@ -5440,32 +5529,7 @@ mod test { .await; let file = PartitionedFile::new("rejected.parquet".to_string(), data_size as u64); - - // `s IS NOT NULL` references a whole struct, which `PushdownChecker` - // flags as non-primitive — `FilterCandidateBuilder::build` returns - // `Ok(None)` and the conjunct lands in `rejected`. - let predicate = logical2physical(&col("s").is_not_null(), &schema); - - let morselizer = ParquetMorselizerBuilder::new() - .with_store(Arc::clone(&store)) - .with_schema(Arc::clone(&schema)) - .with_predicate(predicate) - // The RowFilter path: emulates the post-`try_pushdown_filters` - // state where the parent `FilterExec` has already been removed - // and the scan owns the conjunct. - .with_pushdown_filters(true) - .build(); - - let stream = open_file(&morselizer, file).await.unwrap(); - let (_, rows) = count_batches_and_rows(stream).await; - - // 2 rows have a non-null struct. Before the fix this returned 3 - // (the conjunct was silently dropped). - assert_eq!( - rows, 2, - "expected 2 rows with non-null struct; the rejected conjunct must \ - be applied post-scan, not silently dropped" - ); + (schema, file) } /// Helpers for tests that exercise parquet virtual columns diff --git a/datafusion/datasource-parquet/src/push_decoder.rs b/datafusion/datasource-parquet/src/push_decoder.rs index e38cec26da65a..61464405a06e9 100644 --- a/datafusion/datasource-parquet/src/push_decoder.rs +++ b/datafusion/datasource-parquet/src/push_decoder.rs @@ -58,7 +58,7 @@ use parquet::file::metadata::ParquetMetaData; use datafusion_common::{DataFusionError, Result, internal_err}; use datafusion_physical_expr::expressions::DynamicFilterTracking; -use datafusion_physical_expr::split_conjunction; +use datafusion_physical_expr::utils::split_optional; use datafusion_physical_expr_common::physical_expr::PhysicalExpr; use datafusion_physical_plan::metrics::{BaselineMetrics, Count, Gauge}; use datafusion_pruning::{PruningPredicate, PruningPredicateBuilder}; @@ -412,15 +412,15 @@ impl RowFilterContext { (context, rejected) } Err(e) => { - // Whole-file build failure: route every conjunct post-scan - // rather than silently dropping the predicate. + // Whole-file build failure: route every required conjunct + // post-scan rather than silently dropping the predicate. + // Optional conjuncts are not needed for correctness, thus + // they are not evaluated after the scan. debug!( "Ignoring error prebuilding row filter candidates: {e}; \ - all conjuncts will be evaluated post-scan" + all required conjuncts will be evaluated post-scan" ); - let rejected = - split_conjunction(predicate).into_iter().cloned().collect(); - (None, rejected) + (None, split_optional(predicate).0) } } } diff --git a/datafusion/datasource-parquet/src/row_filter.rs b/datafusion/datasource-parquet/src/row_filter.rs index 96c304a38f661..8cf0a2468c628 100644 --- a/datafusion/datasource-parquet/src/row_filter.rs +++ b/datafusion/datasource-parquet/src/row_filter.rs @@ -78,7 +78,7 @@ use parquet::file::metadata::ParquetMetaData; use datafusion_common::Result; use datafusion_common::cast::as_boolean_array; use datafusion_common::tree_node::TreeNode; -use datafusion_physical_expr::utils::reassign_expr_columns; +use datafusion_physical_expr::utils::{is_optional_filter, reassign_expr_columns}; use datafusion_physical_expr::{PhysicalExpr, split_conjunction}; use datafusion_physical_plan::metrics; @@ -416,11 +416,14 @@ fn size_of_columns(columns: &[usize], metadata: &ParquetMetaData) -> Result, @@ -512,6 +520,26 @@ pub(crate) fn prebuild_row_filter_candidates( let mut candidates: Vec = Vec::with_capacity(predicates.len()); let mut rejected: Vec> = Vec::new(); for predicate in predicates { + // An optional conjunct that cannot be pushed down for this file is not + // needed for correctness: do not use it, and do not send it to the + // post-scan filter. + if is_optional_filter(predicate) { + match FilterCandidateBuilder::new( + Arc::clone(predicate), + Arc::clone(file_schema), + ) + .build(metadata) + { + Ok(Some(candidate)) => candidates.push(candidate), + Ok(None) => {} + Err(e) => { + log::debug!( + "Ignoring optional filter that cannot be pushed down: {e}" + ); + } + } + continue; + } match FilterCandidateBuilder::new(Arc::clone(predicate), Arc::clone(file_schema)) .build(metadata)? { From ab2b6e27efcca86549c7676174b1f60cc2f3c0ba Mon Sep 17 00:00:00 2001 From: Adrian Garcia Badaracco <1755071+adriangb@users.noreply.github.com> Date: Thu, 24 Sep 2026 14:52:54 -0500 Subject: [PATCH 08/42] feat: add filter_stats (FilterCost, Clock) for filter measurements Add the `datafusion_physical_expr::filter_stats` module with the shared primitives that adaptive filter code uses to measure filters at runtime: - `Clock`: a monotonic clock in nanoseconds that tests can replace. `SystemClock` is the real clock. `ManualClock` moves only when a test moves it, thus decisions that use time are deterministic in tests. - `FilterCost`: the rows in, the rows out and the evaluation time of one filter, and the derived cost for each row and rows removed for each nanosecond. - `duration_nanos`: a `Duration` in nanoseconds, saturated to `u64::MAX`. No code uses the module yet, thus behavior does not change. Co-Authored-By: Claude Opus 5.5 --- datafusion/physical-expr/src/filter_stats.rs | 199 +++++++++++++++++++ datafusion/physical-expr/src/lib.rs | 1 + 2 files changed, 200 insertions(+) create mode 100644 datafusion/physical-expr/src/filter_stats.rs diff --git a/datafusion/physical-expr/src/filter_stats.rs b/datafusion/physical-expr/src/filter_stats.rs new file mode 100644 index 0000000000000..f5f5c4f03f8b1 --- /dev/null +++ b/datafusion/physical-expr/src/filter_stats.rs @@ -0,0 +1,199 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Measurements of filter evaluations at runtime. +//! +//! Operators that adapt to the data measure the filters that they evaluate: +//! the rows in, the rows that pass and the evaluation time. This module has +//! the shared parts: +//! +//! * [`Clock`]: a monotonic clock that tests can replace, so that decisions +//! that use time are deterministic in tests. [`SystemClock`] is the real +//! clock and [`ManualClock`] is a clock that only moves when a test moves +//! it. +//! * [`FilterCost`]: the counts and the time of one filter, and the values +//! derived from them (cost for each row, rows removed for each +//! nanosecond). +//! +//! For example, an operator can use them to pause a filter that costs more +//! than it saves, or to change the order of the conjuncts of a predicate. + +use std::fmt::Debug; +use std::sync::Arc; +use std::sync::atomic::{AtomicU64, Ordering}; +use std::time::Duration; + +use datafusion_common::instant::Instant; + +/// A monotonic clock in nanoseconds. +/// +/// Production code uses [`SystemClock`]. Tests use [`ManualClock`] (or their +/// own implementation), thus decisions that use time are deterministic in +/// tests. +pub trait Clock: Debug + Send + Sync { + /// Nanoseconds since an arbitrary fixed point. The value never + /// decreases. + fn now_nanos(&self) -> u64; +} + +/// The real monotonic [`Clock`]. +#[derive(Debug, Clone, Copy)] +pub struct SystemClock { + start: Instant, +} + +impl SystemClock { + /// Creates a clock whose zero is now. + pub fn new() -> Self { + Self { + start: Instant::now(), + } + } + + /// A shared [`SystemClock`], as a trait object. + pub fn shared() -> Arc { + Arc::new(Self::new()) + } +} + +impl Default for SystemClock { + fn default() -> Self { + Self::new() + } +} + +impl Clock for SystemClock { + fn now_nanos(&self) -> u64 { + u64::try_from(self.start.elapsed().as_nanos()).unwrap_or(u64::MAX) + } +} + +/// A [`Clock`] that moves only when [`Self::advance`] is called. For tests. +#[derive(Debug, Default)] +pub struct ManualClock { + nanos: AtomicU64, +} + +impl ManualClock { + /// Creates a clock at zero. + pub fn new() -> Self { + Self::default() + } + + /// Moves the clock forward by `nanos` nanoseconds. + pub fn advance(&self, nanos: u64) { + self.nanos.fetch_add(nanos, Ordering::Relaxed); + } +} + +impl Clock for ManualClock { + fn now_nanos(&self) -> u64 { + self.nanos.load(Ordering::Relaxed) + } +} + +/// Returns the nanoseconds in `elapsed`, saturated to `u64::MAX`. +pub fn duration_nanos(elapsed: Duration) -> u64 { + u64::try_from(elapsed.as_nanos()).unwrap_or(u64::MAX) +} + +/// The measurements of one filter: the rows that it was evaluated on, the +/// rows that passed it, and the evaluation time. +#[derive(Debug, Default, Clone, Copy, PartialEq, Eq)] +pub struct FilterCost { + /// Rows that the filter was evaluated on. + pub rows_in: u64, + /// Rows that passed the filter (`true`; `null` does not pass). + pub rows_out: u64, + /// Evaluation time in nanoseconds. + pub nanos: u64, +} + +impl FilterCost { + /// Adds the result of one evaluation. `rows_out` larger than `rows_in` + /// is used as `rows_in`. + pub fn add(&mut self, rows_in: u64, rows_out: u64, nanos: u64) { + self.rows_in = self.rows_in.saturating_add(rows_in); + self.rows_out = self.rows_out.saturating_add(rows_out.min(rows_in)); + self.nanos = self.nanos.saturating_add(nanos); + } + + /// Rows that the filter removed. + pub fn rows_removed(&self) -> u64 { + self.rows_in.saturating_sub(self.rows_out) + } + + /// Nanoseconds for each evaluated row, or `None` if the filter was not + /// evaluated on any row. + pub fn nanos_per_row(&self) -> Option { + (self.rows_in > 0).then(|| self.nanos as f64 / self.rows_in as f64) + } + + /// Rows removed for each nanosecond, `(1 + rows_in - rows_out) / nanos`, + /// or `None` if the filter was not evaluated on any row. A larger value + /// is a better filter to evaluate first. This is the ranking key of + /// Velox (Pedreira et al., VLDB 2022). The `1 +` ranks a filter that + /// removes no rows by its cost, and a zero time is used as 1 ns. + pub fn rows_removed_per_nano(&self) -> Option { + (self.rows_in > 0) + .then(|| (1 + self.rows_removed()) as f64 / self.nanos.max(1) as f64) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn filter_cost_derived_values() { + let empty = FilterCost::default(); + assert_eq!(empty.nanos_per_row(), None); + assert_eq!(empty.rows_removed_per_nano(), None); + + let mut cost = FilterCost::default(); + cost.add(100, 25, 1_000); + cost.add(100, 200, 1_000); + assert_eq!(cost.rows_in, 200); + // `rows_out` is at most `rows_in` for each evaluation. + assert_eq!(cost.rows_out, 125); + assert_eq!(cost.rows_removed(), 75); + assert_eq!(cost.nanos_per_row(), Some(10.0)); + assert_eq!(cost.rows_removed_per_nano(), Some(76.0 / 2_000.0)); + + // A zero time is used as 1 ns. + let mut free = FilterCost::default(); + free.add(10, 0, 0); + assert_eq!(free.rows_removed_per_nano(), Some(11.0)); + } + + #[test] + fn manual_clock_moves_only_when_advanced() { + let clock = ManualClock::new(); + assert_eq!(clock.now_nanos(), 0); + clock.advance(5); + clock.advance(7); + assert_eq!(clock.now_nanos(), 12); + } + + #[test] + fn system_clock_is_monotonic() { + let clock = SystemClock::new(); + let first = clock.now_nanos(); + assert!(clock.now_nanos() >= first); + assert_eq!(duration_nanos(Duration::from_micros(3)), 3_000); + } +} diff --git a/datafusion/physical-expr/src/lib.rs b/datafusion/physical-expr/src/lib.rs index 80e9f88b510ed..75d9d7ea70c2b 100644 --- a/datafusion/physical-expr/src/lib.rs +++ b/datafusion/physical-expr/src/lib.rs @@ -34,6 +34,7 @@ pub mod binary_map { pub mod async_scalar_function; pub mod equivalence; pub mod expressions; +pub mod filter_stats; pub mod higher_order_function; pub mod intervals; mod partitioning; From ab74ffa5dbea9eb65af61bb6be8768c8b86dcc5d Mon Sep 17 00:00:00 2001 From: Adrian Garcia Badaracco <1755071+adriangb@users.noreply.github.com> Date: Thu, 24 Sep 2026 15:02:40 -0500 Subject: [PATCH 09/42] feat: add OptionalFilterGate to pause optional filters that cost more than they save Add a runtime gate that pauses optional filters (filters that are not needed for correctness, such as hash join and TopK dynamic filters) when they cost more than they save. The gate is a per-stream state machine (Evaluate / Paused with exponential backoff) that restarts evaluation when the filter changes. The gate finds the dynamic filters one time with `DynamicFilterTracking::classify` and then polls their subscriptions, so a check does not walk the filter tree. Gates do not share state. At the end of each window of evaluated batches the gate pauses the filter if the window removed no rows, or if its evaluation time is larger than the work that the removed rows save: `(rows_in - rows_out) * saving_ns_per_row`. The saving for each row is the configured minimum plus an optional value that the consumer measures and updates (`MeasuredRowSaving`). The cost rule has a margin (pause above 1.1x the saving, resume below 0.9x) so that a filter does not switch on and off when cost and saving are almost equal. Add the `datafusion.execution.optional_filter_min_saving_ns_per_row` option (default 20). No operator uses the gate yet, so behavior does not change. Co-Authored-By: Claude Opus 5.5 --- datafusion/common/src/config.rs | 11 + .../expressions/dynamic_filters/tracker.rs | 2 +- datafusion/physical-expr/src/lib.rs | 1 + .../physical-expr/src/optional_filter_gate.rs | 1005 +++++++++++++++++ .../test_files/information_schema.slt | 2 + docs/source/user-guide/configs.md | 1 + 6 files changed, 1021 insertions(+), 1 deletion(-) create mode 100644 datafusion/physical-expr/src/optional_filter_gate.rs diff --git a/datafusion/common/src/config.rs b/datafusion/common/src/config.rs index 913845a4219e4..24d52eb080fad 100644 --- a/datafusion/common/src/config.rs +++ b/datafusion/common/src/config.rs @@ -1184,6 +1184,17 @@ config_namespace! { /// /// Disabled by default, set to a number greater than 0 for enabling it. pub hash_join_buffering_capacity: usize, default = 0 + + /// The assumed work, in nanoseconds, that each row removed by an + /// optional filter saves downstream. Optional filters are filters that + /// are not needed for correctness, such as the dynamic filters that + /// hash joins and TopK push down into scans. When an operator evaluates + /// optional filters adaptively, it pauses an optional filter whose + /// evaluation costs more than the work that it saves. Consumers that can + /// measure the saving (the Parquet scan) add their measured decode cost. + /// The default is about the cost of a hash table probe for one row. The + /// best value depends on the hardware. + pub optional_filter_min_saving_ns_per_row: f64, default = 20.0 } } diff --git a/datafusion/physical-expr/src/expressions/dynamic_filters/tracker.rs b/datafusion/physical-expr/src/expressions/dynamic_filters/tracker.rs index fd4c18b07e2cd..6f0a830d07a85 100644 --- a/datafusion/physical-expr/src/expressions/dynamic_filters/tracker.rs +++ b/datafusion/physical-expr/src/expressions/dynamic_filters/tracker.rs @@ -157,7 +157,7 @@ impl DynamicFilterTracker { } /// `true` once every watched filter has completed and been dropped. - fn is_exhausted(&self) -> bool { + pub(crate) fn is_exhausted(&self) -> bool { self.subscriptions.is_empty() } } diff --git a/datafusion/physical-expr/src/lib.rs b/datafusion/physical-expr/src/lib.rs index 75d9d7ea70c2b..fba82a96b0c46 100644 --- a/datafusion/physical-expr/src/lib.rs +++ b/datafusion/physical-expr/src/lib.rs @@ -37,6 +37,7 @@ pub mod expressions; pub mod filter_stats; pub mod higher_order_function; pub mod intervals; +pub mod optional_filter_gate; mod partitioning; mod physical_expr; pub mod planner; diff --git a/datafusion/physical-expr/src/optional_filter_gate.rs b/datafusion/physical-expr/src/optional_filter_gate.rs new file mode 100644 index 0000000000000..8275430baca00 --- /dev/null +++ b/datafusion/physical-expr/src/optional_filter_gate.rs @@ -0,0 +1,1005 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! A runtime gate that pauses optional filters that cost more than they +//! save. +//! +//! An *optional filter* is a filter that is not needed for correctness, for +//! example a dynamic filter that a hash join or a TopK pushes down into a +//! scan. An operator can skip such a filter and still produce correct +//! results. When the filter removes few rows, or when it is expensive (for +//! example a hash table lookup with many columns), the cost to evaluate it +//! can be larger than the benefit. +//! +//! [`OptionalFilterGate`] decides, batch by batch, if a stream evaluates the +//! filter or skips it. Each stream has its own gate, and gates do not share +//! state. +//! +//! # State machine +//! +//! ```text +//! keep (see "Decision") +//! (reset backoff, new window) +//! +-------+ +//! | | +//! v | +//! +-------------+-+ pause +---------------------+ +//! start ------>| Evaluate |-------------------------------->| Paused | +//! | (window of | (pause for `backoff` batches, | (skip `remaining` | +//! | sample_batches| then double `backoff`) | batches) | +//! | batches) |<--------------------------------| | +//! +---------------+ pause ends: probe with a +---------------------+ +//! fresh window +//! ``` +//! +//! * In `Evaluate`, the gate collects the rows in, the rows out and the +//! evaluation time of `sample_batches` batches (a *window*). Then it +//! decides (see below). To pause, it goes to `Paused` for `backoff` batches +//! and doubles `backoff` (up to `max_pause_batches`). To keep the filter, +//! it stays in `Evaluate`, sets `backoff` to `initial_pause_batches` and +//! starts a new window. +//! * In `Paused`, the gate skips batches. It does not change counters or +//! `backoff` for skipped batches. When the pause ends, the gate evaluates +//! a new window (a *probe*). +//! * Before each batch the gate checks if the filter changed (for example a +//! dynamic filter got new bounds). If so, the gate goes to `Evaluate` with +//! an empty window and sets `backoff` to `initial_pause_batches`. +//! +//! # Decision +//! +//! At the end of each window, the gate pauses the filter if one of these +//! rules is true: +//! +//! 1. The filter removed no rows in the window. +//! 2. The evaluation time of the window (`cost_ns`) is larger than the work +//! that the removed rows save (`saving_ns`): +//! +//! ```text +//! saving_ns = (rows_in - rows_out) * saving_ns_per_row +//! saving_ns_per_row = min_saving_ns_per_row + measured saving +//! ``` +//! +//! `min_saving_ns_per_row` comes from the configuration. It is the work +//! that a removed row saves after the filter, for example a hash table +//! probe in a join. The *measured saving* is optional: a consumer that +//! can measure more work that a removed row saves (the Parquet scan +//! measures the decode time of the columns that the filter does not read) +//! gives it in a shared [`MeasuredRowSaving`] and updates it at any time. +//! +//! To prevent a filter from switching on and off when the cost and the +//! saving are almost equal, this rule has a margin: a running filter is +//! paused only if `cost_ns > saving_ns * 1.1`, and a probe after a pause +//! turns the filter on again only if `cost_ns < saving_ns * 0.9`. +//! +//! Thus a filter that removes most rows but is expensive is paused, and a +//! cheap filter stays on also when it removes only some of the rows. +//! +//! The gate measures time with a [`Clock`]. Tests use a [`ManualClock`], so +//! that the decisions are deterministic. +//! +//! [`ManualClock`]: crate::filter_stats::ManualClock +//! +//! # Change detection +//! +//! The gate walks the filter one time, when it is created, with +//! [`DynamicFilterTracking::classify`]. The walk subscribes to each +//! [`DynamicFilterPhysicalExpr`] in the filter that is not complete. Before +//! each batch, the gate polls these subscriptions with +//! [`DynamicFilterTracker::changed`]. When nothing changed, this is one +//! atomic load for each subscription. The tracker drops a subscription when +//! its filter is complete. A filter without dynamic filters, or with only +//! complete dynamic filters, is never polled and never resets the gate. +//! +//! [`DynamicFilterPhysicalExpr`]: crate::expressions::DynamicFilterPhysicalExpr +//! [`DynamicFilterTracker::changed`]: crate::expressions::DynamicFilterTracker::changed + +use std::sync::Arc; +use std::sync::atomic::{AtomicU64, Ordering}; +use std::time::Duration; + +use datafusion_common::config::ExecutionOptions; +use datafusion_physical_expr_common::physical_expr::PhysicalExpr; + +use crate::expressions::DynamicFilterTracking; +use crate::filter_stats::{Clock, FilterCost, SystemClock, duration_nanos}; + +/// A running filter is paused by the cost rule only if its cost is larger +/// than this multiple of its saving. See the [module documentation](self). +const PAUSE_COST_MARGIN: f64 = 1.1; + +/// A probe turns a paused filter on again (by the cost rule) only if its +/// cost is smaller than this multiple of its saving. +const RESUME_COST_MARGIN: f64 = 0.9; + +/// Configuration of an [`OptionalFilterGate`]. +#[derive(Debug, Clone, Copy, PartialEq)] +pub struct OptionalFilterGateConfig { + /// Number of evaluated batches in one window. The gate makes a decision + /// at the end of each window. Values smaller than 1 are used as 1. + pub sample_batches: usize, + /// Number of batches to skip at the first pause, and after the filter + /// was selective again. Values smaller than 1 are used as 1. + pub initial_pause_batches: usize, + /// Maximum number of batches to skip in one pause. Values smaller than + /// `initial_pause_batches` are used as `initial_pause_batches`. + pub max_pause_batches: usize, + /// Work, in nanoseconds, that each row removed by the filter saves after + /// the filter, at the least. The gate adds the saving that the consumer + /// measures (see [`MeasuredRowSaving`]). The gate pauses a filter whose + /// evaluation time is larger than the saving of the rows that it + /// removes. Negative values are used as 0. + pub min_saving_ns_per_row: f64, +} + +impl Default for OptionalFilterGateConfig { + fn default() -> Self { + Self { + sample_batches: 2, + initial_pause_batches: 4, + max_pause_batches: 32, + min_saving_ns_per_row: 20.0, + } + } +} + +impl From<&ExecutionOptions> for OptionalFilterGateConfig { + /// Uses `optional_filter_min_saving_ns_per_row` from `options`, and the + /// default values for the other fields. + fn from(options: &ExecutionOptions) -> Self { + Self { + min_saving_ns_per_row: options.optional_filter_min_saving_ns_per_row, + ..Default::default() + } + } +} + +impl OptionalFilterGateConfig { + /// Returns a copy with the documented minimum values applied. + fn normalized(self) -> Self { + let sample_batches = self.sample_batches.max(1); + let initial_pause_batches = self.initial_pause_batches.max(1); + let max_pause_batches = self.max_pause_batches.max(initial_pause_batches); + Self { + sample_batches, + initial_pause_batches, + max_pause_batches, + min_saving_ns_per_row: self.min_saving_ns_per_row.max(0.0), + } + } +} + +/// The decision of an [`OptionalFilterGate`] for one batch. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum GateDecision { + /// Evaluate the filter on this batch, then call + /// [`OptionalFilterGate::record`] with the row counts and the time. + Evaluate, + /// Do not evaluate the filter on this batch. Let all rows pass. + Skip, +} + +/// Work, in nanoseconds, that each row removed by an optional filter saves, +/// as measured by the consumer of the filter. The gate adds it to +/// [`OptionalFilterGateConfig::min_saving_ns_per_row`]. +/// +/// The consumer creates one value, gives a clone of the [`Arc`] to the gate +/// with [`OptionalFilterGate::with_measured_saving`], and updates it at any +/// time with [`Self::set_ns_per_row`]. The gate reads it at each decision. +/// For example, the Parquet scan sets it to the time to decode the columns +/// that the filter does not read, for each row. +/// +/// The value is an `f64` in an [`AtomicU64`], thus reads and updates are +/// cheap and lock-free. +#[derive(Debug, Default)] +pub struct MeasuredRowSaving { + /// The bits of the `f64` value. + ns_per_row_bits: AtomicU64, +} + +impl MeasuredRowSaving { + /// Creates a value of 0 ns. + pub fn new() -> Self { + Self::default() + } + + /// Sets the measured saving for each removed row, in nanoseconds. + /// Values that are negative or not finite are used as 0. + pub fn set_ns_per_row(&self, ns_per_row: f64) { + let ns_per_row = if ns_per_row.is_finite() { + ns_per_row.max(0.0) + } else { + 0.0 + }; + self.ns_per_row_bits + .store(ns_per_row.to_bits(), Ordering::Relaxed); + } + + /// The measured saving for each removed row, in nanoseconds. + pub fn ns_per_row(&self) -> f64 { + f64::from_bits(self.ns_per_row_bits.load(Ordering::Relaxed)) + } +} + +/// The state of an [`OptionalFilterGate`]. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum GateState { + /// Evaluate the filter and collect counts and time for the current + /// window. + Evaluate { + window: FilterCost, + batches_in_window: usize, + }, + /// Skip the filter for `remaining_batches` more batches. + Paused { remaining_batches: usize }, +} + +impl GateState { + const fn new_window() -> Self { + Self::Evaluate { + window: FilterCost { + rows_in: 0, + rows_out: 0, + nanos: 0, + }, + batches_in_window: 0, + } + } +} + +/// Decides, batch by batch, if one stream evaluates an optional filter. +/// +/// See the [module documentation](self) for the state machine. A gate is +/// for one stream only. Do not share it between streams. +/// +/// Call [`Self::begin_batch`] before each batch. If it returns +/// [`GateDecision::Evaluate`], evaluate [`Self::filter`] on the batch and +/// then call [`Self::record`] with the row counts and the evaluation time. +/// Measure the time with [`Self::clock`], so that tests can replace it. +#[derive(Debug)] +pub struct OptionalFilterGate { + filter: Arc, + /// The dynamic filters in `filter` that can still change. + tracking: DynamicFilterTracking, + config: OptionalFilterGateConfig, + /// The clock that consumers use to measure the evaluation time. + clock: Arc, + /// The saving that the consumer measures, added to + /// `config.min_saving_ns_per_row`. + measured_saving: Option>, + state: GateState, + /// True while the current window is a probe after a pause. The cost + /// rule then uses [`RESUME_COST_MARGIN`]. + probing: bool, + /// Length of the next pause, in batches. + backoff: usize, + /// True after `begin_batch` returned `Evaluate` and before `record`. + awaiting_record: bool, + pauses: usize, +} + +impl OptionalFilterGate { + /// Creates a gate for `filter`, a boolean expression. The gate starts to + /// evaluate the filter. + /// + /// The gate uses a [`SystemClock`] and no measured saving. See + /// [`Self::with_clock`] and [`Self::with_measured_saving`]. + /// + /// This walks `filter` one time to find its dynamic filters. + pub fn new(filter: Arc, config: OptionalFilterGateConfig) -> Self { + let config = config.normalized(); + let tracking = DynamicFilterTracking::classify(&filter); + Self { + filter, + tracking, + config, + clock: SystemClock::shared(), + measured_saving: None, + state: GateState::new_window(), + probing: false, + backoff: config.initial_pause_batches, + awaiting_record: false, + pauses: 0, + } + } + + /// Uses `clock` as the clock of this gate, see [`Self::clock`]. + pub fn with_clock(mut self, clock: Arc) -> Self { + self.clock = clock; + self + } + + /// Adds `saving` to [`OptionalFilterGateConfig::min_saving_ns_per_row`] + /// at each decision. See [`MeasuredRowSaving`]. + pub fn with_measured_saving(mut self, saving: Arc) -> Self { + self.measured_saving = Some(saving); + self + } + + /// The filter of this gate. + pub fn filter(&self) -> &Arc { + &self.filter + } + + /// The clock to measure the evaluation time that is given to + /// [`Self::record`]. + pub fn clock(&self) -> &Arc { + &self.clock + } + + /// The work, in nanoseconds, that the gate assumes each removed row + /// saves now: the configured minimum plus the measured saving. + pub fn saving_ns_per_row(&self) -> f64 { + let measured = self + .measured_saving + .as_ref() + .map_or(0.0, |saving| saving.ns_per_row()); + self.config.min_saving_ns_per_row + measured + } + + /// Call before each batch. Returns if the caller must evaluate the + /// filter on the batch or skip it. + /// + /// If the result is [`GateDecision::Evaluate`], call [`Self::record`] + /// after the evaluation. + pub fn begin_batch(&mut self) -> GateDecision { + // Only a filter with dynamic filters that are not complete can + // change. When nothing changed, this is one atomic load for each + // such dynamic filter. + if let Some(tracker) = self.tracking.watcher() + && tracker.changed() + { + self.state = GateState::new_window(); + self.probing = false; + self.backoff = self.config.initial_pause_batches; + } + + match &mut self.state { + GateState::Paused { remaining_batches } => { + // Only count down. Counters and backoff change only at + // decision points. + *remaining_batches = remaining_batches.saturating_sub(1); + if *remaining_batches == 0 { + // The next batch is a probe. + self.state = GateState::new_window(); + self.probing = true; + } + self.awaiting_record = false; + GateDecision::Skip + } + GateState::Evaluate { .. } => { + self.awaiting_record = true; + GateDecision::Evaluate + } + } + } + + /// Records the result of an evaluation that [`Self::begin_batch`] + /// requested. `rows_in` is the number of rows evaluated, `rows_out` + /// the number of rows that passed (a null result does not pass) and + /// `elapsed` the evaluation time. + /// + /// Calls without a matching `begin_batch` that returned + /// [`GateDecision::Evaluate`] are ignored. + pub fn record(&mut self, rows_in: usize, rows_out: usize, elapsed: Duration) { + if !std::mem::take(&mut self.awaiting_record) { + return; + } + let GateState::Evaluate { + window, + batches_in_window, + } = &mut self.state + else { + return; + }; + window.add(rows_in as u64, rows_out as u64, duration_nanos(elapsed)); + *batches_in_window += 1; + if *batches_in_window >= self.config.sample_batches { + let window = *window; + self.decide(window); + } + } + + /// Number of times the gate paused the filter. + pub fn pauses(&self) -> usize { + self.pauses + } + + /// True if the gate skips the next batch, unless the filter changes + /// before it. + pub fn is_paused(&self) -> bool { + matches!(self.state, GateState::Paused { .. }) + } + + /// Makes a decision at the end of a window. + fn decide(&mut self, window: FilterCost) { + if window.rows_in == 0 { + // No rows, thus no information. Start a new window. + self.state = GateState::new_window(); + return; + } + if self.should_pause(&window) { + self.start_pause(self.backoff); + } else { + self.state = GateState::new_window(); + self.probing = false; + self.backoff = self.config.initial_pause_batches; + } + } + + /// The decision rules, see the [module documentation](self). + fn should_pause(&self, window: &FilterCost) -> bool { + let rows_removed = window.rows_removed(); + if rows_removed == 0 { + return true; + } + // The filter costs more than it saves. + let cost_ns = window.nanos as f64; + let saving_ns = rows_removed as f64 * self.saving_ns_per_row(); + if self.probing { + // Turn the filter on again only if it is clearly worth its cost. + cost_ns >= saving_ns * RESUME_COST_MARGIN + } else { + cost_ns > saving_ns * PAUSE_COST_MARGIN + } + } + + /// Goes to `Paused` for `pause_batches` batches and doubles the backoff. + fn start_pause(&mut self, pause_batches: usize) { + let pause_batches = pause_batches.max(1); + self.probing = false; + self.state = GateState::Paused { + remaining_batches: pause_batches, + }; + self.backoff = pause_batches + .saturating_mul(2) + .min(self.config.max_pause_batches); + self.pauses += 1; + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::expressions::{BinaryExpr, Column, DynamicFilterPhysicalExpr, col, lit}; + use crate::filter_stats::ManualClock; + use arrow::array::{Array, BooleanArray, Int32Array, RecordBatch}; + use arrow::datatypes::{DataType, Field, Schema}; + use datafusion_common::cast::as_boolean_array; + use datafusion_expr::Operator; + + const ROWS: usize = 1000; + + fn static_filter() -> Arc { + lit(true) + } + + /// A gate with the default configuration and a clock that does not + /// move: the evaluation time is 0, thus only a window that removes no + /// rows pauses the filter. + fn gate_with(filter: Arc) -> OptionalFilterGate { + OptionalFilterGate::new(filter, OptionalFilterGateConfig::default()) + .with_clock(Arc::new(ManualClock::new())) + } + + fn new_gate() -> OptionalFilterGate { + gate_with(static_filter()) + } + + /// Feeds one batch with the given pass ratio and an evaluation time of + /// 0. Returns the decision. + fn feed(gate: &mut OptionalFilterGate, pass_ratio: f64) -> GateDecision { + feed_timed(gate, pass_ratio, 0.0) + } + + /// Feeds one batch with the given pass ratio and an evaluation time of + /// `ns_per_row` for each row. Returns the decision. + fn feed_timed( + gate: &mut OptionalFilterGate, + pass_ratio: f64, + ns_per_row: f64, + ) -> GateDecision { + let decision = gate.begin_batch(); + if decision == GateDecision::Evaluate { + let elapsed = Duration::from_nanos((ROWS as f64 * ns_per_row) as u64); + gate.record(ROWS, (ROWS as f64 * pass_ratio) as usize, elapsed); + } + decision + } + + /// Feeds `n` batches and returns how many the gate evaluated. + fn feed_n(gate: &mut OptionalFilterGate, n: usize, pass_ratio: f64) -> usize { + (0..n) + .filter(|_| feed(gate, pass_ratio) == GateDecision::Evaluate) + .count() + } + + /// Feeds batches until the gate evaluates one. Returns the number of + /// skipped batches. + fn skip_until_probe(gate: &mut OptionalFilterGate, pass_ratio: f64) -> usize { + let mut skipped = 0; + while feed(gate, pass_ratio) == GateDecision::Skip { + skipped += 1; + assert!(skipped < 10_000, "gate never probes"); + } + skipped + } + + /// Evaluates the filter of `gate` on `batch` like a consumer does, or + /// skips it. Returns `None` if the gate skipped the filter. + fn evaluate( + gate: &mut OptionalFilterGate, + batch: &RecordBatch, + ) -> Option { + if gate.begin_batch() == GateDecision::Skip { + return None; + } + let num_rows = batch.num_rows(); + let start = gate.clock().now_nanos(); + let result = gate + .filter() + .evaluate(batch) + .unwrap() + .into_array(num_rows) + .unwrap(); + let result = as_boolean_array(&result).unwrap().clone(); + let elapsed = gate.clock().now_nanos().saturating_sub(start); + // `true_count` does not count nulls. + gate.record(num_rows, result.true_count(), Duration::from_nanos(elapsed)); + Some(result) + } + + #[test] + fn pauses_filter_that_removes_no_rows() { + let mut gate = new_gate(); + assert_eq!(feed(&mut gate, 1.0), GateDecision::Evaluate); + assert!(!gate.is_paused()); + assert_eq!(feed(&mut gate, 1.0), GateDecision::Evaluate); + assert!(gate.is_paused()); + assert_eq!(gate.pauses(), 1); + + // Pauses for `initial_pause_batches` batches. + for _ in 0..4 { + assert_eq!(feed(&mut gate, 1.0), GateDecision::Skip); + } + // Then it probes. + assert_eq!(feed(&mut gate, 1.0), GateDecision::Evaluate); + } + + #[test] + fn keeps_free_filter_that_removes_rows() { + let mut gate = new_gate(); + assert_eq!(feed_n(&mut gate, 100, 0.5), 100); + assert_eq!(gate.pauses(), 0); + // A free filter that removes only 1% of the rows also stays on. + assert_eq!(feed_n(&mut gate, 10, 0.99), 10); + assert_eq!(gate.pauses(), 0); + } + + #[test] + fn backoff_doubles_up_to_cap_and_resets() { + let mut gate = new_gate(); + let mut observed = vec![]; + assert_eq!(feed_n(&mut gate, 2, 1.0), 2); + for _ in 0..6 { + observed.push(skip_until_probe(&mut gate, 1.0)); + // `skip_until_probe` evaluated the first batch of the probe + // window. One more batch closes the window. + assert_eq!(feed(&mut gate, 1.0), GateDecision::Evaluate); + } + assert_eq!(observed, vec![4, 8, 16, 32, 32, 32]); + assert_eq!(gate.pauses(), 7); + + // A selective probe resets the backoff. + assert!(gate.is_paused()); + let skipped = skip_until_probe(&mut gate, 0.1); + assert_eq!(skipped, 32); + assert_eq!(feed(&mut gate, 0.1), GateDecision::Evaluate); + assert!(!gate.is_paused()); + assert_eq!(gate.backoff, 4); + + // The next pause is short again. + assert_eq!(feed_n(&mut gate, 2, 1.0), 2); + assert_eq!(skip_until_probe(&mut gate, 1.0), 4); + } + + fn dynamic_filter() -> (Arc, Arc) { + let column: Arc = Arc::new(Column::new("a", 0)); + let dynamic = Arc::new(DynamicFilterPhysicalExpr::new(vec![column], lit(true))); + let filter = Arc::clone(&dynamic) as Arc; + (dynamic, filter) + } + + fn a_gt(value: i32) -> Arc { + Arc::new(BinaryExpr::new( + Arc::new(Column::new("a", 0)), + Operator::Gt, + lit(value), + )) + } + + #[test] + fn generation_change_restarts_evaluation() { + let (dynamic, filter) = dynamic_filter(); + let mut gate = gate_with(filter); + + // Pause two times: the backoff grows to 16. + assert_eq!(feed_n(&mut gate, 2, 1.0), 2); + skip_until_probe(&mut gate, 1.0); + feed(&mut gate, 1.0); + assert!(gate.is_paused()); + assert_eq!(gate.backoff, 16); + + // A new generation of the filter ends the pause at once. + dynamic.update(a_gt(10)).unwrap(); + assert_eq!(feed(&mut gate, 1.0), GateDecision::Evaluate); + assert!(!gate.is_paused()); + assert_eq!(gate.backoff, 4); + + // The new window has an empty history: one more batch decides. + assert_eq!(feed(&mut gate, 1.0), GateDecision::Evaluate); + assert!(gate.is_paused()); + assert_eq!(skip_until_probe(&mut gate, 1.0), 4); + } + + #[test] + fn static_filter_is_never_watched() { + // A filter without dynamic filters, and a filter whose dynamic + // filters are all complete, can not change: the gate never polls + // them and never resets. + let (dynamic, complete) = dynamic_filter(); + dynamic.mark_complete(); + for (filter, expected_complete) in [(static_filter(), false), (complete, true)] { + let mut gate = gate_with(filter); + match (&gate.tracking, expected_complete) { + (DynamicFilterTracking::Static, false) + | (DynamicFilterTracking::AllComplete, true) => {} + (other, _) => panic!("unexpected tracking {other:?}"), + } + assert!(gate.tracking.watcher().is_none()); + + assert_eq!(feed_n(&mut gate, 2, 1.0), 2); + assert!(gate.is_paused()); + assert_eq!(skip_until_probe(&mut gate, 1.0), 4); + assert_eq!(feed(&mut gate, 1.0), GateDecision::Evaluate); + assert_eq!(skip_until_probe(&mut gate, 1.0), 8); + assert_eq!(gate.pauses(), 2); + } + } + + #[test] + fn completed_filter_stops_being_watched() { + let (dynamic, filter) = dynamic_filter(); + let mut gate = gate_with(filter); + assert!(gate.tracking.watcher().is_some()); + assert_eq!(feed_n(&mut gate, 2, 1.0), 2); + assert!(gate.is_paused()); + + // The final update restarts the evaluation one time. + dynamic.update(a_gt(10)).unwrap(); + dynamic.mark_complete(); + assert_eq!(feed(&mut gate, 1.0), GateDecision::Evaluate); + assert!(!gate.is_paused()); + // The tracker dropped the subscription of the complete filter. + assert!(gate.tracking.watcher().unwrap().is_exhausted()); + + // No more resets: the gate pauses and backs off as usual. + assert_eq!(feed(&mut gate, 1.0), GateDecision::Evaluate); + assert!(gate.is_paused()); + assert_eq!(skip_until_probe(&mut gate, 1.0), 4); + assert_eq!(feed(&mut gate, 1.0), GateDecision::Evaluate); + assert_eq!(skip_until_probe(&mut gate, 1.0), 8); + } + + #[test] + fn topk_like_filter_stays_on() { + // The filter gets tighter over time and changes often, like a TopK + // dynamic filter. At the start it removes almost no rows. + let (dynamic, filter) = dynamic_filter(); + let mut gate = gate_with(filter); + for i in 0..100 { + if i % 2 == 0 { + dynamic.update(a_gt(i)).unwrap(); + } + let pass_ratio = 1.0 - (i as f64 / 100.0); + assert_eq!( + feed(&mut gate, pass_ratio), + GateDecision::Evaluate, + "batch {i}" + ); + } + assert_eq!(gate.pauses(), 0); + } + + #[test] + fn skewed_input_probe_re_enables_filter() { + let mut gate = new_gate(); + // Data that the filter does not remove first: the gate pauses with + // growing backoff. + assert_eq!(feed_n(&mut gate, 2, 1.0), 2); + for _ in 0..3 { + skip_until_probe(&mut gate, 1.0); + feed(&mut gate, 1.0); + } + assert!(gate.is_paused()); + + // Then the data becomes selective. A probe finds it. + let skipped = skip_until_probe(&mut gate, 0.05); + assert!(skipped <= 32); + assert_eq!(feed(&mut gate, 0.05), GateDecision::Evaluate); + assert!(!gate.is_paused()); + // The filter stays on for the rest of the input. + assert_eq!(feed_n(&mut gate, 50, 0.05), 50); + } + + /// Regression test: a paused gate must not change its backoff or its + /// counters for each skipped batch. Only decisions change them. + #[test] + fn paused_gate_does_not_grow_backoff_while_skipping() { + let mut gate = new_gate(); + assert_eq!(feed_n(&mut gate, 2, 1.0), 2); + assert!(gate.is_paused()); + let backoff = gate.backoff; + let pauses = gate.pauses(); + + for _ in 0..3 { + assert_eq!(gate.begin_batch(), GateDecision::Skip); + // A stray `record` during a pause is ignored. + gate.record(ROWS, 0, Duration::from_secs(1)); + assert_eq!(gate.backoff, backoff); + assert_eq!(gate.pauses(), pauses); + } + assert_eq!(gate.begin_batch(), GateDecision::Skip); + assert_eq!(gate.backoff, backoff); + // The pause is over: the next batch is evaluated. + assert_eq!(gate.begin_batch(), GateDecision::Evaluate); + } + + #[test] + fn empty_window_makes_no_decision() { + let mut gate = new_gate(); + for _ in 0..10 { + assert_eq!(gate.begin_batch(), GateDecision::Evaluate); + gate.record(0, 0, Duration::from_micros(1)); + } + assert_eq!(gate.pauses(), 0); + } + + fn batch(values: Vec>) -> RecordBatch { + let schema = Arc::new(Schema::new(vec![Field::new("a", DataType::Int32, true)])); + RecordBatch::try_new(schema, vec![Arc::new(Int32Array::from(values))]).unwrap() + } + + fn col_gt(value: i32) -> Arc { + let schema = Schema::new(vec![Field::new("a", DataType::Int32, true)]); + Arc::new(BinaryExpr::new( + col("a", &schema).unwrap(), + Operator::Gt, + lit(value), + )) + } + + #[test] + fn evaluate_selective_filter() { + let mut gate = gate_with(col_gt(8)); + let input = batch((0..10).map(Some).collect()); + for _ in 0..10 { + let result = evaluate(&mut gate, &input).expect("evaluated"); + assert_eq!(result.true_count(), 1); + } + assert_eq!(gate.pauses(), 0); + } + + #[test] + fn evaluate_filter_that_removes_no_rows_skips() { + let mut gate = gate_with(col_gt(0)); + let input = batch((1..=10).map(Some).collect()); + assert!(evaluate(&mut gate, &input).is_some()); + assert!(evaluate(&mut gate, &input).is_some()); + for _ in 0..4 { + assert!(evaluate(&mut gate, &input).is_none()); + } + assert!(evaluate(&mut gate, &input).is_some()); + } + + #[test] + fn evaluate_counts_null_as_not_passing() { + let mut gate = gate_with(col_gt(0)); + // 9 of 10 rows are null: the filter removes them. + let mut values = vec![None; 9]; + values.push(Some(5)); + let input = batch(values); + for _ in 0..10 { + let result = evaluate(&mut gate, &input).expect("evaluated"); + assert_eq!(result.null_count(), 9); + } + assert_eq!(gate.pauses(), 0); + } + + #[test] + fn config_from_execution_options() { + let mut options = ExecutionOptions::default(); + assert_eq!( + OptionalFilterGateConfig::from(&options), + OptionalFilterGateConfig::default() + ); + options.optional_filter_min_saving_ns_per_row = 7.5; + let config = OptionalFilterGateConfig::from(&options); + assert_eq!(config.min_saving_ns_per_row, 7.5); + assert_eq!(config.sample_batches, 2); + } + + // The tests below use the default `min_saving_ns_per_row` of 20 ns. For + // a window of 1000-row batches with pass ratio `p` and cost `c` ns for + // each row, the gate compares `c` with `(1 - p) * 20` ns for each row. + + /// Like the dynamic filter of a hash join with a multi-column key: it + /// removes 93% of the rows, but costs 70 ns for each row. Each removed + /// row saves only 20 ns, thus 18.6 ns for each evaluated row. + #[test] + fn selective_but_expensive_filter_pauses() { + let mut gate = new_gate(); + assert_eq!(feed_timed(&mut gate, 0.07, 70.0), GateDecision::Evaluate); + assert_eq!(feed_timed(&mut gate, 0.07, 70.0), GateDecision::Evaluate); + assert!(gate.is_paused()); + assert_eq!(gate.pauses(), 1); + + // The probes find the same cost: the pauses get longer. + assert_eq!(skip_until_probe(&mut gate, 0.07), 4); + assert_eq!(feed_timed(&mut gate, 0.07, 70.0), GateDecision::Evaluate); + // `skip_until_probe` evaluated the first batch with no time. The + // window costs 70 µs and saves 1860 * 20 ns = 37.2 µs: still too + // much. + assert!(gate.is_paused()); + assert_eq!(skip_until_probe(&mut gate, 0.07), 8); + } + + /// A bound check that costs 2 ns for each row and removes 30% of the + /// rows: it saves 6 ns for each row, thus it stays on. + #[test] + fn cheap_weakly_selective_filter_stays_on() { + let mut gate = new_gate(); + for _ in 0..100 { + assert_eq!(feed_timed(&mut gate, 0.7, 2.0), GateDecision::Evaluate); + } + assert_eq!(gate.pauses(), 0); + } + + /// Cheap bounds that cost 1 ns for each row and remove 12% of the rows: + /// they save 2.4 ns for each row, thus they stay on. There is no + /// separate rule for the fraction of rows that pass. + #[test] + fn cheap_bounds_that_remove_few_rows_stay_on() { + let mut gate = new_gate(); + for _ in 0..100 { + assert_eq!(feed_timed(&mut gate, 0.88, 1.0), GateDecision::Evaluate); + } + assert_eq!(gate.pauses(), 0); + } + + /// Like the dynamic filter of a TopK: a cheap comparison that removes + /// almost all rows. It stays on. + #[test] + fn cheap_very_selective_filter_stays_on() { + let mut gate = new_gate(); + for _ in 0..100 { + assert_eq!(feed_timed(&mut gate, 0.001, 3.0), GateDecision::Evaluate); + } + assert_eq!(gate.pauses(), 0); + } + + /// The consumer measures a larger saving (for example the Parquet scan + /// measures the decode time of the columns that the filter does not + /// read). The next probe turns the expensive filter on again. + #[test] + fn measured_saving_re_enables_filter_at_next_probe() { + let saving = Arc::new(MeasuredRowSaving::new()); + let mut gate = new_gate().with_measured_saving(Arc::clone(&saving)); + assert_eq!(gate.saving_ns_per_row(), 20.0); + assert_eq!(feed_timed(&mut gate, 0.07, 70.0), GateDecision::Evaluate); + assert_eq!(feed_timed(&mut gate, 0.07, 70.0), GateDecision::Evaluate); + assert!(gate.is_paused()); + + // Each removed row now saves 20 + 80 = 100 ns: 93 ns for each + // evaluated row, more than the cost of 70 ns with the margin. + saving.set_ns_per_row(80.0); + assert_eq!(gate.saving_ns_per_row(), 100.0); + // The pause does not end early. + for _ in 0..4 { + assert_eq!(feed_timed(&mut gate, 0.07, 70.0), GateDecision::Skip); + } + assert_eq!(feed_timed(&mut gate, 0.07, 70.0), GateDecision::Evaluate); + assert_eq!(feed_timed(&mut gate, 0.07, 70.0), GateDecision::Evaluate); + assert!(!gate.is_paused()); + assert_eq!(gate.backoff, 4); + for _ in 0..20 { + assert_eq!(feed_timed(&mut gate, 0.07, 70.0), GateDecision::Evaluate); + } + + // Invalid measured values are used as 0. + saving.set_ns_per_row(f64::NAN); + assert_eq!(gate.saving_ns_per_row(), 20.0); + saving.set_ns_per_row(-5.0); + assert_eq!(gate.saving_ns_per_row(), 20.0); + } + + /// When the cost is near the saving, the gate keeps its state: a running + /// filter stays on and a paused filter stays paused. + #[test] + fn cost_check_has_hysteresis() { + // The filter removes 50% of the rows: the saving is 10 ns for each + // evaluated row. A cost of 10.5 ns is between 0.9 and 1.1 times the + // saving. + let mut gate = new_gate(); + for _ in 0..20 { + assert_eq!(feed_timed(&mut gate, 0.5, 10.5), GateDecision::Evaluate); + } + assert!(!gate.is_paused()); + + // Pause it with a larger cost. + assert_eq!(feed_timed(&mut gate, 0.5, 30.0), GateDecision::Evaluate); + assert_eq!(feed_timed(&mut gate, 0.5, 30.0), GateDecision::Evaluate); + assert!(gate.is_paused()); + // A probe with the cost near the saving does not turn it on. + for _ in 0..4 { + assert_eq!(feed_timed(&mut gate, 0.5, 10.5), GateDecision::Skip); + } + assert_eq!(feed_timed(&mut gate, 0.5, 10.5), GateDecision::Evaluate); + assert_eq!(feed_timed(&mut gate, 0.5, 10.5), GateDecision::Evaluate); + assert!(gate.is_paused()); + assert_eq!(gate.pauses(), 2); + // A probe with a clearly lower cost turns it on. + for _ in 0..8 { + assert_eq!(feed_timed(&mut gate, 0.5, 8.5), GateDecision::Skip); + } + assert_eq!(feed_timed(&mut gate, 0.5, 8.5), GateDecision::Evaluate); + assert_eq!(feed_timed(&mut gate, 0.5, 8.5), GateDecision::Evaluate); + assert!(!gate.is_paused()); + } + + /// A clock that moves by a fixed step each time it is read. + #[derive(Debug)] + struct SteppingClock { + now: AtomicU64, + step: u64, + } + + impl Clock for SteppingClock { + fn now_nanos(&self) -> u64 { + self.now.fetch_add(self.step, Ordering::Relaxed) + } + } + + /// A consumer measures the time with the clock of the gate. + #[test] + fn consumer_measures_time_with_gate_clock() { + // Each evaluation takes 10 µs for 10 rows: 1000 ns for each row. + let clock = Arc::new(SteppingClock { + now: AtomicU64::new(0), + step: 10_000, + }); + let mut gate = + OptionalFilterGate::new(col_gt(8), OptionalFilterGateConfig::default()) + .with_clock(clock); + let input = batch((0..10).map(Some).collect()); + // The filter removes 90% of the rows, but it costs 1000 ns for each + // row, and each removed row saves 20 ns. + assert!(evaluate(&mut gate, &input).is_some()); + assert!(evaluate(&mut gate, &input).is_some()); + assert!(gate.is_paused()); + assert!(evaluate(&mut gate, &input).is_none()); + } +} diff --git a/datafusion/sqllogictest/test_files/information_schema.slt b/datafusion/sqllogictest/test_files/information_schema.slt index 113f63c71257c..1ecc7a97d310e 100644 --- a/datafusion/sqllogictest/test_files/information_schema.slt +++ b/datafusion/sqllogictest/test_files/information_schema.slt @@ -231,6 +231,7 @@ datafusion.execution.max_spill_file_size_bytes 134217728 datafusion.execution.meta_fetch_concurrency 32 datafusion.execution.minimum_parallel_output_files 4 datafusion.execution.objectstore_writer_buffer_size 10485760 +datafusion.execution.optional_filter_min_saving_ns_per_row 20 datafusion.execution.parquet.allow_single_file_parallelism true datafusion.execution.parquet.binary_as_string false datafusion.execution.parquet.bloom_filter_fpp NULL @@ -392,6 +393,7 @@ datafusion.execution.max_spill_file_size_bytes 134217728 Maximum size in bytes f datafusion.execution.meta_fetch_concurrency 32 Number of files to read in parallel when inferring schema and statistics datafusion.execution.minimum_parallel_output_files 4 Guarantees a minimum level of output files running in parallel. RecordBatches will be distributed in round robin fashion to each parallel writer. Each writer is closed and a new file opened once soft_max_rows_per_output_file is reached. datafusion.execution.objectstore_writer_buffer_size 10485760 Size (bytes) of data buffer DataFusion uses when writing output files. This affects the size of the data chunks that are uploaded to remote object stores (e.g. AWS S3). If very large (>= 100 GiB) output files are being written, it may be necessary to increase this size to avoid errors from the remote end point. +datafusion.execution.optional_filter_min_saving_ns_per_row 20 The assumed work, in nanoseconds, that each row removed by an optional filter saves downstream. Optional filters are filters that are not needed for correctness, such as the dynamic filters that hash joins and TopK push down into scans. When an operator evaluates optional filters adaptively, it pauses an optional filter whose evaluation costs more than the work that it saves. Consumers that can measure the saving (the Parquet scan) add their measured decode cost. The default is about the cost of a hash table probe for one row. The best value depends on the hardware. datafusion.execution.parquet.allow_single_file_parallelism true (writing) Controls whether DataFusion will attempt to speed up writing parquet files by serializing them in parallel. Each column in each row group in each output file are serialized in parallel leveraging a maximum possible core count of n_files\*n_row_groups\*n_columns. datafusion.execution.parquet.binary_as_string false (reading) If true, parquet reader will read columns of `Binary/LargeBinary` with `Utf8`, and `BinaryView` with `Utf8View`. Parquet files generated by some legacy writers do not correctly set the UTF8 flag for strings, causing string columns to be loaded as BLOB instead. The parquet reader has special optimizations for `Utf8` validation, so reading such columns as strings is significantly faster than reading them as binary and then casting to string. datafusion.execution.parquet.bloom_filter_fpp NULL (writing) Sets bloom filter false positive probability. If NULL, uses default parquet writer setting diff --git a/docs/source/user-guide/configs.md b/docs/source/user-guide/configs.md index ef4ea00129f29..178e3d109466c 100644 --- a/docs/source/user-guide/configs.md +++ b/docs/source/user-guide/configs.md @@ -145,6 +145,7 @@ The following configuration settings are available: | datafusion.execution.objectstore_writer_buffer_size | 10485760 | Size (bytes) of data buffer DataFusion uses when writing output files. This affects the size of the data chunks that are uploaded to remote object stores (e.g. AWS S3). If very large (>= 100 GiB) output files are being written, it may be necessary to increase this size to avoid errors from the remote end point. | | datafusion.execution.enable_ansi_mode | false | Whether to enable ANSI SQL mode. The flag is experimental and relevant only for DataFusion Spark built-in functions When `enable_ansi_mode` is set to `true`, the query engine follows ANSI SQL semantics for expressions, casting, and error handling. This means: - **Strict type coercion rules:** implicit casts between incompatible types are disallowed. - **Standard SQL arithmetic behavior:** operations such as division by zero, numeric overflow, or invalid casts raise runtime errors rather than returning `NULL` or adjusted values. - **Consistent ANSI behavior** for string concatenation, comparisons, and `NULL` handling. When `enable_ansi_mode` is `false` (the default), the engine uses a more permissive, non-ANSI mode designed for user convenience and backward compatibility. In this mode: - Implicit casts between types are allowed (e.g., string to integer when possible). - Arithmetic operations are more lenient — for example, `abs()` on the minimum representable integer value returns the input value instead of raising overflow. - Division by zero or invalid casts may return `NULL` instead of failing. # Default `false` — ANSI SQL mode is disabled by default. | | datafusion.execution.hash_join_buffering_capacity | 0 | How many bytes to buffer in the probe side of hash joins while the build side is concurrently being built. Without this, hash joins will wait until the full materialization of the build side before polling the probe side. This is useful in scenarios where the query is not completely CPU bounded, allowing to do some early work concurrently and reducing the latency of the query. Note that when hash join buffering is enabled, the probe side will start eagerly polling data, not giving time for the producer side of dynamic filters to produce any meaningful predicate. Queries with dynamic filters might see performance degradation. Disabled by default, set to a number greater than 0 for enabling it. | +| datafusion.execution.optional_filter_min_saving_ns_per_row | 20 | The assumed work, in nanoseconds, that each row removed by an optional filter saves downstream. Optional filters are filters that are not needed for correctness, such as the dynamic filters that hash joins and TopK push down into scans. When an operator evaluates optional filters adaptively, it pauses an optional filter whose evaluation costs more than the work that it saves. Consumers that can measure the saving (the Parquet scan) add their measured decode cost. The default is about the cost of a hash table probe for one row. The best value depends on the hardware. | | datafusion.optimizer.enable_distinct_aggregation_soft_limit | true | When set to true, the optimizer will push a limit operation into grouped aggregations which have no aggregate expressions, as a soft limit, emitting groups once the limit is reached, before all rows in the group are read. | | datafusion.optimizer.enable_round_robin_repartition | true | When set to true, the physical plan optimizer will try to add round robin repartitioning to increase parallelism to leverage more CPU cores | | datafusion.optimizer.enable_topk_aggregation | true | When set to true, the optimizer will attempt to perform limit operations during aggregations, if possible | From 6ae18bcd6c93919a889475a98361945757b812bc Mon Sep 17 00:00:00 2001 From: Adrian Garcia Badaracco <1755071+adriangb@users.noreply.github.com> Date: Fri, 25 Sep 2026 11:41:04 -0500 Subject: [PATCH 10/42] feat(gate): add a measured overhead for each evaluated row to the cost A consumer can now give the gate a fixed cost for each evaluated row in addition to the evaluation time, with `MeasuredRowSaving::set_overhead_ns_per_row`. The gate adds it to the cost of each window. The Parquet scan uses it for the fixed cost of a row filter stage, which is larger than the evaluation time of a cheap predicate. PR: #25674 Co-Authored-By: Claude Opus 5.5 --- .../physical-expr/src/optional_filter_gate.rs | 115 ++++++++++++++---- 1 file changed, 92 insertions(+), 23 deletions(-) diff --git a/datafusion/physical-expr/src/optional_filter_gate.rs b/datafusion/physical-expr/src/optional_filter_gate.rs index 8275430baca00..7bcc8e4f7edd3 100644 --- a/datafusion/physical-expr/src/optional_filter_gate.rs +++ b/datafusion/physical-expr/src/optional_filter_gate.rs @@ -65,20 +65,24 @@ //! rules is true: //! //! 1. The filter removed no rows in the window. -//! 2. The evaluation time of the window (`cost_ns`) is larger than the work -//! that the removed rows save (`saving_ns`): +//! 2. The cost of the window (`cost_ns`) is larger than the work that the +//! removed rows save (`saving_ns`): //! //! ```text +//! cost_ns = evaluation time + rows_in * measured overhead //! saving_ns = (rows_in - rows_out) * saving_ns_per_row //! saving_ns_per_row = min_saving_ns_per_row + measured saving //! ``` //! //! `min_saving_ns_per_row` comes from the configuration. It is the work //! that a removed row saves after the filter, for example a hash table -//! probe in a join. The *measured saving* is optional: a consumer that -//! can measure more work that a removed row saves (the Parquet scan -//! measures the decode time of the columns that the filter does not read) -//! gives it in a shared [`MeasuredRowSaving`] and updates it at any time. +//! probe in a join. The *measured saving* and the *measured overhead* are +//! optional: a consumer that can measure more work that a removed row +//! saves (the Parquet scan measures the decode time of the columns that +//! the filter does not read), or a fixed cost for each evaluated row in +//! addition to the evaluation time (the Parquet scan has a cost for each +//! row filter stage), gives them in a shared [`MeasuredRowSaving`] and +//! updates them at any time. //! //! To prevent a filter from switching on and off when the cost and the //! saving are almost equal, this rule has a margin: a running filter is @@ -192,22 +196,42 @@ pub enum GateDecision { Skip, } -/// Work, in nanoseconds, that each row removed by an optional filter saves, -/// as measured by the consumer of the filter. The gate adds it to -/// [`OptionalFilterGateConfig::min_saving_ns_per_row`]. +/// The terms of the cost rule that the consumer of an optional filter +/// measures: +/// +/// * The *saving*: work, in nanoseconds, that each row removed by the filter +/// saves. The gate adds it to +/// [`OptionalFilterGateConfig::min_saving_ns_per_row`]. For example, the +/// Parquet scan sets it to the time to decode the columns that the filter +/// does not read, for each removed row that the decoder can skip. +/// * The *overhead*: work, in nanoseconds, that the consumer does for each +/// evaluated row because it evaluates the filter, in addition to the +/// evaluation time. The gate adds it to the cost of each window. For +/// example, the Parquet scan sets it to the fixed cost of a row filter +/// stage when the filter is a row filter predicate. /// /// The consumer creates one value, gives a clone of the [`Arc`] to the gate /// with [`OptionalFilterGate::with_measured_saving`], and updates it at any -/// time with [`Self::set_ns_per_row`]. The gate reads it at each decision. -/// For example, the Parquet scan sets it to the time to decode the columns -/// that the filter does not read, for each row. +/// time with [`Self::set_ns_per_row`] and [`Self::set_overhead_ns_per_row`]. +/// The gate reads it at each decision. /// -/// The value is an `f64` in an [`AtomicU64`], thus reads and updates are +/// Each value is an `f64` in an [`AtomicU64`], thus reads and updates are /// cheap and lock-free. #[derive(Debug, Default)] pub struct MeasuredRowSaving { - /// The bits of the `f64` value. + /// The bits of the `f64` saving for each removed row. ns_per_row_bits: AtomicU64, + /// The bits of the `f64` overhead for each evaluated row. + overhead_ns_per_row_bits: AtomicU64, +} + +/// `value` if it is finite and not negative, else 0. +fn non_negative(value: f64) -> f64 { + if value.is_finite() { + value.max(0.0) + } else { + 0.0 + } } impl MeasuredRowSaving { @@ -219,19 +243,26 @@ impl MeasuredRowSaving { /// Sets the measured saving for each removed row, in nanoseconds. /// Values that are negative or not finite are used as 0. pub fn set_ns_per_row(&self, ns_per_row: f64) { - let ns_per_row = if ns_per_row.is_finite() { - ns_per_row.max(0.0) - } else { - 0.0 - }; self.ns_per_row_bits - .store(ns_per_row.to_bits(), Ordering::Relaxed); + .store(non_negative(ns_per_row).to_bits(), Ordering::Relaxed); } /// The measured saving for each removed row, in nanoseconds. pub fn ns_per_row(&self) -> f64 { f64::from_bits(self.ns_per_row_bits.load(Ordering::Relaxed)) } + + /// Sets the measured overhead for each evaluated row, in nanoseconds. + /// Values that are negative or not finite are used as 0. + pub fn set_overhead_ns_per_row(&self, ns_per_row: f64) { + self.overhead_ns_per_row_bits + .store(non_negative(ns_per_row).to_bits(), Ordering::Relaxed); + } + + /// The measured overhead for each evaluated row, in nanoseconds. + pub fn overhead_ns_per_row(&self) -> f64 { + f64::from_bits(self.overhead_ns_per_row_bits.load(Ordering::Relaxed)) + } } /// The state of an [`OptionalFilterGate`]. @@ -322,8 +353,9 @@ impl OptionalFilterGate { self } - /// Adds `saving` to [`OptionalFilterGateConfig::min_saving_ns_per_row`] - /// at each decision. See [`MeasuredRowSaving`]. + /// Adds the saving of `saving` to + /// [`OptionalFilterGateConfig::min_saving_ns_per_row`] and its overhead + /// to the cost at each decision. See [`MeasuredRowSaving`]. pub fn with_measured_saving(mut self, saving: Arc) -> Self { self.measured_saving = Some(saving); self @@ -350,6 +382,14 @@ impl OptionalFilterGate { self.config.min_saving_ns_per_row + measured } + /// The work, in nanoseconds for each evaluated row, that the gate adds + /// to the evaluation time now: the measured overhead. + pub fn overhead_ns_per_row(&self) -> f64 { + self.measured_saving + .as_ref() + .map_or(0.0, |saving| saving.overhead_ns_per_row()) + } + /// Call before each batch. Returns if the caller must evaluate the /// filter on the batch or skip it. /// @@ -447,7 +487,8 @@ impl OptionalFilterGate { return true; } // The filter costs more than it saves. - let cost_ns = window.nanos as f64; + let cost_ns = + window.nanos as f64 + window.rows_in as f64 * self.overhead_ns_per_row(); let saving_ns = rows_removed as f64 * self.saving_ns_per_row(); if self.probing { // Turn the filter on again only if it is clearly worth its cost. @@ -936,6 +977,34 @@ mod tests { assert_eq!(gate.saving_ns_per_row(), 20.0); } + /// The consumer measures an overhead for each evaluated row (for example + /// the fixed cost of a Parquet row filter stage). A filter that is worth + /// its evaluation time alone is paused when the overhead is added. + #[test] + fn measured_overhead_adds_to_cost() { + let saving = Arc::new(MeasuredRowSaving::new()); + let mut gate = new_gate().with_measured_saving(Arc::clone(&saving)); + // Removes 50% of the rows: saves 10 ns for each evaluated row. The + // evaluation costs 5 ns for each row. + for _ in 0..10 { + assert_eq!(feed_timed(&mut gate, 0.5, 5.0), GateDecision::Evaluate); + } + assert!(!gate.is_paused()); + + // With 8 ns of overhead, the cost is 13 ns for each row. + saving.set_overhead_ns_per_row(8.0); + assert_eq!(gate.overhead_ns_per_row(), 8.0); + assert_eq!(feed_timed(&mut gate, 0.5, 5.0), GateDecision::Evaluate); + assert_eq!(feed_timed(&mut gate, 0.5, 5.0), GateDecision::Evaluate); + assert!(gate.is_paused()); + + // Invalid measured values are used as 0. + saving.set_overhead_ns_per_row(f64::INFINITY); + assert_eq!(gate.overhead_ns_per_row(), 0.0); + saving.set_overhead_ns_per_row(-1.0); + assert_eq!(gate.overhead_ns_per_row(), 0.0); + } + /// When the cost is near the saving, the gate keeps its state: a running /// filter stays on and a paused filter stays paused. #[test] From ac838c45be4f1d720fef7dd5c3b2308f995b722f Mon Sep 17 00:00:00 2001 From: Adrian Garcia Badaracco <1755071+adriangb@users.noreply.github.com> Date: Fri, 25 Sep 2026 12:10:21 -0500 Subject: [PATCH 11/42] feat(gate): share pauses between the gates of one plan site Each gate paid for its own first window and its own probes. A scan opens its files at the same time, thus a filter that removes nothing cost one window in each file (TPC-H Q9: five join filters that remove no rows and a CASE routing filter that costs 83 ns for each row, in each of 12 files). `SharedGateVerdict` holds the last pause (or end of a pause) of the gates of one plan site in one atomic word. A gate without evidence of its own (before its first decision, after a filter change, after a pause) uses a pause that another gate published after the last verdict that it saw: a new gate starts paused, and a gate in its first window or in a probe window stops and pauses. Thus usually only one gate probes after a pause. A gate that keeps the filter does not use the pauses of other gates (skewed data). A filter change clears the shared pause, because it was measured on the old filter. PR: #25674 Co-Authored-By: Claude Opus 5.5 --- .../physical-expr/src/optional_filter_gate.rs | 309 +++++++++++++++++- 1 file changed, 306 insertions(+), 3 deletions(-) diff --git a/datafusion/physical-expr/src/optional_filter_gate.rs b/datafusion/physical-expr/src/optional_filter_gate.rs index 7bcc8e4f7edd3..0d91bf9b7888f 100644 --- a/datafusion/physical-expr/src/optional_filter_gate.rs +++ b/datafusion/physical-expr/src/optional_filter_gate.rs @@ -26,8 +26,9 @@ //! can be larger than the benefit. //! //! [`OptionalFilterGate`] decides, batch by batch, if a stream evaluates the -//! filter or skips it. Each stream has its own gate, and gates do not share -//! state. +//! filter or skips it. Each stream has its own gate. The gates of one plan +//! site (for example all files and partitions of one scan) can share their +//! pauses, see "Shared verdict". //! //! # State machine //! @@ -110,6 +111,32 @@ //! //! [`DynamicFilterPhysicalExpr`]: crate::expressions::DynamicFilterPhysicalExpr //! [`DynamicFilterTracker::changed`]: crate::expressions::DynamicFilterTracker::changed +//! +//! # Shared verdict +//! +//! Each gate pays for at least one window before its first decision. A scan +//! opens many files at the same time, and a filter that removes no rows +//! (for example a hash join filter on a column that has only matching +//! values) then costs one window in each file. Also each gate probes on its +//! own after each pause. +//! +//! Thus the gates of one plan site can share a [`SharedGateVerdict`] (see +//! [`OptionalFilterGate::with_shared_verdict`]). Each gate publishes its +//! pauses and the end of its pauses there. A gate that has no evidence of +//! its own that the filter is worth its cost (before its first decision, +//! after a change of the filter, and after a decision to pause) uses a +//! pause that another gate published after the last shared verdict that it +//! saw: +//! +//! * A new gate starts paused if the shared verdict is a pause. +//! * A gate in its first window, or in a probe window after a pause, stops +//! the window and pauses. Thus after a pause of all gates, usually only +//! the first gate at the end of its pause probes the filter. +//! +//! A gate that keeps the filter does not use the pauses of other gates: +//! with skewed data the filter can be worth its cost for some files only. +//! A gate that sees a change of the filter clears a shared pause, because +//! it was measured on the old filter, and starts again without evidence. use std::sync::Arc; use std::sync::atomic::{AtomicU64, Ordering}; @@ -265,6 +292,89 @@ impl MeasuredRowSaving { } } +/// The last pause (or end of a pause) that the gates of one plan site +/// published, see "Shared verdict" in the [module documentation](self). +/// +/// Create one value for each plan site (for example each optional filter of +/// one scan), and give a clone of the [`Arc`] to each gate of the site with +/// [`OptionalFilterGate::with_shared_verdict`]. +/// +/// The value is one [`AtomicU64`], thus it is lock-free. Gates write it only +/// at decisions that pause the filter or end a pause, and read it before a +/// batch only while they have no evidence of their own. +#[derive(Debug, Default)] +pub struct SharedGateVerdict { + /// A [`Verdict`] and its sequence number, see [`Verdict::pack`]. + word: AtomicU64, +} + +/// One published verdict of a [`SharedGateVerdict`]. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum Verdict { + /// The filter is not paused (or nothing was published yet). + Keep, + /// A gate paused the filter for this number of batches. + Pause(usize), +} + +impl Verdict { + /// Largest pause length that fits in the packed word. + const MAX_BATCHES: usize = u32::MAX as usize; + + /// Packs the verdict with the sequence number `seq`: the sequence + /// number in the low 32 bits, and the pause length (0 for + /// [`Verdict::Keep`]) in the high 32 bits. + fn pack(self, seq: u32) -> u64 { + let batches = match self { + Self::Keep => 0, + Self::Pause(batches) => batches.clamp(1, Self::MAX_BATCHES) as u64, + }; + (batches << 32) | u64::from(seq) + } + + /// The verdict and the sequence number in `word`. + fn unpack(word: u64) -> (Self, u32) { + let verdict = match (word >> 32) as usize { + 0 => Self::Keep, + batches => Self::Pause(batches), + }; + (verdict, word as u32) + } +} + +impl SharedGateVerdict { + /// Creates a shared verdict without any published pause. + pub fn new() -> Self { + Self::default() + } + + /// True if the last published verdict is a pause. + pub fn is_paused(&self) -> bool { + matches!(self.load().0, Verdict::Pause(_)) + } + + /// The current verdict and its sequence number. + fn load(&self) -> (Verdict, u32) { + Verdict::unpack(self.word.load(Ordering::Acquire)) + } + + /// Publishes `verdict` if `replace` returns true for the current + /// verdict. Returns the sequence number of the published verdict. + fn publish_if( + &self, + verdict: Verdict, + replace: impl Fn(Verdict) -> bool, + ) -> Option { + self.word + .fetch_update(Ordering::AcqRel, Ordering::Acquire, |word| { + let (current, seq) = Verdict::unpack(word); + replace(current).then(|| verdict.pack(seq.wrapping_add(1))) + }) + .ok() + .map(|previous| (previous as u32).wrapping_add(1)) + } +} + /// The state of an [`OptionalFilterGate`]. #[derive(Debug, Clone, Copy, PartialEq, Eq)] enum GateState { @@ -320,6 +430,17 @@ pub struct OptionalFilterGate { /// True after `begin_batch` returned `Evaluate` and before `record`. awaiting_record: bool, pauses: usize, + /// The verdict that this gate shares with the other gates of its plan + /// site, if any. + shared: Option>, + /// Sequence number of the last shared verdict that this gate published + /// or saw. + shared_seq: u32, + /// True while this gate has no evidence of its own that the filter is + /// worth its cost: before its first decision, after a change of the + /// filter, and after a decision that paused the filter. Then it uses + /// the shared pauses of the other gates. + uses_shared_pauses: bool, } impl OptionalFilterGate { @@ -344,9 +465,25 @@ impl OptionalFilterGate { backoff: config.initial_pause_batches, awaiting_record: false, pauses: 0, + shared: None, + shared_seq: 0, + uses_shared_pauses: true, } } + /// Shares the pauses of this gate with the other gates of the same plan + /// site, see "Shared verdict" in the [module documentation](self). If + /// the shared verdict is a pause, the gate starts paused. + pub fn with_shared_verdict(mut self, shared: Arc) -> Self { + let (verdict, seq) = shared.load(); + self.shared_seq = seq; + if let Verdict::Pause(batches) = verdict { + self.start_pause(batches); + } + self.shared = Some(shared); + self + } + /// Uses `clock` as the clock of this gate, see [`Self::clock`]. pub fn with_clock(mut self, clock: Arc) -> Self { self.clock = clock; @@ -405,6 +542,17 @@ impl OptionalFilterGate { self.state = GateState::new_window(); self.probing = false; self.backoff = self.config.initial_pause_batches; + self.uses_shared_pauses = true; + if let Some(shared) = &self.shared { + // A shared pause was measured on the old filter. + let paused = |verdict| matches!(verdict, Verdict::Pause(_)); + if let Some(seq) = shared.publish_if(Verdict::Keep, paused) { + self.shared_seq = seq; + } + } + } + if !self.is_paused() { + self.use_shared_pause(); } match &mut self.state { @@ -472,11 +620,48 @@ impl OptionalFilterGate { return; } if self.should_pause(&window) { - self.start_pause(self.backoff); + let batches = self.backoff; + self.start_pause(batches); + self.uses_shared_pauses = true; + self.publish(Verdict::Pause(batches)); } else { self.state = GateState::new_window(); self.probing = false; self.backoff = self.config.initial_pause_batches; + if std::mem::take(&mut self.uses_shared_pauses) { + // The first decision, or the end of a pause. + self.publish(Verdict::Keep); + } + } + } + + /// Before a batch that the gate would evaluate: if this gate uses the + /// shared pauses and another gate published a pause after the last + /// shared verdict that this gate saw, pauses the filter for the same + /// length. The current window is dropped. + fn use_shared_pause(&mut self) { + if !self.uses_shared_pauses { + return; + } + let Some(shared) = &self.shared else { + return; + }; + let (verdict, seq) = shared.load(); + if seq == self.shared_seq { + return; + } + self.shared_seq = seq; + if let Verdict::Pause(batches) = verdict { + self.start_pause(batches); + } + } + + /// Publishes `verdict` to the shared verdict, if any. + fn publish(&mut self, verdict: Verdict) { + if let Some(shared) = &self.shared { + self.shared_seq = shared + .publish_if(verdict, |_| true) + .expect("an unconditional update always succeeds"); } } @@ -1039,6 +1224,124 @@ mod tests { assert!(!gate.is_paused()); } + fn shared_gate( + filter: Arc, + shared: &Arc, + ) -> OptionalFilterGate { + gate_with(filter).with_shared_verdict(Arc::clone(shared)) + } + + #[test] + fn verdict_pack_round_trip() { + for verdict in [Verdict::Keep, Verdict::Pause(1), Verdict::Pause(32)] { + assert_eq!(Verdict::unpack(verdict.pack(7)), (verdict, 7)); + assert_eq!(Verdict::unpack(verdict.pack(u32::MAX)), (verdict, u32::MAX)); + } + // Pause lengths are at least 1 and at most `MAX_BATCHES`. + assert_eq!( + Verdict::unpack(Verdict::Pause(0).pack(1)).0, + Verdict::Pause(1) + ); + assert_eq!( + Verdict::unpack(Verdict::Pause(usize::MAX).pack(1)).0, + Verdict::Pause(Verdict::MAX_BATCHES) + ); + } + + /// A new gate starts from the shared pause of the other gates. + #[test] + fn new_gate_starts_with_shared_pause() { + let shared = Arc::new(SharedGateVerdict::new()); + let mut first = shared_gate(static_filter(), &shared); + assert!(!first.is_paused()); + assert_eq!(feed_n(&mut first, 2, 1.0), 2); + assert!(first.is_paused()); + assert!(shared.is_paused()); + + // A new gate starts paused for the same length, then probes. + let mut second = shared_gate(static_filter(), &shared); + assert!(second.is_paused()); + assert_eq!(second.pauses(), 1); + assert_eq!(skip_until_probe(&mut second, 0.5), 4); + // The probe keeps the filter: new gates evaluate again. + assert_eq!(feed(&mut second, 0.5), GateDecision::Evaluate); + assert!(!second.is_paused()); + assert!(!shared.is_paused()); + assert!(!shared_gate(static_filter(), &shared).is_paused()); + } + + /// Gates that start at the same time: a gate in its first window uses + /// the pause that another gate published, instead of the rest of its + /// own window, and only one gate probes after the pause. + #[test] + fn first_window_and_probe_use_shared_pause() { + let shared = Arc::new(SharedGateVerdict::new()); + let mut gates: Vec<_> = (0..4) + .map(|_| shared_gate(static_filter(), &shared)) + .collect(); + // All gates evaluate their first batch. + for gate in &mut gates { + assert_eq!(feed(gate, 1.0), GateDecision::Evaluate); + } + // The first gate completes its window and pauses. + assert_eq!(feed(&mut gates[0], 1.0), GateDecision::Evaluate); + assert!(gates[0].is_paused()); + // The others do not evaluate the second batch of their window. + for gate in &mut gates[1..] { + assert_eq!(feed(gate, 1.0), GateDecision::Skip); + assert!(gate.is_paused()); + } + + // Only the first gate probes at the end of its pause. The others + // use its verdict: a pause of 8 batches. + assert_eq!(skip_until_probe(&mut gates[0], 1.0), 4); + assert_eq!(feed(&mut gates[0], 1.0), GateDecision::Evaluate); + assert!(gates[0].is_paused()); + for gate in &mut gates[1..] { + let skipped = (0..20) + .take_while(|_| feed(gate, 1.0) == GateDecision::Skip) + .count(); + // 3 more batches of the first pause, then 8. + assert_eq!(skipped, 3 + 8); + assert_eq!(gate.pauses(), 2); + } + } + + /// A gate that keeps the filter does not use the pauses of other gates + /// (skewed data). + #[test] + fn running_gate_ignores_shared_pause() { + let shared = Arc::new(SharedGateVerdict::new()); + let mut selective = shared_gate(static_filter(), &shared); + assert_eq!(feed_n(&mut selective, 2, 0.1), 2); + assert!(!selective.is_paused()); + + let mut other = shared_gate(static_filter(), &shared); + assert_eq!(feed_n(&mut other, 2, 1.0), 2); + assert!(other.is_paused()); + assert!(shared.is_paused()); + + assert_eq!(feed_n(&mut selective, 20, 0.1), 20); + assert_eq!(selective.pauses(), 0); + } + + /// A change of the filter clears the shared pause: it was measured on + /// the old filter. + #[test] + fn change_clears_shared_pause() { + let (dynamic, filter) = dynamic_filter(); + let shared = Arc::new(SharedGateVerdict::new()); + let mut gate = shared_gate(Arc::clone(&filter), &shared); + assert_eq!(feed_n(&mut gate, 2, 1.0), 2); + assert!(shared.is_paused()); + + dynamic.update(a_gt(1)).unwrap(); + // The gate sees the change at its next batch. + assert_eq!(feed(&mut gate, 0.5), GateDecision::Evaluate); + assert!(!shared.is_paused()); + assert!(!shared_gate(filter, &shared).is_paused()); + } + /// A clock that moves by a fixed step each time it is read. #[derive(Debug)] struct SteppingClock { From 36c73a0d6a1d92809204ac04cc332b718cf79f46 Mon Sep 17 00:00:00 2001 From: Adrian Garcia Badaracco <1755071+adriangb@users.noreply.github.com> Date: Fri, 25 Sep 2026 12:44:36 -0500 Subject: [PATCH 12/42] feat(gate): decide only on windows of at least MIN_OBSERVED_ROWS rows A gate decided after `sample_batches` batches, whatever their size. After a selective row filter a batch can have 2 to 7 rows, and the fixed cost of each call then looks like 600 to 8000 ns for each row: ClickBench Q23 paused the TopK filter on `EventTime` (0.4 ns for each row on full batches) on such windows, and the shared verdict spread these pauses to the other files. All decisions (pause, keep, probe) now need a window of at least `sample_batches` batches and `MIN_OBSERVED_ROWS` rows. The constant moves to `filter_stats`, so that the gate and the Parquet filter placement use the same sample size. All published shared pauses come from such windows. PR: #25674 Co-Authored-By: Claude Opus 5.5 --- datafusion/physical-expr/src/filter_stats.rs | 9 +++ .../physical-expr/src/optional_filter_gate.rs | 64 ++++++++++++++----- 2 files changed, 57 insertions(+), 16 deletions(-) diff --git a/datafusion/physical-expr/src/filter_stats.rs b/datafusion/physical-expr/src/filter_stats.rs index f5f5c4f03f8b1..0ea219f4166e0 100644 --- a/datafusion/physical-expr/src/filter_stats.rs +++ b/datafusion/physical-expr/src/filter_stats.rs @@ -106,6 +106,15 @@ impl Clock for ManualClock { } } +/// Minimum number of evaluated rows before an adaptive decision uses the +/// measurements of a filter. It is one batch of the default +/// `datafusion.execution.batch_size`. With fewer rows, the evaluation time +/// is dominated by the fixed cost of each call (for example 2 to 7 rows of +/// a batch after a selective row filter took 600 to 8000 ns for each row in +/// ClickBench Q23, against 0.4 ns for each row on full batches), and the +/// fraction of removed rows is not reliable. +pub const MIN_OBSERVED_ROWS: u64 = 8192; + /// Returns the nanoseconds in `elapsed`, saturated to `u64::MAX`. pub fn duration_nanos(elapsed: Duration) -> u64 { u64::try_from(elapsed.as_nanos()).unwrap_or(u64::MAX) diff --git a/datafusion/physical-expr/src/optional_filter_gate.rs b/datafusion/physical-expr/src/optional_filter_gate.rs index 0d91bf9b7888f..5da48bcdb80af 100644 --- a/datafusion/physical-expr/src/optional_filter_gate.rs +++ b/datafusion/physical-expr/src/optional_filter_gate.rs @@ -48,7 +48,8 @@ //! ``` //! //! * In `Evaluate`, the gate collects the rows in, the rows out and the -//! evaluation time of `sample_batches` batches (a *window*). Then it +//! evaluation time of at least `sample_batches` batches and at least +//! [`MIN_OBSERVED_ROWS`] rows (a *window*). Then it //! decides (see below). To pause, it goes to `Paused` for `backoff` batches //! and doubles `backoff` (up to `max_pause_batches`). To keep the filter, //! it stays in `Evaluate`, sets `backoff` to `initial_pause_batches` and @@ -146,7 +147,9 @@ use datafusion_common::config::ExecutionOptions; use datafusion_physical_expr_common::physical_expr::PhysicalExpr; use crate::expressions::DynamicFilterTracking; -use crate::filter_stats::{Clock, FilterCost, SystemClock, duration_nanos}; +use crate::filter_stats::{ + Clock, FilterCost, MIN_OBSERVED_ROWS, SystemClock, duration_nanos, +}; /// A running filter is paused by the cost rule only if its cost is larger /// than this multiple of its saving. See the [module documentation](self). @@ -595,7 +598,12 @@ impl OptionalFilterGate { }; window.add(rows_in as u64, rows_out as u64, duration_nanos(elapsed)); *batches_in_window += 1; - if *batches_in_window >= self.config.sample_batches { + // A window has at least `sample_batches` batches and + // `MIN_OBSERVED_ROWS` rows: fewer rows are not enough evidence for a + // decision. + if *batches_in_window >= self.config.sample_batches + && window.rows_in >= MIN_OBSERVED_ROWS + { let window = *window; self.decide(window); } @@ -707,7 +715,14 @@ mod tests { use datafusion_common::cast::as_boolean_array; use datafusion_expr::Operator; - const ROWS: usize = 1000; + /// Rows of each test batch: two batches make a window of + /// `MIN_OBSERVED_ROWS` rows. + const ROWS: usize = MIN_OBSERVED_ROWS as usize / 2; + + /// `ROWS` values of `value(i)`. + fn values(value: impl Fn(i32) -> Option) -> Vec> { + (0..ROWS as i32).map(value).collect() + } fn static_filter() -> Arc { lit(true) @@ -994,6 +1009,25 @@ mod tests { assert_eq!(gate.begin_batch(), GateDecision::Evaluate); } + /// Small batches (for example after a selective row filter) have a + /// large fixed cost for each row. The window grows until it has + /// `MIN_OBSERVED_ROWS` rows: small batches do not decide early. + #[test] + fn window_needs_min_observed_rows() { + let mut gate = new_gate(); + let small = 3; + let batches = MIN_OBSERVED_ROWS as usize / small; + for _ in 0..batches { + assert_eq!(gate.begin_batch(), GateDecision::Evaluate); + // 8000 ns for each row, and no row removed. + gate.record(small, small, Duration::from_nanos(8000 * small as u64)); + } + assert!(!gate.is_paused()); + assert_eq!(gate.begin_batch(), GateDecision::Evaluate); + gate.record(small, small, Duration::from_nanos(8000 * small as u64)); + assert!(gate.is_paused()); + } + #[test] fn empty_window_makes_no_decision() { let mut gate = new_gate(); @@ -1021,10 +1055,10 @@ mod tests { #[test] fn evaluate_selective_filter() { let mut gate = gate_with(col_gt(8)); - let input = batch((0..10).map(Some).collect()); + let input = batch(values(|i| Some(i % 10))); for _ in 0..10 { let result = evaluate(&mut gate, &input).expect("evaluated"); - assert_eq!(result.true_count(), 1); + assert_eq!(result.true_count(), ROWS / 10); } assert_eq!(gate.pauses(), 0); } @@ -1032,7 +1066,7 @@ mod tests { #[test] fn evaluate_filter_that_removes_no_rows_skips() { let mut gate = gate_with(col_gt(0)); - let input = batch((1..=10).map(Some).collect()); + let input = batch(values(|i| Some(i % 10 + 1))); assert!(evaluate(&mut gate, &input).is_some()); assert!(evaluate(&mut gate, &input).is_some()); for _ in 0..4 { @@ -1045,12 +1079,10 @@ mod tests { fn evaluate_counts_null_as_not_passing() { let mut gate = gate_with(col_gt(0)); // 9 of 10 rows are null: the filter removes them. - let mut values = vec![None; 9]; - values.push(Some(5)); - let input = batch(values); + let input = batch(values(|i| (i % 10 == 0).then_some(5))); for _ in 0..10 { let result = evaluate(&mut gate, &input).expect("evaluated"); - assert_eq!(result.null_count(), 9); + assert_eq!(result.null_count(), ROWS - ROWS.div_ceil(10)); } assert_eq!(gate.pauses(), 0); } @@ -1087,8 +1119,8 @@ mod tests { assert_eq!(skip_until_probe(&mut gate, 0.07), 4); assert_eq!(feed_timed(&mut gate, 0.07, 70.0), GateDecision::Evaluate); // `skip_until_probe` evaluated the first batch with no time. The - // window costs 70 µs and saves 1860 * 20 ns = 37.2 µs: still too - // much. + // window costs 70 ns for each row of one batch and saves 0.93 * 20 + // ns for each row of two batches: still too much. assert!(gate.is_paused()); assert_eq!(skip_until_probe(&mut gate, 0.07), 8); } @@ -1358,15 +1390,15 @@ mod tests { /// A consumer measures the time with the clock of the gate. #[test] fn consumer_measures_time_with_gate_clock() { - // Each evaluation takes 10 µs for 10 rows: 1000 ns for each row. + // Each evaluation takes 1000 ns for each row. let clock = Arc::new(SteppingClock { now: AtomicU64::new(0), - step: 10_000, + step: 1000 * ROWS as u64, }); let mut gate = OptionalFilterGate::new(col_gt(8), OptionalFilterGateConfig::default()) .with_clock(clock); - let input = batch((0..10).map(Some).collect()); + let input = batch(values(|i| Some(i % 10))); // The filter removes 90% of the rows, but it costs 1000 ns for each // row, and each removed row saves 20 ns. assert!(evaluate(&mut gate, &input).is_some()); From a352cb705a3df7603373bd08b32fe76e9737633e Mon Sep 17 00:00:00 2001 From: Adrian Garcia Badaracco <1755071+adriangb@users.noreply.github.com> Date: Fri, 25 Sep 2026 13:51:27 -0500 Subject: [PATCH 13/42] feat(gate): use the measured work of the filter producer as the saving The gate assumed that each removed row saves `min_saving_ns_per_row` (20 ns) after the filter. For hash join dynamic filters this is the probe work of the join, and it is much smaller in star joins with small dimension tables: 3.5 to 8 ns for each probe row on the TPC-DS SF1 `date_dim` joins (Q65, Q67), 2 ns on Q90, 17 ns on TPC-H Q9. Filters that cost 3 to 7 ns for each row and remove 80% of the rows thus stayed on, and cost more than the join work they saved (8-18% slower than `pruning_only` on the bot). `RemovedRowWork` (in `filter_stats`) is the work that the producer of a filter does for each row that the filter removes, as the producer measures it. Each `DynamicFilterPhysicalExpr` has one, shared by all its derived filters. The gate uses the smallest measured work of the dynamic filters in its filter as the saving of a removed row, and `min_saving_ns_per_row` only until the producer has measured `MIN_OBSERVED_ROWS` rows (a prior). PR: #25674 Co-Authored-By: Claude Opus 5.5 --- .../src/expressions/dynamic_filters/mod.rs | 20 +++++ datafusion/physical-expr/src/filter_stats.rs | 53 ++++++++++++ .../physical-expr/src/optional_filter_gate.rs | 84 ++++++++++++++++--- 3 files changed, 146 insertions(+), 11 deletions(-) diff --git a/datafusion/physical-expr/src/expressions/dynamic_filters/mod.rs b/datafusion/physical-expr/src/expressions/dynamic_filters/mod.rs index 4d384e952e0c0..2cc08779153b6 100644 --- a/datafusion/physical-expr/src/expressions/dynamic_filters/mod.rs +++ b/datafusion/physical-expr/src/expressions/dynamic_filters/mod.rs @@ -21,6 +21,7 @@ use std::{fmt::Display, hash::Hash, sync::Arc}; use tokio::sync::watch; use crate::PhysicalExpr; +use crate::filter_stats::RemovedRowWork; use arrow::datatypes::{DataType, Schema}; #[cfg(feature = "proto")] use datafusion_common::internal_datafusion_err; @@ -94,6 +95,10 @@ pub struct DynamicFilterPhysicalExpr { /// But this can have overhead in production, so it's only included in our tests. data_type: Arc>>, nullable: Arc>>, + /// The work that the producer of the filter does for each row that the + /// filter removes, see [`Self::removed_row_work`]. Shared by all + /// derived filters. + removed_row_work: Arc, } impl std::fmt::Debug for DynamicFilterPhysicalExpr { @@ -223,9 +228,20 @@ impl DynamicFilterPhysicalExpr { state_watch, data_type: Arc::new(RwLock::new(None)), nullable: Arc::new(RwLock::new(None)), + removed_row_work: Arc::default(), } } + /// The work that the producer of this filter does for each row that + /// the filter removes before it (for example the probe of a hash join), + /// as the producer measures it. A consumer that decides if the filter + /// is worth its cost uses it as the saving of a removed row (see + /// [`OptionalFilterGate`](crate::optional_filter_gate::OptionalFilterGate)). + /// The producer records its work in it. + pub fn removed_row_work(&self) -> &Arc { + &self.removed_row_work + } + fn remap_children( children: &[Arc], remapped_children: Option<&Vec>>, @@ -490,6 +506,7 @@ impl DynamicFilterPhysicalExpr { state_watch, data_type: Arc::new(RwLock::new(None)), nullable: Arc::new(RwLock::new(None)), + removed_row_work: Arc::default(), } } } @@ -518,6 +535,7 @@ impl PhysicalExpr for DynamicFilterPhysicalExpr { state_watch: self.state_watch.clone(), data_type: Arc::clone(&self.data_type), nullable: Arc::clone(&self.nullable), + removed_row_work: Arc::clone(&self.removed_row_work), })) } @@ -614,6 +632,7 @@ impl PhysicalExpr for DynamicFilterPhysicalExpr { state_watch: _, // Runtime channel, recreated from inner state by from_parts(). data_type: _, // Cached test invariant, recomputed from the expression. nullable: _, // Cached test invariant, recomputed from the expression. + removed_row_work: _, // Runtime measurement of the producer. } = self; let children = children @@ -837,6 +856,7 @@ impl DynamicFilterPhysicalExpr { state_watch: self.state_watch.clone(), data_type: Arc::clone(&self.data_type), nullable: Arc::clone(&self.nullable), + removed_row_work: Arc::clone(&self.removed_row_work), } } } diff --git a/datafusion/physical-expr/src/filter_stats.rs b/datafusion/physical-expr/src/filter_stats.rs index 0ea219f4166e0..f209a62acd17e 100644 --- a/datafusion/physical-expr/src/filter_stats.rs +++ b/datafusion/physical-expr/src/filter_stats.rs @@ -163,10 +163,63 @@ impl FilterCost { } } +/// The work, in nanoseconds for each row, that an operator does on the rows +/// that a filter removes before them, as measured by that operator. +/// +/// For example, a hash join computes the hashes of the join keys of each +/// probe row and looks them up in its hash table. A row that the dynamic +/// filter of the join removes in the scan does not get this work, thus this +/// is the saving of a removed row. The work that depends on a match (the +/// output of a matched row) is not in it: the filter does not remove +/// matched rows. The producer of a dynamic filter measures it and the +/// consumers of the filter read it (see +/// [`DynamicFilterPhysicalExpr::removed_row_work`]). +/// +/// Lock-free. +/// +/// [`DynamicFilterPhysicalExpr::removed_row_work`]: crate::expressions::DynamicFilterPhysicalExpr::removed_row_work +#[derive(Debug, Default)] +pub struct RemovedRowWork { + rows: AtomicU64, + nanos: AtomicU64, +} + +impl RemovedRowWork { + /// Creates an empty measurement. + pub fn new() -> Self { + Self::default() + } + + /// Adds `rows` rows and `nanos` nanoseconds of work. The two can be + /// recorded separately (for example the rows when a batch arrives and + /// the time of each step). + pub fn record(&self, rows: u64, nanos: u64) { + self.rows.fetch_add(rows, Ordering::Relaxed); + self.nanos.fetch_add(nanos, Ordering::Relaxed); + } + + /// The work for each row, or `None` before [`MIN_OBSERVED_ROWS`] rows. + pub fn ns_per_row(&self) -> Option { + let rows = self.rows.load(Ordering::Relaxed); + (rows >= MIN_OBSERVED_ROWS) + .then(|| self.nanos.load(Ordering::Relaxed) as f64 / rows as f64) + } +} + #[cfg(test)] mod tests { use super::*; + #[test] + fn removed_row_work_needs_min_observed_rows() { + let work = RemovedRowWork::new(); + work.record(MIN_OBSERVED_ROWS - 1, 0); + work.record(0, 3 * (MIN_OBSERVED_ROWS - 1)); + assert_eq!(work.ns_per_row(), None); + work.record(1, 3); + assert_eq!(work.ns_per_row(), Some(3.0)); + } + #[test] fn filter_cost_derived_values() { let empty = FilterCost::default(); diff --git a/datafusion/physical-expr/src/optional_filter_gate.rs b/datafusion/physical-expr/src/optional_filter_gate.rs index 5da48bcdb80af..15787f101a382 100644 --- a/datafusion/physical-expr/src/optional_filter_gate.rs +++ b/datafusion/physical-expr/src/optional_filter_gate.rs @@ -73,12 +73,20 @@ //! ```text //! cost_ns = evaluation time + rows_in * measured overhead //! saving_ns = (rows_in - rows_out) * saving_ns_per_row -//! saving_ns_per_row = min_saving_ns_per_row + measured saving +//! saving_ns_per_row = producer work + measured saving //! ``` //! -//! `min_saving_ns_per_row` comes from the configuration. It is the work -//! that a removed row saves after the filter, for example a hash table -//! probe in a join. The *measured saving* and the *measured overhead* are +//! The *producer work* is the work that a removed row saves after the +//! filter, in the operator that produced the filter, for example the +//! hash and the hash table lookup of a probe row in a hash join. The +//! producer measures it ([`RemovedRowWork`], see +//! [`DynamicFilterPhysicalExpr::removed_row_work`]): a hash join with a +//! small build side does 2 to 8 ns of work for each probe row (TPC-DS +//! SF1 star joins), a join with a large build side much more. Until the +//! producer has measured [`MIN_OBSERVED_ROWS`] rows, the gate uses +//! `min_saving_ns_per_row` from the configuration (a prior). With more +//! than one dynamic filter in the filter, the smallest measured work is +//! used. The *measured saving* and the *measured overhead* are //! optional: a consumer that can measure more work that a removed row //! saves (the Parquet scan measures the decode time of the columns that //! the filter does not read), or a fixed cost for each evaluated row in @@ -146,9 +154,11 @@ use std::time::Duration; use datafusion_common::config::ExecutionOptions; use datafusion_physical_expr_common::physical_expr::PhysicalExpr; -use crate::expressions::DynamicFilterTracking; +use datafusion_common::tree_node::{TreeNode, TreeNodeRecursion}; + +use crate::expressions::{DynamicFilterPhysicalExpr, DynamicFilterTracking}; use crate::filter_stats::{ - Clock, FilterCost, MIN_OBSERVED_ROWS, SystemClock, duration_nanos, + Clock, FilterCost, MIN_OBSERVED_ROWS, RemovedRowWork, SystemClock, duration_nanos, }; /// A running filter is paused by the cost rule only if its cost is larger @@ -172,7 +182,8 @@ pub struct OptionalFilterGateConfig { /// `initial_pause_batches` are used as `initial_pause_batches`. pub max_pause_batches: usize, /// Work, in nanoseconds, that each row removed by the filter saves after - /// the filter, at the least. The gate adds the saving that the consumer + /// the filter, until the producer of the filter has measured it (see + /// [`RemovedRowWork`]). The gate adds the saving that the consumer /// measures (see [`MeasuredRowSaving`]). The gate pauses a filter whose /// evaluation time is larger than the saving of the rows that it /// removes. Negative values are used as 0. @@ -421,9 +432,11 @@ pub struct OptionalFilterGate { config: OptionalFilterGateConfig, /// The clock that consumers use to measure the evaluation time. clock: Arc, - /// The saving that the consumer measures, added to - /// `config.min_saving_ns_per_row`. + /// The saving that the consumer measures, added to the producer work. measured_saving: Option>, + /// The work that the producers of the dynamic filters in `filter` do + /// for each removed row, see [`RemovedRowWork`]. + producer_work: Vec>, state: GateState, /// True while the current window is a probe after a pause. The cost /// rule then uses [`RESUME_COST_MARGIN`]. @@ -457,12 +470,22 @@ impl OptionalFilterGate { pub fn new(filter: Arc, config: OptionalFilterGateConfig) -> Self { let config = config.normalized(); let tracking = DynamicFilterTracking::classify(&filter); + let mut producer_work = vec![]; + filter + .apply(|expr| { + if let Some(dynamic) = expr.downcast_ref::() { + producer_work.push(Arc::clone(dynamic.removed_row_work())); + } + Ok(TreeNodeRecursion::Continue) + }) + .expect("the closure is infallible"); Self { filter, tracking, config, clock: SystemClock::shared(), measured_saving: None, + producer_work, state: GateState::new_window(), probing: false, backoff: config.initial_pause_batches, @@ -513,13 +536,26 @@ impl OptionalFilterGate { } /// The work, in nanoseconds, that the gate assumes each removed row - /// saves now: the configured minimum plus the measured saving. + /// saves now: the work of the producer (measured, or the configured + /// `min_saving_ns_per_row` before the measurement) plus the measured + /// saving of the consumer. pub fn saving_ns_per_row(&self) -> f64 { let measured = self .measured_saving .as_ref() .map_or(0.0, |saving| saving.ns_per_row()); - self.config.min_saving_ns_per_row + measured + self.producer_work_ns_per_row() + measured + } + + /// The work of the producer for each removed row: the smallest measured + /// [`RemovedRowWork`] of the dynamic filters in the filter, or + /// `min_saving_ns_per_row` if none is measured yet. + fn producer_work_ns_per_row(&self) -> f64 { + self.producer_work + .iter() + .filter_map(|work| work.ns_per_row()) + .reduce(f64::min) + .unwrap_or(self.config.min_saving_ns_per_row) } /// The work, in nanoseconds for each evaluated row, that the gate adds @@ -1374,6 +1410,32 @@ mod tests { assert!(!shared_gate(filter, &shared).is_paused()); } + /// The producer of a dynamic filter measures its work for each row that + /// the filter removes: once measured, it replaces the configured + /// `min_saving_ns_per_row`. + #[test] + fn producer_work_replaces_configured_saving() { + let (dynamic, filter) = dynamic_filter(); + let mut gate = gate_with(filter); + assert_eq!(gate.saving_ns_per_row(), 20.0); + // Removes 80% at 5 ns for each row: 5 < 0.8 * 20 * 1.1, it stays on. + for _ in 0..4 { + assert_eq!(feed_timed(&mut gate, 0.2, 5.0), GateDecision::Evaluate); + } + assert!(!gate.is_paused()); + + // The producer measures 4 ns for each removed row: 0.8 * 4 < 5. + let work = dynamic.removed_row_work(); + work.record(MIN_OBSERVED_ROWS, 4 * MIN_OBSERVED_ROWS); + assert_eq!(gate.saving_ns_per_row(), 4.0); + assert_eq!(feed_timed(&mut gate, 0.2, 5.0), GateDecision::Evaluate); + assert_eq!(feed_timed(&mut gate, 0.2, 5.0), GateDecision::Evaluate); + assert!(gate.is_paused()); + + // A filter without dynamic filters uses the configuration. + assert_eq!(new_gate().saving_ns_per_row(), 20.0); + } + /// A clock that moves by a fixed step each time it is read. #[derive(Debug)] struct SteppingClock { From ded1dd654377b5e56ac329c4669d4ae12be36c48 Mon Sep 17 00:00:00 2001 From: Adrian Garcia Badaracco <1755071+adriangb@users.noreply.github.com> Date: Thu, 24 Sep 2026 18:03:42 -0500 Subject: [PATCH 14/42] feat: add optional_filter_mode config Add `OptionalFilterMode` and the `datafusion.execution.optional_filter_mode` option (default `always`): - `always`: evaluate optional filters like any other pushed-down filter (today's behavior). - `adaptive`: evaluate each optional filter behind an `OptionalFilterGate`, which pauses it while it costs more than it saves or removes no rows. - `pruning_only`: use optional filters only for statistics pruning. The Parquet scan and `FilterExec` read this option. This commit alone does not change behavior. Co-Authored-By: Claude Opus 5.5 --- datafusion/common/src/config.rs | 70 +++++++++++++++++++ .../test_files/information_schema.slt | 2 + docs/source/user-guide/configs.md | 1 + 3 files changed, 73 insertions(+) diff --git a/datafusion/common/src/config.rs b/datafusion/common/src/config.rs index 24d52eb080fad..f05f4034f424f 100644 --- a/datafusion/common/src/config.rs +++ b/datafusion/common/src/config.rs @@ -883,6 +883,61 @@ impl Display for MapKeyDedupPolicy { } } +/// How DataFusion evaluates optional filters. +/// +/// Optional filters are filters that are not needed for correctness, such as +/// dynamic filters pushed down by hash joins and TopK. See +/// [`ExecutionOptions::optional_filter_mode`]. +#[derive(Debug, Default, Clone, Copy, PartialEq, Eq)] +pub enum OptionalFilterMode { + /// Evaluate optional filters like any other pushed-down filter. + #[default] + Always, + /// Pause optional filters that cost more than they save. Try them again + /// at intervals to find out if they became worth their cost. + Adaptive, + /// Use optional filters only for statistics pruning (for example of + /// files, row groups and pages). Never evaluate them row by row. + PruningOnly, +} + +impl FromStr for OptionalFilterMode { + type Err = DataFusionError; + + fn from_str(s: &str) -> Result { + match s.to_ascii_lowercase().as_str() { + "always" => Ok(Self::Always), + "adaptive" => Ok(Self::Adaptive), + "pruning_only" => Ok(Self::PruningOnly), + other => Err(DataFusionError::Configuration(format!( + "Invalid optional filter mode: {other}. Expected one of: always, adaptive, pruning_only" + ))), + } + } +} + +impl ConfigField for OptionalFilterMode { + fn visit(&self, v: &mut V, key: &str, description: &'static str) { + v.some(key, self, description) + } + + fn set(&mut self, _: &str, value: &str) -> Result<()> { + *self = OptionalFilterMode::from_str(value)?; + Ok(()) + } +} + +impl Display for OptionalFilterMode { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + let str = match self { + Self::Always => "always", + Self::Adaptive => "adaptive", + Self::PruningOnly => "pruning_only", + }; + write!(f, "{str}") + } +} + impl From for Option { fn from(c: SpillCompression) -> Self { match c { @@ -1185,6 +1240,21 @@ config_namespace! { /// Disabled by default, set to a number greater than 0 for enabling it. pub hash_join_buffering_capacity: usize, default = 0 + /// Controls how DataFusion evaluates filters that are not needed for + /// correctness, such as the dynamic filters that hash joins and TopK + /// push down into scans. `always` evaluates these filters like any + /// other pushed-down filter. `adaptive` pauses these filters when they + /// cost more than they save (see + /// `datafusion.execution.optional_filter_min_saving_ns_per_row`) or when + /// they remove no rows, and tries them again at intervals. `pruning_only` + /// uses these filters only to prune files, row groups and pages with + /// statistics, and never evaluates them row by row. + /// + /// This option is most important when + /// `datafusion.execution.parquet.pushdown_filters` is true, because then + /// the Parquet reader evaluates pushed-down filters row by row. + pub optional_filter_mode: OptionalFilterMode, default = OptionalFilterMode::Always + /// The assumed work, in nanoseconds, that each row removed by an /// optional filter saves downstream. Optional filters are filters that /// are not needed for correctness, such as the dynamic filters that diff --git a/datafusion/sqllogictest/test_files/information_schema.slt b/datafusion/sqllogictest/test_files/information_schema.slt index 1ecc7a97d310e..3f6ea66b6f1d0 100644 --- a/datafusion/sqllogictest/test_files/information_schema.slt +++ b/datafusion/sqllogictest/test_files/information_schema.slt @@ -232,6 +232,7 @@ datafusion.execution.meta_fetch_concurrency 32 datafusion.execution.minimum_parallel_output_files 4 datafusion.execution.objectstore_writer_buffer_size 10485760 datafusion.execution.optional_filter_min_saving_ns_per_row 20 +datafusion.execution.optional_filter_mode always datafusion.execution.parquet.allow_single_file_parallelism true datafusion.execution.parquet.binary_as_string false datafusion.execution.parquet.bloom_filter_fpp NULL @@ -394,6 +395,7 @@ datafusion.execution.meta_fetch_concurrency 32 Number of files to read in parall datafusion.execution.minimum_parallel_output_files 4 Guarantees a minimum level of output files running in parallel. RecordBatches will be distributed in round robin fashion to each parallel writer. Each writer is closed and a new file opened once soft_max_rows_per_output_file is reached. datafusion.execution.objectstore_writer_buffer_size 10485760 Size (bytes) of data buffer DataFusion uses when writing output files. This affects the size of the data chunks that are uploaded to remote object stores (e.g. AWS S3). If very large (>= 100 GiB) output files are being written, it may be necessary to increase this size to avoid errors from the remote end point. datafusion.execution.optional_filter_min_saving_ns_per_row 20 The assumed work, in nanoseconds, that each row removed by an optional filter saves downstream. Optional filters are filters that are not needed for correctness, such as the dynamic filters that hash joins and TopK push down into scans. When an operator evaluates optional filters adaptively, it pauses an optional filter whose evaluation costs more than the work that it saves. Consumers that can measure the saving (the Parquet scan) add their measured decode cost. The default is about the cost of a hash table probe for one row. The best value depends on the hardware. +datafusion.execution.optional_filter_mode always Controls how DataFusion evaluates filters that are not needed for correctness, such as the dynamic filters that hash joins and TopK push down into scans. `always` evaluates these filters like any other pushed-down filter. `adaptive` pauses these filters when they cost more than they save (see `datafusion.execution.optional_filter_min_saving_ns_per_row`) or when they remove no rows, and tries them again at intervals. `pruning_only` uses these filters only to prune files, row groups and pages with statistics, and never evaluates them row by row. This option is most important when `datafusion.execution.parquet.pushdown_filters` is true, because then the Parquet reader evaluates pushed-down filters row by row. datafusion.execution.parquet.allow_single_file_parallelism true (writing) Controls whether DataFusion will attempt to speed up writing parquet files by serializing them in parallel. Each column in each row group in each output file are serialized in parallel leveraging a maximum possible core count of n_files\*n_row_groups\*n_columns. datafusion.execution.parquet.binary_as_string false (reading) If true, parquet reader will read columns of `Binary/LargeBinary` with `Utf8`, and `BinaryView` with `Utf8View`. Parquet files generated by some legacy writers do not correctly set the UTF8 flag for strings, causing string columns to be loaded as BLOB instead. The parquet reader has special optimizations for `Utf8` validation, so reading such columns as strings is significantly faster than reading them as binary and then casting to string. datafusion.execution.parquet.bloom_filter_fpp NULL (writing) Sets bloom filter false positive probability. If NULL, uses default parquet writer setting diff --git a/docs/source/user-guide/configs.md b/docs/source/user-guide/configs.md index 178e3d109466c..3c87a2888e86f 100644 --- a/docs/source/user-guide/configs.md +++ b/docs/source/user-guide/configs.md @@ -145,6 +145,7 @@ The following configuration settings are available: | datafusion.execution.objectstore_writer_buffer_size | 10485760 | Size (bytes) of data buffer DataFusion uses when writing output files. This affects the size of the data chunks that are uploaded to remote object stores (e.g. AWS S3). If very large (>= 100 GiB) output files are being written, it may be necessary to increase this size to avoid errors from the remote end point. | | datafusion.execution.enable_ansi_mode | false | Whether to enable ANSI SQL mode. The flag is experimental and relevant only for DataFusion Spark built-in functions When `enable_ansi_mode` is set to `true`, the query engine follows ANSI SQL semantics for expressions, casting, and error handling. This means: - **Strict type coercion rules:** implicit casts between incompatible types are disallowed. - **Standard SQL arithmetic behavior:** operations such as division by zero, numeric overflow, or invalid casts raise runtime errors rather than returning `NULL` or adjusted values. - **Consistent ANSI behavior** for string concatenation, comparisons, and `NULL` handling. When `enable_ansi_mode` is `false` (the default), the engine uses a more permissive, non-ANSI mode designed for user convenience and backward compatibility. In this mode: - Implicit casts between types are allowed (e.g., string to integer when possible). - Arithmetic operations are more lenient — for example, `abs()` on the minimum representable integer value returns the input value instead of raising overflow. - Division by zero or invalid casts may return `NULL` instead of failing. # Default `false` — ANSI SQL mode is disabled by default. | | datafusion.execution.hash_join_buffering_capacity | 0 | How many bytes to buffer in the probe side of hash joins while the build side is concurrently being built. Without this, hash joins will wait until the full materialization of the build side before polling the probe side. This is useful in scenarios where the query is not completely CPU bounded, allowing to do some early work concurrently and reducing the latency of the query. Note that when hash join buffering is enabled, the probe side will start eagerly polling data, not giving time for the producer side of dynamic filters to produce any meaningful predicate. Queries with dynamic filters might see performance degradation. Disabled by default, set to a number greater than 0 for enabling it. | +| datafusion.execution.optional_filter_mode | always | Controls how DataFusion evaluates filters that are not needed for correctness, such as the dynamic filters that hash joins and TopK push down into scans. `always` evaluates these filters like any other pushed-down filter. `adaptive` pauses these filters when they cost more than they save (see `datafusion.execution.optional_filter_min_saving_ns_per_row`) or when they remove no rows, and tries them again at intervals. `pruning_only` uses these filters only to prune files, row groups and pages with statistics, and never evaluates them row by row. This option is most important when `datafusion.execution.parquet.pushdown_filters` is true, because then the Parquet reader evaluates pushed-down filters row by row. | | datafusion.execution.optional_filter_min_saving_ns_per_row | 20 | The assumed work, in nanoseconds, that each row removed by an optional filter saves downstream. Optional filters are filters that are not needed for correctness, such as the dynamic filters that hash joins and TopK push down into scans. When an operator evaluates optional filters adaptively, it pauses an optional filter whose evaluation costs more than the work that it saves. Consumers that can measure the saving (the Parquet scan) add their measured decode cost. The default is about the cost of a hash table probe for one row. The best value depends on the hardware. | | datafusion.optimizer.enable_distinct_aggregation_soft_limit | true | When set to true, the optimizer will push a limit operation into grouped aggregations which have no aggregate expressions, as a soft limit, emitting groups once the limit is reached, before all rows in the group are read. | | datafusion.optimizer.enable_round_robin_repartition | true | When set to true, the physical plan optimizer will try to add round robin repartitioning to increase parallelism to leverage more CPU cores | From 0a51a930ac4b2af3373c48d1f0a525cc8f0b4604 Mon Sep 17 00:00:00 2001 From: Adrian Garcia Badaracco <1755071+adriangb@users.noreply.github.com> Date: Thu, 24 Sep 2026 18:03:59 -0500 Subject: [PATCH 15/42] feat: let the Parquet scan skip optional filters that are not worth their cost Make the Parquet scan the first consumer of optional filters (conjuncts wrapped in `OptionalFilterPhysicalExpr`) when row level filter pushdown (`datafusion.execution.parquet.pushdown_filters`) is enabled. The `datafusion.execution.optional_filter_mode` option controls the row filter: - `always` (default): optional conjuncts are normal RowFilter predicates. Behavior does not change. - `adaptive`: each optional conjunct is a separate RowFilter predicate, after all required predicates, with an `OptionalFilterGate`. When the gate skips a batch, the predicate lets all rows pass without evaluation. Each file has its own gates; gates do not share state. - `pruning_only`: optional conjuncts are not in the RowFilter. In `adaptive` mode, the gate times each evaluation of the predicate and pauses a filter that removes no rows or that costs more than it saves. The saving of a removed row is the configured minimum (`datafusion.execution.optional_filter_min_saving_ns_per_row`) plus the decode time of the output columns that the filter does not read. The scan estimates this decode time from the compressed size of those column chunks (file metadata) and a decode speed in ns per compressed byte that it measures over the whole scan (time to produce each output batch from the decoder, which does not include the row filter). `ParquetSource::try_pushdown_filters` reads the mode and the minimum saving from the session configuration. In all modes, required conjuncts do not change, and statistics pruning (files, row groups, pages) uses optional conjuncts as before. An optional conjunct that cannot be pushed down for a file is dropped for that file. Add the lazily registered metrics `optional_filter_rows_skipped`, `optional_filter_pauses` and `optional_filter_eval_time`. Co-Authored-By: Claude Opus 5.5 --- datafusion/core/tests/parquet/mod.rs | 1 + .../core/tests/parquet/optional_filters.rs | 400 ++++++++ datafusion/datasource-parquet/src/metrics.rs | 88 ++ datafusion/datasource-parquet/src/mod.rs | 1 + .../datasource-parquet/src/opener/mod.rs | 90 +- .../datasource-parquet/src/optional_filter.rs | 246 +++++ .../datasource-parquet/src/push_decoder.rs | 46 +- .../datasource-parquet/src/row_filter.rs | 897 +++++++++++++++++- datafusion/datasource-parquet/src/source.rs | 86 ++ .../test_files/optional_filters.slt | 135 +++ docs/source/user-guide/explain-usage.md | 3 + 11 files changed, 1909 insertions(+), 84 deletions(-) create mode 100644 datafusion/core/tests/parquet/optional_filters.rs create mode 100644 datafusion/datasource-parquet/src/optional_filter.rs create mode 100644 datafusion/sqllogictest/test_files/optional_filters.slt diff --git a/datafusion/core/tests/parquet/mod.rs b/datafusion/core/tests/parquet/mod.rs index 308d7176c455e..77153c42b7ff5 100644 --- a/datafusion/core/tests/parquet/mod.rs +++ b/datafusion/core/tests/parquet/mod.rs @@ -53,6 +53,7 @@ mod expr_adapter; mod external_access_plan; mod file_statistics; mod filter_pushdown; +mod optional_filters; mod ordering; mod page_pruning; mod row_group_pruning; diff --git a/datafusion/core/tests/parquet/optional_filters.rs b/datafusion/core/tests/parquet/optional_filters.rs new file mode 100644 index 0000000000000..bff9366d4fa1f --- /dev/null +++ b/datafusion/core/tests/parquet/optional_filters.rs @@ -0,0 +1,400 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! End-to-end tests for *optional filters* in the Parquet scan with row +//! level filter pushdown (`pushdown_filters = true`). +//! +//! An optional filter is a predicate conjunct wrapped in an +//! `OptionalFilterPhysicalExpr`. The scan handles it as +//! `datafusion.execution.optional_filter_mode` says: +//! +//! * `always`: like all other filters. +//! * `adaptive`: an `OptionalFilterGate` skips it while it removes no rows, +//! or while it costs more than it saves. +//! * `pruning_only`: only statistics pruning uses it. +//! +//! No operator makes optional filters yet, thus these tests push an +//! `Optional(...)` predicate into the `ParquetSource` directly, with +//! `FileSource::try_pushdown_filters` and the session configuration. + +use std::sync::Arc; + +use arrow::array::{Array, Int32Array, Int64Array, RecordBatch}; +use arrow::compute::concat_batches; +use arrow_schema::{DataType, Field, Schema, SchemaRef}; +use datafusion::datasource::listing::PartitionedFile; +use datafusion::datasource::object_store::ObjectStoreUrl; +use datafusion::datasource::physical_plan::ParquetSource; +use datafusion::physical_plan::filter::FilterExec; +use datafusion::physical_plan::{ExecutionPlan, collect, execute_stream}; +use datafusion::prelude::{SessionConfig, SessionContext}; +use datafusion_common::ScalarValue; +use datafusion_common::config::{ConfigOptions, OptionalFilterMode}; +use datafusion_datasource::file::FileSource; +use datafusion_datasource::file_scan_config::FileScanConfigBuilder; +use datafusion_datasource::source::DataSourceExec; +use datafusion_expr::Operator; +use datafusion_physical_expr::PhysicalExpr; +use datafusion_physical_expr::expressions::{ + BinaryExpr, Column, DynamicFilterPhysicalExpr, OptionalFilterPhysicalExpr, lit, +}; +use futures::StreamExt; +use parquet::arrow::ArrowWriter; +use parquet::file::properties::WriterProperties; +use tempfile::NamedTempFile; + +use crate::parquet::utils::MetricsFinder; + +const ROW_GROUPS: usize = 10; +const ROWS_PER_ROW_GROUP: usize = 2000; +/// Each row filter evaluation sees one batch of this many rows. +const BATCH_SIZE: usize = 100; + +/// A file with `ROW_GROUPS` row groups. Column `a` is `i % 100` (each row +/// group has all values `0..100`, thus statistics cannot prune it), `rg` is +/// the row group index and `v` is the row number. +fn write_file() -> (NamedTempFile, SchemaRef) { + let schema = Arc::new(Schema::new(vec![ + Field::new("a", DataType::Int32, false), + Field::new("rg", DataType::Int32, false), + Field::new("v", DataType::Int64, false), + ])); + let file = NamedTempFile::new().unwrap(); + let props = WriterProperties::builder() + .set_max_row_group_row_count(Some(ROWS_PER_ROW_GROUP)) + .build(); + let mut writer = + ArrowWriter::try_new(file.reopen().unwrap(), Arc::clone(&schema), Some(props)) + .unwrap(); + for rg in 0..ROW_GROUPS { + let start = rg * ROWS_PER_ROW_GROUP; + let rows = start..start + ROWS_PER_ROW_GROUP; + let a: Int32Array = rows.clone().map(|i| (i % 100) as i32).collect(); + let rg_col = Int32Array::from(vec![rg as i32; ROWS_PER_ROW_GROUP]); + let v: Int64Array = rows.map(|i| i as i64).collect(); + let batch = RecordBatch::try_new( + Arc::clone(&schema), + vec![Arc::new(a), Arc::new(rg_col), Arc::new(v)], + ) + .unwrap(); + writer.write(&batch).unwrap(); + } + let metadata = writer.close().unwrap(); + assert_eq!(metadata.num_row_groups(), ROW_GROUPS); + (file, schema) +} + +fn col_a(schema: &Schema) -> Arc { + Arc::new(Column::new_with_schema("a", schema).unwrap()) +} + +fn a_op(schema: &Schema, op: Operator, value: i32) -> Arc { + Arc::new(BinaryExpr::new( + col_a(schema), + op, + lit(ScalarValue::Int32(Some(value))), + )) +} + +/// `a % 100 < 100`: true for all rows. Statistics cannot prove this, thus +/// the scan evaluates the filter for each row (it does not skip the row +/// filter for fully matched row groups). +fn removes_no_rows(schema: &Schema) -> Arc { + let a_mod_100: Arc = Arc::new(BinaryExpr::new( + col_a(schema), + Operator::Modulo, + lit(ScalarValue::Int32(Some(100))), + )); + Arc::new(BinaryExpr::new( + a_mod_100, + Operator::Lt, + lit(ScalarValue::Int32(Some(100))), + )) +} + +fn optional(inner: Arc) -> Arc { + Arc::new(OptionalFilterPhysicalExpr::new(inner)) +} + +/// A scan of all columns of `file`. The gates of the optional filters +/// assume that a removed row saves a very large amount of work, thus the +/// cost check (which uses the wall clock) never pauses a filter that +/// removes rows, and the tests are deterministic. See [`scan_with`]. +fn scan( + file: &NamedTempFile, + schema: &SchemaRef, + predicate: Arc, + mode: OptionalFilterMode, +) -> Arc { + scan_with(file, schema, predicate, mode, 1e9, None) +} + +/// A scan of the columns `projection` (all columns if `None`) of `file`, +/// with `min_saving_ns_per_row` for the gates of the optional filters. +fn scan_with( + file: &NamedTempFile, + schema: &SchemaRef, + predicate: Arc, + mode: OptionalFilterMode, + min_saving_ns_per_row: f64, + projection: Option>, +) -> Arc { + let mut options = ConfigOptions::default(); + options.execution.parquet.pushdown_filters = true; + options.execution.optional_filter_mode = mode; + options.execution.optional_filter_min_saving_ns_per_row = min_saving_ns_per_row; + let source = ParquetSource::new(Arc::clone(schema)) + .try_pushdown_filters(vec![predicate], &options) + .unwrap() + .updated_node + .expect("the scan accepts the predicate"); + let path = file.path().to_str().unwrap().to_string(); + let size = std::fs::metadata(&path).unwrap().len(); + let config = FileScanConfigBuilder::new(ObjectStoreUrl::local_filesystem(), source) + .with_file(PartitionedFile::new(path, size)) + .with_projection_indices(projection) + .unwrap() + .build(); + DataSourceExec::from_data_source(config) +} + +fn session() -> SessionContext { + SessionContext::new_with_config(SessionConfig::new().with_batch_size(BATCH_SIZE)) +} + +fn metric(plan: &dyn ExecutionPlan, name: &str) -> usize { + MetricsFinder::find_metrics(plan) + .unwrap() + .sum_by_name(name) + .map_or(0, |v| v.as_usize()) +} + +/// Sorted values of column `v`. +fn values(batches: &[RecordBatch]) -> Vec { + let mut values: Vec = batches + .iter() + .flat_map(|batch| { + let v = batch + .column_by_name("v") + .unwrap() + .as_any() + .downcast_ref::() + .unwrap(); + v.values().to_vec() + }) + .collect(); + values.sort_unstable(); + values +} + +/// An optional filter that removes no rows is paused by the gate in +/// `adaptive` mode. The consumer of the scan (here a +/// `FilterExec`, like a hash join for a join dynamic filter) applies the +/// filter again, thus the results are the same in all modes. +#[tokio::test] +async fn optional_filter_that_removes_no_rows_is_paused() { + let (file, schema) = write_file(); + let total_rows = ROW_GROUPS * ROWS_PER_ROW_GROUP; + let removed_rows = 0; + + let mut results = vec![]; + for mode in [ + OptionalFilterMode::Always, + OptionalFilterMode::Adaptive, + OptionalFilterMode::PruningOnly, + ] { + let filter = removes_no_rows(&schema); + let scan = scan(&file, &schema, optional(Arc::clone(&filter)), mode); + let plan: Arc = + Arc::new(FilterExec::try_new(filter, Arc::clone(&scan)).unwrap()); + let batches = collect(plan, session().task_ctx()).await.unwrap(); + let values = values(&batches); + assert_eq!(values.len(), total_rows - removed_rows, "mode {mode}"); + + let pruned = metric(scan.as_ref(), "pushdown_rows_pruned"); + let skipped = metric(scan.as_ref(), "optional_filter_rows_skipped"); + let pauses = metric(scan.as_ref(), "optional_filter_pauses"); + match mode { + OptionalFilterMode::Always => { + assert_eq!(pruned, removed_rows); + assert_eq!(skipped, 0); + assert_eq!(pauses, 0); + } + OptionalFilterMode::Adaptive => { + assert!(skipped > total_rows / 2, "skipped {skipped} rows"); + assert!(pauses > 0); + assert_eq!(pruned, removed_rows); + // Each row is either skipped or evaluated. + let matched = metric(scan.as_ref(), "pushdown_rows_matched"); + assert_eq!(matched + pruned, total_rows); + } + OptionalFilterMode::PruningOnly => { + // No row filter at all. + assert_eq!(pruned, 0); + assert_eq!(metric(scan.as_ref(), "pushdown_rows_matched"), 0); + assert_eq!(skipped, 0); + assert_eq!(pauses, 0); + } + } + results.push(values); + } + assert_eq!(results[0], results[1]); + assert_eq!(results[0], results[2]); +} + +/// A selective optional filter is never paused, and the scan output is the +/// same as in `always` mode. +#[tokio::test] +async fn selective_optional_filter_is_not_paused() { + let (file, schema) = write_file(); + let predicate = optional(a_op(&schema, Operator::Lt, 10)); + let mut results = vec![]; + for mode in [OptionalFilterMode::Always, OptionalFilterMode::Adaptive] { + let scan = scan(&file, &schema, Arc::clone(&predicate), mode); + let batches = collect(Arc::clone(&scan), session().task_ctx()) + .await + .unwrap(); + assert_eq!(metric(scan.as_ref(), "optional_filter_rows_skipped"), 0); + assert_eq!(metric(scan.as_ref(), "optional_filter_pauses"), 0); + results.push(values(&batches)); + } + assert_eq!(results[0].len(), ROW_GROUPS * ROWS_PER_ROW_GROUP / 10); + assert_eq!(results[0], results[1]); +} + +/// A selective optional filter is paused when it costs more than it saves. +/// The scan reads only column `a`, which the filter reads too: a removed row +/// saves no decode time. With `min_saving_ns_per_row = 0`, a removed row +/// saves nothing, thus any evaluation time is too much and the gate pauses +/// `a < 10` although it keeps only 10% of the rows. The consumer of the scan +/// applies the filter again, thus the result does not change. +#[tokio::test] +async fn selective_optional_filter_is_paused_when_it_costs_more_than_it_saves() { + let (file, schema) = write_file(); + let filter = a_op(&schema, Operator::Lt, 10); + let scan = scan_with( + &file, + &schema, + optional(Arc::clone(&filter)), + OptionalFilterMode::Adaptive, + 0.0, + Some(vec![0]), + ); + let plan: Arc = + Arc::new(FilterExec::try_new(filter, Arc::clone(&scan)).unwrap()); + let batches = collect(plan, session().task_ctx()).await.unwrap(); + let rows: usize = batches.iter().map(|b| b.num_rows()).sum(); + assert_eq!(rows, ROW_GROUPS * ROWS_PER_ROW_GROUP / 10); + assert!(metric(scan.as_ref(), "optional_filter_pauses") > 0); + assert!(metric(scan.as_ref(), "optional_filter_rows_skipped") > 0); + assert!(metric(scan.as_ref(), "optional_filter_eval_time") > 0); +} + +/// A required conjunct stays a normal row filter predicate in all modes. +#[tokio::test] +async fn required_conjunct_is_unchanged() { + let (file, schema) = write_file(); + // a < 50 AND Optional(a != 99) + let predicate: Arc = Arc::new(BinaryExpr::new( + a_op(&schema, Operator::Lt, 50), + Operator::And, + optional(a_op(&schema, Operator::NotEq, 99)), + )); + for mode in [ + OptionalFilterMode::Always, + OptionalFilterMode::Adaptive, + OptionalFilterMode::PruningOnly, + ] { + let scan = scan(&file, &schema, Arc::clone(&predicate), mode); + let batches = collect(Arc::clone(&scan), session().task_ctx()) + .await + .unwrap(); + assert_eq!( + values(&batches).len(), + ROW_GROUPS * ROWS_PER_ROW_GROUP / 2, + "mode {mode}" + ); + } +} + +/// An optional dynamic filter that changes during the scan: the gate paused +/// the first version of the filter (which removes no rows), and evaluates the filter +/// again when the filter changes. +#[tokio::test] +async fn optional_dynamic_filter_update_restarts_evaluation() { + let (file, schema) = write_file(); + let dynamic = Arc::new(DynamicFilterPhysicalExpr::new( + vec![col_a(&schema)], + removes_no_rows(&schema), + )); + let scan = scan( + &file, + &schema, + optional(Arc::clone(&dynamic) as Arc), + OptionalFilterMode::Adaptive, + ); + let mut stream = execute_stream(Arc::clone(&scan), session().task_ctx()).unwrap(); + + // Read until the scan is in the third row group. The gate paused the + // filter in the first row group. + let mut before_update = vec![]; + let mut max_rg_before_update = 0; + while max_rg_before_update < 2 { + let batch = stream.next().await.unwrap().unwrap(); + max_rg_before_update = max_rg_before_update.max(max_rg(&batch)); + before_update.push(batch); + } + assert!(metric(scan.as_ref(), "optional_filter_rows_skipped") > 0); + assert!(metric(scan.as_ref(), "optional_filter_pauses") > 0); + + // The filter becomes selective. + dynamic.update(a_op(&schema, Operator::Lt, 10)).unwrap(); + + let mut after_update = vec![]; + while let Some(batch) = stream.next().await { + after_update.push(batch.unwrap()); + } + let after_update = concat_batches(&after_update[0].schema(), &after_update).unwrap(); + // The scan evaluates the filter of a row group before it returns the + // first batch of that row group. Thus all row groups after the current + // one use the new filter. + let a = column_i32(&after_update, "a"); + let rg = column_i32(&after_update, "rg"); + let mut checked_rows = 0; + for (a, rg) in a.iter().zip(rg.iter()) { + if *rg > max_rg_before_update { + assert!(*a < 10, "row with a = {a} in row group {rg} passed"); + checked_rows += 1; + } + } + let later_row_groups = ROW_GROUPS - 1 - max_rg_before_update as usize; + assert_eq!(checked_rows, later_row_groups * ROWS_PER_ROW_GROUP / 10); +} + +fn column_i32<'a>(batch: &'a RecordBatch, name: &str) -> &'a [i32] { + batch + .column_by_name(name) + .unwrap() + .as_any() + .downcast_ref::() + .unwrap() + .values() +} + +fn max_rg(batch: &RecordBatch) -> i32 { + column_i32(batch, "rg").iter().copied().max().unwrap_or(0) +} diff --git a/datafusion/datasource-parquet/src/metrics.rs b/datafusion/datasource-parquet/src/metrics.rs index 4abadd2c4e13a..e73b3de165985 100644 --- a/datafusion/datasource-parquet/src/metrics.rs +++ b/datafusion/datasource-parquet/src/metrics.rs @@ -16,6 +16,7 @@ // under the License. use std::sync::Arc; +use std::time::Duration; use datafusion_physical_plan::metrics::{ Count, ExecutionPlanMetricsSet, Gauge, Label, MetricBuilder, MetricCategory, @@ -491,3 +492,90 @@ impl RowFilterSkippedFullyMatchedMetric { count.add(1); } } + +/// Lazily-registered counters for optional filters that an +/// [`OptionalFilterGate`](datafusion_physical_expr::optional_filter_gate::OptionalFilterGate) +/// evaluates in the Parquet `RowFilter` (when +/// `datafusion.execution.optional_filter_mode` is `adaptive`): +/// +/// * `optional_filter_rows_skipped`: rows for which the gate skipped an +/// optional filter (all these rows passed that filter without evaluation). +/// * `optional_filter_pauses`: number of times a gate paused an optional +/// filter (because it removed no rows or cost more than it saved). +/// * `optional_filter_eval_time`: time spent to evaluate optional filters. +/// `row_pushdown_eval_time` includes this time. +/// +/// Like [`RowFilterSkippedFullyMatchedMetric`], each counter is registered +/// only when it first fires, so scans that never skip an optional filter do +/// not show zero-valued counters in `EXPLAIN ANALYZE`. +#[derive(Debug)] +pub(crate) struct OptionalFilterMetrics { + metrics: ExecutionPlanMetricsSet, + partition: usize, + filename: String, + rows_skipped: Option, + pauses: Option, + eval_time: Option