From d112ecc0b55b259888043aecf415108880dacee0 Mon Sep 17 00:00:00 2001 From: "xavier.lee" Date: Wed, 26 Aug 2026 00:08:31 -0400 Subject: [PATCH 1/5] refactor: separate aggregate group completion from input ordering --- datafusion/core/tests/dataframe/mod.rs | 8 +- .../core/tests/fuzz_cases/aggregate_fuzz.rs | 19 +- .../aggregation_fuzzer/query_builder.rs | 4 +- .../enforce_distribution.rs | 8 +- .../limited_distinct_aggregation.rs | 2 +- .../tests/physical_optimizer/pushdown_sort.rs | 4 +- .../benches/dictionary_group_values.rs | 14 +- .../benches/ordered_group_values.rs | 22 +- .../physical-plan/benches/partial_ordering.rs | 16 +- ...inal_table.rs => clustered_final_table.rs} | 30 +- ...al_table.rs => clustered_partial_table.rs} | 28 +- ...gle_table.rs => clustered_single_table.rs} | 22 +- .../aggregates/aggregate_hash_table/common.rs | 4 +- ...{common_ordered.rs => common_clustered.rs} | 92 +++-- .../aggregates/aggregate_hash_table/mod.rs | 12 +- .../aggregate_hash_table/partial_table.rs | 4 +- ...al_stream.rs => clustered_final_stream.rs} | 95 ++--- ..._stream.rs => clustered_partial_stream.rs} | 95 ++--- ...e_stream.rs => clustered_single_stream.rs} | 228 +++++------ .../src/aggregates/group_values/mod.rs | 24 +- .../{ordered.rs => clustered.rs} | 82 ++-- .../group_values/multi_group_by/list.rs | 12 +- .../group_values/multi_group_by/mod.rs | 4 +- .../src/aggregates/grouped_hash_stream.rs | 60 ++- .../src/aggregates/hash_stream.rs | 29 +- .../physical-plan/src/aggregates/mod.rs | 368 ++++++++++++------ .../src/aggregates/order/full.rs | 16 +- .../physical-plan/src/aggregates/order/mod.rs | 167 ++++---- .../src/aggregates/order/partial.rs | 192 ++++----- .../src/aggregates/partial_reduce_stream.rs | 5 +- .../src/aggregates/single_stream.rs | 13 +- .../physical-plan/src/aggregates/spill.rs | 49 +-- .../physical-plan/src/recursive_query.rs | 4 +- .../test_files/agg_func_substitute.slt | 12 +- .../sqllogictest/test_files/aggregate.slt | 12 +- .../sqllogictest/test_files/group_by.slt | 58 +-- datafusion/sqllogictest/test_files/joins.slt | 8 +- datafusion/sqllogictest/test_files/order.slt | 2 +- .../test_files/ordered_aggregate_spill.slt | 24 +- .../test_files/preserve_file_partitioning.slt | 12 +- .../test_files/range_sorted_time_bin_agg.slt | 14 +- .../repartition_subset_satisfaction.slt | 12 +- .../sqllogictest/test_files/sort_pushdown.slt | 10 +- datafusion/sqllogictest/test_files/unnest.slt | 4 +- datafusion/sqllogictest/test_files/window.slt | 2 +- .../library-user-guide/upgrading/56.0.0.md | 22 ++ 46 files changed, 1081 insertions(+), 843 deletions(-) rename datafusion/physical-plan/src/aggregates/aggregate_hash_table/{ordered_final_table.rs => clustered_final_table.rs} (76%) rename datafusion/physical-plan/src/aggregates/aggregate_hash_table/{ordered_partial_table.rs => clustered_partial_table.rs} (76%) rename datafusion/physical-plan/src/aggregates/aggregate_hash_table/{ordered_single_table.rs => clustered_single_table.rs} (79%) rename datafusion/physical-plan/src/aggregates/aggregate_hash_table/{common_ordered.rs => common_clustered.rs} (84%) rename datafusion/physical-plan/src/aggregates/{ordered_final_stream.rs => clustered_final_stream.rs} (89%) rename datafusion/physical-plan/src/aggregates/{ordered_partial_stream.rs => clustered_partial_stream.rs} (81%) rename datafusion/physical-plan/src/aggregates/{ordered_single_stream.rs => clustered_single_stream.rs} (77%) rename datafusion/physical-plan/src/aggregates/group_values/multi_group_by/{ordered.rs => clustered.rs} (87%) diff --git a/datafusion/core/tests/dataframe/mod.rs b/datafusion/core/tests/dataframe/mod.rs index 43ceee3444ced..9ffe34893b9e1 100644 --- a/datafusion/core/tests/dataframe/mod.rs +++ b/datafusion/core/tests/dataframe/mod.rs @@ -3422,9 +3422,9 @@ async fn union_with_mix_of_presorted_and_explicitly_resorted_inputs_with_reparti assert_snapshot!( union_with_mix_of_presorted_and_explicitly_resorted_inputs_impl(false).await?, @r" - AggregateExec: mode=Final, gby=[id@0 as id], aggr=[], ordering_mode=Sorted + AggregateExec: mode=Final, gby=[id@0 as id], aggr=[], group_completion_mode=Full SortPreservingMergeExec: [id@0 ASC NULLS LAST] - AggregateExec: mode=Partial, gby=[id@0 as id], aggr=[], ordering_mode=Sorted + AggregateExec: mode=Partial, gby=[id@0 as id], aggr=[], group_completion_mode=Full UnionExec DataSourceExec: file_groups={1 group: [[{testdata}/alltypes_tiny_pages.parquet]]}, projection=[id], output_ordering=[id@0 ASC NULLS LAST], file_type=parquet SortExec: expr=[id@0 ASC NULLS LAST], preserve_partitioning=[false] @@ -3440,9 +3440,9 @@ async fn union_with_mix_of_presorted_and_explicitly_resorted_inputs_with_reparti assert_snapshot!( union_with_mix_of_presorted_and_explicitly_resorted_inputs_impl(true).await?, @r" - AggregateExec: mode=Final, gby=[id@0 as id], aggr=[], ordering_mode=Sorted + AggregateExec: mode=Final, gby=[id@0 as id], aggr=[], group_completion_mode=Full SortPreservingMergeExec: [id@0 ASC NULLS LAST] - AggregateExec: mode=Partial, gby=[id@0 as id], aggr=[], ordering_mode=Sorted + AggregateExec: mode=Partial, gby=[id@0 as id], aggr=[], group_completion_mode=Full UnionExec DataSourceExec: file_groups={1 group: [[{testdata}/alltypes_tiny_pages.parquet]]}, projection=[id], output_ordering=[id@0 ASC NULLS LAST], file_type=parquet SortExec: expr=[id@0 ASC NULLS LAST], preserve_partitioning=[false] diff --git a/datafusion/core/tests/fuzz_cases/aggregate_fuzz.rs b/datafusion/core/tests/fuzz_cases/aggregate_fuzz.rs index afbb1b21d0856..5db48946ea3cb 100644 --- a/datafusion/core/tests/fuzz_cases/aggregate_fuzz.rs +++ b/datafusion/core/tests/fuzz_cases/aggregate_fuzz.rs @@ -41,7 +41,6 @@ use datafusion_common_runtime::JoinSet; use datafusion_functions_aggregate::sum::sum_udaf; use datafusion_physical_expr::PhysicalSortExpr; use datafusion_physical_expr::expressions::{Column, col, lit}; -use datafusion_physical_plan::InputOrderMode; use test_utils::{StringBatchGenerator, add_empty_batches}; use datafusion_execution::TaskContext; @@ -49,7 +48,7 @@ use datafusion_execution::memory_pool::FairSpillPool; use datafusion_execution::runtime_env::RuntimeEnvBuilder; use datafusion_physical_expr::aggregate::AggregateExprBuilder; use datafusion_physical_plan::aggregates::{ - AggregateExec, AggregateMode, PhysicalGroupBy, + AggregateExec, AggregateMode, GroupCompletionMode, PhysicalGroupBy, }; use datafusion_physical_plan::metrics::MetricValue; use datafusion_physical_plan::{ExecutionPlan, collect, displayable}; @@ -303,7 +302,7 @@ async fn streaming_aggregate_test() { /// two `AggregateExec` variants produce the same result: the pipeline breaking /// one over unordered input (`PartialHashAggregateStream`) and the /// non-pipeline breaking one over ordered input -/// (`OrderedPartialAggregateStream`). +/// (`ClusteredPartialAggregateStream`). async fn run_aggregate_test(input1: Vec, group_by_columns: Vec<&str>) { let schema = input1[0].schema(); let session_config = SessionConfig::new().with_batch_size(50); @@ -354,8 +353,8 @@ async fn run_aggregate_test(input1: Vec, group_by_columns: Vec<&str .unwrap(), ); assert_ne!( - aggregate_exec_running.input_order_mode(), - &InputOrderMode::Linear, + aggregate_exec_running.group_completion_mode(), + &GroupCompletionMode::None, "running aggregate should observe ordered input for group_by: {group_by:?}" ); @@ -555,13 +554,17 @@ async fn verify_ordered_aggregate(frame: &DataFrame, expected_sort: bool) { fn f_down(&mut self, node: &'n Self::Node) -> Result { if let Some(exec) = node.downcast_ref::() { + assert_eq!( + exec.properties().output_ordering().is_some(), + self.expected_sort + ); if self.expected_sort { assert!(matches!( - exec.input_order_mode(), - InputOrderMode::PartiallySorted(_) | InputOrderMode::Sorted + exec.group_completion_mode(), + GroupCompletionMode::Partial(_) | GroupCompletionMode::Full )); } else { - assert_eq!(*exec.input_order_mode(), InputOrderMode::Linear); + assert_eq!(*exec.group_completion_mode(), GroupCompletionMode::None); } } Ok(TreeNodeRecursion::Continue) diff --git a/datafusion/core/tests/fuzz_cases/aggregation_fuzzer/query_builder.rs b/datafusion/core/tests/fuzz_cases/aggregation_fuzzer/query_builder.rs index 2b2d2483d73f9..debeceb433293 100644 --- a/datafusion/core/tests/fuzz_cases/aggregation_fuzzer/query_builder.rs +++ b/datafusion/core/tests/fuzz_cases/aggregation_fuzzer/query_builder.rs @@ -93,9 +93,9 @@ pub struct QueryBuilder { /// ... /// ``` /// - /// More details can see [`GroupOrdering`]. + /// More details can see [`GroupCompletion`]. /// - /// [`GroupOrdering`]: datafusion_physical_plan::aggregates::order::GroupOrdering + /// [`GroupCompletion`]: datafusion_physical_plan::aggregates::order::GroupCompletion dataset_sort_keys: Vec>, /// If we will also test the no grouping case like: diff --git a/datafusion/core/tests/physical_optimizer/enforce_distribution.rs b/datafusion/core/tests/physical_optimizer/enforce_distribution.rs index c280a0aa5ef80..66e14ec8b1181 100644 --- a/datafusion/core/tests/physical_optimizer/enforce_distribution.rs +++ b/datafusion/core/tests/physical_optimizer/enforce_distribution.rs @@ -4867,9 +4867,9 @@ fn preserve_ordering_for_streaming_sorted_aggregate() -> Result<()> { let plan_distrib = test_config.to_plan(physical_plan.clone(), &DISTRIB_DISTRIB_SORT); assert_plan!(plan_distrib, @r" - AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[COUNT(b)], ordering_mode=Sorted + AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[COUNT(b)], group_completion_mode=Full RepartitionExec: partitioning=Hash([a@0], 2), input_partitions=2, preserve_order=true, sort_exprs=a@0 ASC - AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[COUNT(b)], ordering_mode=Sorted + AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[COUNT(b)], group_completion_mode=Full DataSourceExec: file_groups={2 groups: [[x], [y]]}, projection=[a, b, c, d, e], output_ordering=[a@0 ASC], file_type=parquet "); @@ -4901,9 +4901,9 @@ fn preserve_ordering_for_streaming_partially_sorted_aggregate() -> Result<()> { let plan_distrib = test_config.to_plan(physical_plan.clone(), &DISTRIB_DISTRIB_SORT); assert_plan!(plan_distrib, @r" - AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b], aggr=[COUNT(c)], ordering_mode=PartiallySorted([0]) + AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b], aggr=[COUNT(c)], group_completion_mode=Partial([0]) RepartitionExec: partitioning=Hash([a@0, b@1], 2), input_partitions=2, preserve_order=true, sort_exprs=a@0 ASC - AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b], aggr=[COUNT(c)], ordering_mode=PartiallySorted([0]) + AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b], aggr=[COUNT(c)], group_completion_mode=Partial([0]) DataSourceExec: file_groups={2 groups: [[x], [y]]}, projection=[a, b, c, d, e], output_ordering=[a@0 ASC], file_type=parquet "); diff --git a/datafusion/core/tests/physical_optimizer/limited_distinct_aggregation.rs b/datafusion/core/tests/physical_optimizer/limited_distinct_aggregation.rs index 1d5737d2431ab..6164445b98e2d 100644 --- a/datafusion/core/tests/physical_optimizer/limited_distinct_aggregation.rs +++ b/datafusion/core/tests/physical_optimizer/limited_distinct_aggregation.rs @@ -520,7 +520,7 @@ fn test_has_order_by() -> Result<()> { actual, @r" LocalLimitExec: fetch=10 - AggregateExec: mode=Single, gby=[a@0 as a], aggr=[], ordering_mode=Sorted + AggregateExec: mode=Single, gby=[a@0 as a], aggr=[], group_completion_mode=Full DataSourceExec: file_groups={1 group: [[x]]}, projection=[a, b, c, d, e], output_ordering=[a@0 ASC], file_type=parquet " ); diff --git a/datafusion/core/tests/physical_optimizer/pushdown_sort.rs b/datafusion/core/tests/physical_optimizer/pushdown_sort.rs index b72563a942ae3..97cfa05d12785 100644 --- a/datafusion/core/tests/physical_optimizer/pushdown_sort.rs +++ b/datafusion/core/tests/physical_optimizer/pushdown_sort.rs @@ -731,13 +731,13 @@ fn test_pushdown_through_blocking_node() { OptimizationTest: input: - SortExec: expr=[a@0 ASC], preserve_partitioning=[false] - - AggregateExec: mode=Final, gby=[a@0 as a], aggr=[COUNT(b)], ordering_mode=Sorted + - AggregateExec: mode=Final, gby=[a@0 as a], aggr=[COUNT(b)], group_completion_mode=Full - SortExec: expr=[a@0 DESC NULLS LAST], preserve_partitioning=[false] - DataSourceExec: file_groups={1 group: [[x]]}, projection=[a, b, c, d, e], output_ordering=[a@0 ASC], file_type=parquet output: Ok: - SortExec: expr=[a@0 ASC], preserve_partitioning=[false] - - AggregateExec: mode=Final, gby=[a@0 as a], aggr=[COUNT(b)], ordering_mode=Sorted + - AggregateExec: mode=Final, gby=[a@0 as a], aggr=[COUNT(b)], group_completion_mode=Full - SortExec: expr=[a@0 DESC NULLS LAST], preserve_partitioning=[false] - DataSourceExec: file_groups={1 group: [[x]]}, projection=[a, b, c, d, e], file_type=parquet, sort_order_for_reorder=[a@0 DESC NULLS LAST], reverse_row_groups=true " diff --git a/datafusion/physical-plan/benches/dictionary_group_values.rs b/datafusion/physical-plan/benches/dictionary_group_values.rs index 502869787ee40..c9ff2cf34b7dc 100644 --- a/datafusion/physical-plan/benches/dictionary_group_values.rs +++ b/datafusion/physical-plan/benches/dictionary_group_values.rs @@ -29,7 +29,7 @@ use criterion::{ }; use datafusion_expr::EmitTo; use datafusion_physical_plan::aggregates::group_values::new_group_values; -use datafusion_physical_plan::aggregates::order::{GroupOrdering, GroupOrderingFull}; +use datafusion_physical_plan::aggregates::order::{GroupCompletion, GroupCompletionFull}; use rand::rngs::StdRng; use rand::seq::SliceRandom; use rand::{Rng, SeedableRng}; @@ -106,7 +106,7 @@ fn bench_intern_emit(c: &mut Criterion) { b.iter_batched_ref( || { ( - new_group_values(schema.clone(), &GroupOrdering::None) + new_group_values(schema.clone(), &GroupCompletion::None) .unwrap(), Vec::::with_capacity(size), ) @@ -151,7 +151,7 @@ fn bench_repeated_intern_emit(c: &mut Criterion) { b.iter_batched_ref( || { ( - new_group_values(schema.clone(), &GroupOrdering::None) + new_group_values(schema.clone(), &GroupCompletion::None) .unwrap(), Vec::::with_capacity(size), ) @@ -172,7 +172,7 @@ fn bench_repeated_intern_emit(c: &mut Criterion) { group.finish(); } -// GroupOrdering::Full -> GroupValuesColumn::: scalar append_val/equal_to path. +// GroupCompletion::Full -> GroupValuesColumn::: scalar append_val/equal_to path. fn bench_scalar_append_equal(c: &mut Criterion) { let mut group = c.benchmark_group("dict_scalar_append_equal"); let schema = dict_schema(); @@ -192,7 +192,7 @@ fn bench_scalar_append_equal(c: &mut Criterion) { ( new_group_values( schema.clone(), - &GroupOrdering::Full(GroupOrderingFull::new()), + &GroupCompletion::Full(GroupCompletionFull::new()), ) .unwrap(), Vec::::with_capacity(size), @@ -227,7 +227,7 @@ fn bench_take_n(c: &mut Criterion) { b.iter_batched_ref( || { ( - new_group_values(schema.clone(), &GroupOrdering::None).unwrap(), + new_group_values(schema.clone(), &GroupCompletion::None).unwrap(), Vec::::with_capacity(size), ) }, @@ -298,7 +298,7 @@ fn bench_shared_values_arc(c: &mut Criterion) { b.iter_batched_ref( || { ( - new_group_values(schema.clone(), &GroupOrdering::None) + new_group_values(schema.clone(), &GroupCompletion::None) .unwrap(), Vec::::with_capacity(size), ) diff --git a/datafusion/physical-plan/benches/ordered_group_values.rs b/datafusion/physical-plan/benches/ordered_group_values.rs index 30af725cddcfc..c7eb02b93826e 100644 --- a/datafusion/physical-plan/benches/ordered_group_values.rs +++ b/datafusion/physical-plan/benches/ordered_group_values.rs @@ -38,12 +38,12 @@ use datafusion_physical_expr::expressions::col; use datafusion_physical_expr::{LexOrdering, PhysicalSortExpr}; use datafusion_physical_plan::aggregates::group_values::multi_group_by::GroupValuesColumn; use datafusion_physical_plan::aggregates::group_values::{GroupValues, new_group_values}; -use datafusion_physical_plan::aggregates::order::GroupOrdering; +use datafusion_physical_plan::aggregates::order::GroupCompletion; use datafusion_physical_plan::aggregates::{ - AggregateExec, AggregateMode, PhysicalGroupBy, + AggregateExec, AggregateMode, GroupCompletionMode, PhysicalGroupBy, }; use datafusion_physical_plan::test::TestMemoryExec; -use datafusion_physical_plan::{ExecutionPlan, InputOrderMode, collect}; +use datafusion_physical_plan::{ExecutionPlan, collect}; use tokio::runtime::Runtime; const ROWS: usize = 131_072; @@ -112,8 +112,10 @@ fn grouping(c: &mut Criterion) { let values: Box = if selected { new_group_values( Arc::clone(&schema), - &GroupOrdering::try_new(&InputOrderMode::Sorted) - .unwrap(), + &GroupCompletion::try_new( + &GroupCompletionMode::Full, + ) + .unwrap(), ) .unwrap() } else { @@ -158,7 +160,7 @@ fn check_case(schema: &SchemaRef, batches: &[Vec]) -> (usize, usize) { let mut hashed = GroupValuesColumn::::try_new(Arc::clone(schema)).unwrap(); let mut selected = new_group_values( Arc::clone(schema), - &GroupOrdering::try_new(&InputOrderMode::Sorted).unwrap(), + &GroupCompletion::try_new(&GroupCompletionMode::Full).unwrap(), ) .unwrap(); let mut expected = Vec::new(); @@ -188,7 +190,7 @@ fn aggregate_plan( schema: &SchemaRef, keys: Vec>, sort_columns: &[&str], - input_order_mode: &InputOrderMode, + group_completion_mode: &GroupCompletionMode, ) -> Arc { let mut fields = schema.fields().to_vec(); fields.push(Arc::new(Field::new("v", DataType::Int64, false))); @@ -232,7 +234,7 @@ fn aggregate_plan( schema, ) .unwrap(); - assert_eq!(plan.input_order_mode(), input_order_mode); + assert_eq!(plan.group_completion_mode(), group_completion_mode); Arc::new(plan) } @@ -246,7 +248,7 @@ fn aggregation(c: &mut Criterion) { for run_length in [1, 8, 128, 8192] { let (schema, keys) = inputs(run_length, 8192, strings); let plan = - aggregate_plan(&schema, keys, &["a", "b"], &InputOrderMode::Sorted); + aggregate_plan(&schema, keys, &["a", "b"], &GroupCompletionMode::Full); let name = format!("{}_run{run_length}", if strings { "string" } else { "int" }); group.bench_function(name, |b| { @@ -291,7 +293,7 @@ fn partially_ordered_aggregation(c: &mut Criterion) { &schema, keys, &["a"], - &InputOrderMode::PartiallySorted(vec![0]), + &GroupCompletionMode::Partial(vec![0]), ); let runtime = Runtime::new().unwrap(); let mut group = c.benchmark_group("partially_ordered_aggregate_exec"); diff --git a/datafusion/physical-plan/benches/partial_ordering.rs b/datafusion/physical-plan/benches/partial_ordering.rs index bdadd6274b75e..ef90904ca1a76 100644 --- a/datafusion/physical-plan/benches/partial_ordering.rs +++ b/datafusion/physical-plan/benches/partial_ordering.rs @@ -18,7 +18,7 @@ use std::sync::Arc; use arrow::array::{ArrayRef, Int32Array}; -use datafusion_physical_plan::aggregates::order::GroupOrderingPartial; +use datafusion_physical_plan::aggregates::order::GroupCompletionPartial; use criterion::{Criterion, criterion_group, criterion_main}; @@ -34,20 +34,20 @@ fn create_test_arrays(num_columns: usize) -> Vec { .collect() } fn bench_new_groups(c: &mut Criterion) { - let mut group = c.benchmark_group("group_ordering_partial"); + let mut group = c.benchmark_group("group_completion_partial"); - // Test with 1, 2, 4, and 8 order indices + // Test with 1, 2, 4, and 8 grouping indices for num_columns in [1, 2, 4, 8] { - let order_indices: Vec = (0..num_columns).collect(); + let grouping_indices: Vec = (0..num_columns).collect(); - group.bench_function(format!("order_indices_{num_columns}"), |b| { + group.bench_function(format!("grouping_indices_{num_columns}"), |b| { let batch_group_values = create_test_arrays(num_columns); let group_indices: Vec = (0..BATCH_SIZE).collect(); b.iter(|| { - let mut ordering = - GroupOrderingPartial::try_new(order_indices.clone()).unwrap(); - ordering + let mut completion = + GroupCompletionPartial::try_new(grouping_indices.clone()).unwrap(); + completion .new_groups(&batch_group_values, &group_indices, BATCH_SIZE) .unwrap(); }); diff --git a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/ordered_final_table.rs b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_final_table.rs similarity index 76% rename from datafusion/physical-plan/src/aggregates/aggregate_hash_table/ordered_final_table.rs rename to datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_final_table.rs index 9cf6497e4843e..0e41489a3ec7b 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/ordered_final_table.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_final_table.rs @@ -15,9 +15,9 @@ // specific language governing permissions and limitations // under the License. -//! Aggregate table for final aggregation when partial-state input is ordered. +//! Aggregate table for final aggregation when partial-state input is clustered. //! -//! See comments in [`super::ordered_partial_table`] for details. +//! See comments in [`super::clustered_partial_table`] for details. use std::sync::Arc; @@ -25,12 +25,12 @@ use arrow::datatypes::SchemaRef; use arrow::record_batch::RecordBatch; use datafusion_common::Result; -use crate::InputOrderMode; use crate::aggregates::aggregate_hash_table::FinalMarker; +use crate::aggregates::order::GroupCompletionMode; use crate::aggregates::{AggregateExec, AggregateMode, group_values::AccumulatorPhase}; use super::common::HashAggregateAccumulator; -use super::common_ordered::{OrderedAggregateTable, OrderedAggregateTableMetrics}; +use super::common_clustered::{ClusteredAggregateTable, ClusteredAggregateTableMetrics}; /// Implementation specific to final aggregation, where the table stores partial /// aggregate states and the input rows are also partial states. @@ -40,28 +40,28 @@ use super::common_ordered::{OrderedAggregateTable, OrderedAggregateTableMetrics} /// - Aggregate table stores: `k, sum(x), count(x)` /// - Input rows: `k, sum(x), count(x)` /// -/// See comments at [`OrderedAggregateTable`] for details. -impl OrderedAggregateTable { - pub(in crate::aggregates) fn new_with_input_order( +/// See comments at [`ClusteredAggregateTable`] for details. +impl ClusteredAggregateTable { + pub(in crate::aggregates) fn new_with_group_completion( agg: &AggregateExec, input_schema: &SchemaRef, output_schema: SchemaRef, - input_order_mode: &InputOrderMode, - metrics: OrderedAggregateTableMetrics, + group_completion_mode: &GroupCompletionMode, + metrics: ClusteredAggregateTableMetrics, ) -> Result { Self::new_for_mode( agg, input_schema, output_schema, Arc::clone(input_schema), - input_order_mode, + group_completion_mode, &AggregateMode::Final, vec![None; agg.aggr_expr().len()], metrics, ) } - /// Merges one partial-state input batch and updates ordering information for + /// Merges one partial-state input batch and updates completion state for /// any newly observed groups. pub(in crate::aggregates) fn aggregate_batch( &mut self, @@ -69,7 +69,7 @@ impl OrderedAggregateTable { ) -> Result<()> { let evaluated_batch = self.evaluate_batch(batch)?; // `PhysicalGroupBy::as_final()` removes grouping sets while planning - // final aggregation, so final ordered aggregation sees one grouping. + // final aggregation, so clustered final aggregation sees one grouping. debug_assert_eq!(evaluated_batch.grouping_set_args.len(), 1); self.aggregate_evaluated_batch( &evaluated_batch, @@ -78,8 +78,8 @@ impl OrderedAggregateTable { ) } - /// Materializes final results for all groups proven complete by the input - /// ordering, leaving the active ordered-key range in the table. + /// Materializes final results for all completed groups, leaving + /// the active contiguous-key range in the table. /// /// Returns None if there are no completed groups. pub(in crate::aggregates) fn take_completed_result_batch( @@ -88,7 +88,7 @@ impl OrderedAggregateTable { if self.is_empty() { return Ok(None); } - let Some(emit_to) = self.group_ordering().emit_to() else { + let Some(emit_to) = self.group_completion().emit_to() else { return Ok(None); }; self.materialize_groups( diff --git a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/ordered_partial_table.rs b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_partial_table.rs similarity index 76% rename from datafusion/physical-plan/src/aggregates/aggregate_hash_table/ordered_partial_table.rs rename to datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_partial_table.rs index 3756c4b8e7868..db3b5f91274fa 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/ordered_partial_table.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_partial_table.rs @@ -15,13 +15,13 @@ // specific language governing permissions and limitations // under the License. -//! Aggregate table for partial aggregation when input is ordered by group keys. +//! Aggregate table for partial aggregation when input is clustered by group keys. //! -//! See the [`super::common_ordered`] comments for the high-level ideas. +//! See the [`super::common_clustered`] comments for the high-level ideas. //! -//! This operator handles input that is ordered by group keys: -//! - Fully ordered: `GROUP BY a, b`, input is `ORDER BY a, b` -//! - Partially ordered: `GROUP BY a, b`, input is `ORDER BY a` +//! Ordering can establish either group-completion mode: +//! - Full: `GROUP BY a, b`, input is `ORDER BY a, b` +//! - Partial: `GROUP BY a, b`, input is `ORDER BY a` //! //! When a group key combination is exhausted, this table eagerly flushes the //! completed groups to improve memory efficiency. @@ -41,7 +41,7 @@ use crate::aggregates::{ }; use super::common::HashAggregateAccumulator; -use super::common_ordered::{OrderedAggregateTable, OrderedAggregateTableMetrics}; +use super::common_clustered::{ClusteredAggregateTable, ClusteredAggregateTableMetrics}; /// Implementation specific to partial aggregation, where the table stores /// partial aggregate states and the input rows are raw rows. @@ -51,8 +51,8 @@ use super::common_ordered::{OrderedAggregateTable, OrderedAggregateTableMetrics} /// - Aggregate table stores: `k, sum(x), count(x)` /// - Input rows: `k, x` /// -/// See comments at [`OrderedAggregateTable`] for details. -impl OrderedAggregateTable { +/// See comments at [`ClusteredAggregateTable`] for details. +impl ClusteredAggregateTable { pub(in crate::aggregates) fn new( agg: &AggregateExec, partition: usize, @@ -60,20 +60,20 @@ impl OrderedAggregateTable { ) -> Result { let input_schema = agg.input().schema(); let state_schema = Arc::clone(&output_schema); - let metrics = OrderedAggregateTableMetrics::new(agg, partition); + let metrics = ClusteredAggregateTableMetrics::new(agg, partition); Self::new_for_mode( agg, &input_schema, output_schema, state_schema, - &agg.input_order_mode, + &agg.group_completion_mode, &AggregateMode::Partial, agg.filter_expr().to_vec(), metrics, ) } - /// Aggregates one raw input batch and updates ordering information for any + /// Aggregates one raw input batch and updates completion state for any /// newly observed groups. pub(in crate::aggregates) fn aggregate_batch( &mut self, @@ -87,15 +87,15 @@ impl OrderedAggregateTable { ) } - /// Materializes all groups proven complete by the input ordering, leaving - /// the active ordered-key range in the table. + /// Materializes all completed groups, leaving + /// the active contiguous-key range in the table. pub(in crate::aggregates) fn take_completed_state_batch( &mut self, ) -> Result> { if self.is_empty() { return Ok(None); } - let Some(emit_to) = self.group_ordering().emit_to() else { + let Some(emit_to) = self.group_completion().emit_to() else { return Ok(None); }; self.materialize_groups( diff --git a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/ordered_single_table.rs b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_single_table.rs similarity index 79% rename from datafusion/physical-plan/src/aggregates/aggregate_hash_table/ordered_single_table.rs rename to datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_single_table.rs index 89859bc036d83..108a21a5d063b 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/ordered_single_table.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_single_table.rs @@ -15,9 +15,9 @@ // specific language governing permissions and limitations // under the License. -//! Aggregate table for single aggregation when raw input is ordered. +//! Aggregate table for single aggregation when raw input is clustered. //! -//! See comments in [`super::ordered_partial_table`] for details. +//! See comments in [`super::clustered_partial_table`] for details. use arrow::datatypes::SchemaRef; use arrow::record_batch::RecordBatch; @@ -27,7 +27,7 @@ use crate::aggregates::aggregate_hash_table::SingleMarker; use crate::aggregates::{AggregateExec, AggregateMode, group_values::AccumulatorPhase}; use super::common::HashAggregateAccumulator; -use super::common_ordered::{OrderedAggregateTable, OrderedAggregateTableMetrics}; +use super::common_clustered::{ClusteredAggregateTable, ClusteredAggregateTableMetrics}; /// Implementation specific to single aggregation, where the table stores final /// aggregate values and the input rows are raw rows. @@ -37,8 +37,8 @@ use super::common_ordered::{OrderedAggregateTable, OrderedAggregateTableMetrics} /// - Aggregate table stores: `k, avg(x)` /// - Input rows: `k, x` /// -/// See comments at [`OrderedAggregateTable`] for details. -impl OrderedAggregateTable { +/// See comments at [`ClusteredAggregateTable`] for details. +impl ClusteredAggregateTable { pub(in crate::aggregates) fn new( agg: &AggregateExec, partition: usize, @@ -51,20 +51,20 @@ impl OrderedAggregateTable { )); let input_schema = agg.input().schema(); - let metrics = OrderedAggregateTableMetrics::new(agg, partition); + let metrics = ClusteredAggregateTableMetrics::new(agg, partition); Self::new_for_mode( agg, &input_schema, output_schema, state_schema, - &agg.input_order_mode, + &agg.group_completion_mode, &agg.mode, agg.filter_expr().to_vec(), metrics, ) } - /// Aggregates one raw input batch and updates ordering information for any + /// Aggregates one raw input batch and updates completion state for any /// newly observed groups. pub(in crate::aggregates) fn aggregate_batch( &mut self, @@ -78,8 +78,8 @@ impl OrderedAggregateTable { ) } - /// Materializes final results for all groups proven complete by the input - /// ordering, leaving the active ordered-key range in the table. + /// Materializes final results for all completed groups, leaving + /// the active contiguous-key range in the table. /// /// Returns None if there are no completed groups. pub(in crate::aggregates) fn take_completed_result_batch( @@ -88,7 +88,7 @@ impl OrderedAggregateTable { if self.is_empty() { return Ok(None); } - let Some(emit_to) = self.group_ordering().emit_to() else { + let Some(emit_to) = self.group_completion().emit_to() else { return Ok(None); }; self.materialize_groups( diff --git a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common.rs b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common.rs index b25caf815eed1..5f33500140c04 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common.rs @@ -38,7 +38,7 @@ use crate::aggregates::group_values::{ AccumulatorPhase, AggregateAccumulatorMetrics, AggregateArgumentMetrics, GroupByMetrics, GroupValues, new_group_values, }; -use crate::aggregates::order::GroupOrdering; +use crate::aggregates::order::GroupCompletion; use crate::aggregates::{ AggregateExec, PhysicalGroupBy, aggregate_expressions, evaluate_group_by, group_id_array, max_duplicate_ordinal, @@ -180,7 +180,7 @@ impl AggregateHashTable { .collect::>()?; let group_schema = agg.group_by().group_schema(&input_schema)?; - let group_values = new_group_values(group_schema, &GroupOrdering::None)?; + let group_values = new_group_values(group_schema, &GroupCompletion::None)?; Ok(Self { group_by_metrics: metrics.group_by, diff --git a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common_ordered.rs b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common_clustered.rs similarity index 84% rename from datafusion/physical-plan/src/aggregates/aggregate_hash_table/common_ordered.rs rename to datafusion/physical-plan/src/aggregates/aggregate_hash_table/common_clustered.rs index 86c052009faea..c719c868c9abc 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common_ordered.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common_clustered.rs @@ -15,8 +15,7 @@ // specific language governing permissions and limitations // under the License. -//! Common utilities for aggregate tables used in aggregations that inputs are ordered -//! by the groups. +//! Common utilities for aggregate tables with input clustered by group keys. use std::marker::PhantomData; use std::sync::Arc; @@ -27,13 +26,12 @@ use datafusion_common::Result; use datafusion_execution::memory_pool::proxy::VecAllocExt; use datafusion_expr::{AggregateMetrics, EmitTo}; -use crate::InputOrderMode; use crate::PhysicalExpr; use crate::aggregates::group_values::{ AccumulatorPhase, AggregateAccumulatorMetrics, AggregateArgumentMetrics, GroupByMetrics, GroupValues, new_group_values, }; -use crate::aggregates::order::GroupOrdering; +use crate::aggregates::order::{GroupCompletion, GroupCompletionMode}; use crate::aggregates::{ AggregateExec, AggregateMode, PhysicalGroupBy, aggregate_expressions, evaluate_group_by, @@ -46,14 +44,14 @@ use super::common::{ }; #[derive(Clone)] -pub(in crate::aggregates) struct OrderedAggregateTableMetrics { +pub(in crate::aggregates) struct ClusteredAggregateTableMetrics { pub(super) group_by: GroupByMetrics, pub(super) aggregate_arguments: AggregateArgumentMetrics, pub(super) accumulator: Arc, pub(super) submetrics: Vec>, } -impl OrderedAggregateTableMetrics { +impl ClusteredAggregateTableMetrics { pub(in crate::aggregates) fn new(agg: &AggregateExec, partition: usize) -> Self { let metrics = AggregateTableMetrics::new(agg, partition); Self { @@ -76,20 +74,20 @@ impl OrderedAggregateTableMetrics { } } -/// Aggregate table shared by the ordered single, partial and final paths. +/// Aggregate table shared by the clustered single, partial and final paths. /// -/// # Ordering optimization +/// # Group completion optimization /// -/// The table consumes input batches while `GroupOrdering` tracks which groups +/// The table consumes input batches while [`GroupCompletion`] tracks which groups /// are proven complete. Completed groups can be emitted before the input stream -/// ends, which keeps memory bounded by the active ordered key range. +/// ends, so completed groups no longer occupy the table. /// /// # Single, partial and final variant difference /// /// The partial and final aggregate tables implement the two stages of grouped /// aggregation, while the single aggregate table implements both stages in one /// table. See -/// [`OrderedPartialAggregateStream`](crate::aggregates::ordered_partial_stream::OrderedPartialAggregateStream) +/// [`ClusteredPartialAggregateStream`](crate::aggregates::clustered_partial_stream::ClusteredPartialAggregateStream) /// for the high-level plan shape. /// /// Example: `AVG(v) FILTER (WHERE v>0) GROUP BY k` @@ -111,15 +109,15 @@ impl OrderedAggregateTableMetrics { /// /// # Marker Type /// -/// `OrderedAggrMode` selects the aggregate semantics. For example, -/// `OrderedAggregateTable::::new(...)` consumes raw rows +/// `AggrMode` selects the aggregate semantics. For example, +/// `ClusteredAggregateTable::::new(...)` consumes raw rows /// and emits partial states, while -/// `OrderedAggregateTable::::new_with_input_order(...)` +/// `ClusteredAggregateTable::::new_with_group_completion(...)` /// consumes partial states and emits final values. /// /// Shared methods live on `impl`; single/partial/final behavior lives on /// marker-specific impls. -pub(in crate::aggregates) struct OrderedAggregateTable { +pub(in crate::aggregates) struct ClusteredAggregateTable { /// Output schema: group columns followed by aggregate state or final values. pub(super) output_schema: SchemaRef, @@ -139,26 +137,26 @@ pub(in crate::aggregates) struct OrderedAggregateTable { /// Optional internal metrics owned by each aggregate expression. pub(super) aggregate_submetrics: Vec>, - /// Group keys, ordering state, and accumulator states. - pub(super) buffer: OrderedAggregateTableBuffer, + /// Group keys, completion state, and accumulator states. + pub(super) buffer: ClusteredAggregateTableBuffer, - _mode: PhantomData, + _mode: PhantomData, } -/// Buffer for the ordered aggregate table's group keys and accumulator states. +/// Buffer for the clustered aggregate table's group keys and accumulator states. /// -/// It accumulates input during aggregation and emits output rows as soon as the -/// input ordering proves those groups are complete. +/// It accumulates input during aggregation and emits output rows as soon as +/// groups are known to be complete. /// -/// [`GroupOrdering`] tracks when and how to do early emit. +/// [`GroupCompletion`] tracks when and how to do early emit. /// [`GroupValues`] stores the physical group-key layout, while /// [`datafusion_expr::GroupsAccumulator`] stores per-group aggregate state. -pub(super) struct OrderedAggregateTableBuffer { +pub(super) struct ClusteredAggregateTableBuffer { /// GROUP BY expressions evaluated against input batches. pub(super) group_by: Arc, - /// Tracks how far ordered input allows this table to drain safely. - pub(super) group_ordering: GroupOrdering, + /// Tracks which groups are complete and can be emitted safely. + pub(super) group_completion: GroupCompletion, /// Interned group keys, in the same group-id order used by accumulators. pub(super) group_values: Box, @@ -174,24 +172,24 @@ pub(super) struct OrderedAggregateTableBuffer { } /// Methods shared by all aggregate modes -impl OrderedAggregateTable { +impl ClusteredAggregateTable { #[expect( clippy::too_many_arguments, - reason = "keeps ordered single, partial and final table construction explicit" + reason = "keeps clustered single, partial and final table construction explicit" )] pub(super) fn new_for_mode( agg: &AggregateExec, input_schema: &SchemaRef, output_schema: SchemaRef, state_schema: SchemaRef, - input_order_mode: &InputOrderMode, + group_completion_mode: &GroupCompletionMode, aggregate_mode: &AggregateMode, filters: Vec>>, - metrics: OrderedAggregateTableMetrics, + metrics: ClusteredAggregateTableMetrics, ) -> Result { - let group_ordering = GroupOrdering::try_new(input_order_mode)?; + let group_completion = GroupCompletion::try_new(group_completion_mode)?; let group_schema = agg.group_by().group_schema(input_schema)?; - let group_values = new_group_values(group_schema, &group_ordering)?; + let group_values = new_group_values(group_schema, &group_completion)?; let aggregate_arguments = aggregate_expressions( agg.aggr_expr(), aggregate_mode, @@ -223,9 +221,9 @@ impl OrderedAggregateTable { aggregate_argument_metrics: metrics.aggregate_arguments, aggregate_accumulator_metrics: metrics.accumulator, aggregate_submetrics: metrics.submetrics, - buffer: OrderedAggregateTableBuffer { + buffer: ClusteredAggregateTableBuffer { group_by: Arc::clone(agg.group_by()), - group_ordering, + group_completion, group_values, group_indices: vec![], accumulators, @@ -268,15 +266,15 @@ impl OrderedAggregateTable { /// Called after the input stream is exhausted and the last batch has been /// aggregated. /// - /// Updates the internal `GroupOrdering` so it can continue emitting until + /// Updates the internal [`GroupCompletion`] so it can continue emitting until /// the buffer is empty. pub(in crate::aggregates) fn input_done(&mut self) { - self.buffer.group_ordering.input_done(); + self.buffer.group_completion.input_done(); } - /// Returns the ordering state used to decide how memory pressure is handled. - pub(in crate::aggregates) fn group_ordering(&self) -> &GroupOrdering { - &self.buffer.group_ordering + /// Returns the completion state used to decide how memory pressure is handled. + pub(in crate::aggregates) fn group_completion(&self) -> &GroupCompletion { + &self.buffer.group_completion } /// Number of groups currently buffered. @@ -297,12 +295,12 @@ impl OrderedAggregateTable { .map(|acc| acc.size()) .sum::() + self.buffer.group_values.size() - + self.buffer.group_ordering.size() + + self.buffer.group_completion.size() + self.buffer.group_indices.allocated_size() } - pub(in crate::aggregates) fn metrics(&self) -> OrderedAggregateTableMetrics { - OrderedAggregateTableMetrics { + pub(in crate::aggregates) fn metrics(&self) -> ClusteredAggregateTableMetrics { + ClusteredAggregateTableMetrics { group_by: self.group_by_metrics.clone(), aggregate_arguments: self.aggregate_argument_metrics.clone(), accumulator: Arc::clone(&self.aggregate_accumulator_metrics), @@ -311,9 +309,9 @@ impl OrderedAggregateTable { } /// Takes every intermediate aggregate state and resets the table so it can - /// continue with a new ordered input segment. + /// continue with a new clustered input segment. /// - /// Unlike normal ordered emission, this operation is allowed to take the + /// Unlike normal group-completion emission, this operation is allowed to take the /// active (incomplete) groups. Partial aggregation can pass those states to /// its final stage, while single and final aggregation sort and spill them /// before replay. @@ -346,7 +344,7 @@ impl OrderedAggregateTable { self.buffer.group_values.clear_shrink(0); self.buffer.group_indices.clear(); self.buffer.group_indices.shrink_to_fit(); - self.buffer.group_ordering.reset(); + self.buffer.group_completion.reset(); Ok(Some(batch)) } @@ -374,7 +372,7 @@ impl OrderedAggregateTable { .intern(group_values, &mut self.buffer.group_indices)?; let total_num_groups = self.buffer.group_values.len(); if total_num_groups > starting_num_groups { - self.buffer.group_ordering.new_groups( + self.buffer.group_completion.new_groups( group_values, &self.buffer.group_indices, total_num_groups, @@ -418,10 +416,10 @@ impl OrderedAggregateTable { let accumulator_metrics = Arc::clone(&self.aggregate_accumulator_metrics); let output = self.group_by_metrics.time_emitting(|| { let mut output = self.buffer.group_values.emit(emit_to)?; - // `EmitTo::All` is only used after `input_done`, when the ordering + // `EmitTo::All` is only used after `input_done`, when the completion // state no longer tracks group indexes. if let EmitTo::First(n) = emit_to { - self.buffer.group_ordering.remove_groups(n); + self.buffer.group_completion.remove_groups(n); } for (idx, acc) in self.buffer.accumulators.iter_mut().enumerate() { diff --git a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/mod.rs b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/mod.rs index 960b7498f1eb2..a356ea5c79d10 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/mod.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/mod.rs @@ -15,12 +15,12 @@ // specific language governing permissions and limitations // under the License. +mod clustered_final_table; +mod clustered_partial_table; +mod clustered_single_table; mod common; -mod common_ordered; +mod common_clustered; mod final_table; -mod ordered_final_table; -mod ordered_partial_table; -mod ordered_single_table; mod partial_reduce_table; mod partial_table; mod single_table; @@ -101,7 +101,9 @@ pub(super) use common::{ AggregateHashTable, FinalMarker, PartialMarker, PartialReduceMarker, PartialSkipMarker, SingleMarker, create_group_accumulator, }; -pub(super) use common_ordered::{OrderedAggregateTable, OrderedAggregateTableMetrics}; +pub(super) use common_clustered::{ + ClusteredAggregateTable, ClusteredAggregateTableMetrics, +}; #[cfg(test)] mod tests { diff --git a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/partial_table.rs b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/partial_table.rs index 397b766f41697..b61b954e4d527 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/partial_table.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/partial_table.rs @@ -23,7 +23,7 @@ use arrow::record_batch::RecordBatch; use datafusion_common::{Result, assert_eq_or_internal_err}; use crate::aggregates::group_values::{AccumulatorPhase, new_group_values}; -use crate::aggregates::order::GroupOrdering; +use crate::aggregates::order::GroupCompletion; use crate::aggregates::{AggregateExec, evaluate_group_by}; use super::common::{ @@ -78,7 +78,7 @@ impl AggregateHashTable { ) -> Result> { let state = self.state.building(); let group_schema = state.group_by.group_schema(&self.input_schema)?; - let group_values = new_group_values(group_schema, &GroupOrdering::None)?; + let group_values = new_group_values(group_schema, &GroupCompletion::None)?; let accumulators = state .accumulators .iter() diff --git a/datafusion/physical-plan/src/aggregates/ordered_final_stream.rs b/datafusion/physical-plan/src/aggregates/clustered_final_stream.rs similarity index 89% rename from datafusion/physical-plan/src/aggregates/ordered_final_stream.rs rename to datafusion/physical-plan/src/aggregates/clustered_final_stream.rs index daa5a85323a85..33c7a87791a1c 100644 --- a/datafusion/physical-plan/src/aggregates/ordered_final_stream.rs +++ b/datafusion/physical-plan/src/aggregates/clustered_final_stream.rs @@ -15,7 +15,8 @@ // specific language governing permissions and limitations // under the License. -//! Final aggregate stream for ordered partial-state input. +//! Final aggregate stream for partial-state input with group-completion +//! guarantees. use std::sync::Arc; @@ -28,24 +29,25 @@ use futures::stream::StreamExt; use super::AggregateExec; use super::aggregate_hash_table::{ - FinalMarker, OrderedAggregateTable, OrderedAggregateTableMetrics, + ClusteredAggregateTable, ClusteredAggregateTableMetrics, FinalMarker, }; +use super::order::GroupCompletionMode; use super::spill::AggregateSpill; +use crate::SendableRecordBatchStream; use crate::aggregates::AggregateMode; use crate::metrics::{BaselineMetrics, SpillMetrics}; use crate::stream::{ObservedStream, RecordBatchStreamAdapter}; -use crate::{InputOrderMode, SendableRecordBatchStream}; -/// Final aggregate stream for `InputOrderMode::Sorted` and -/// `InputOrderMode::PartiallySorted`. +/// Final aggregate stream for [`GroupCompletionMode::Partial`] and +/// [`GroupCompletionMode::Full`]. /// -/// See comments at [`super::ordered_partial_stream::OrderedPartialAggregateStream`] for details. +/// See comments at [`super::clustered_partial_stream::ClusteredPartialAggregateStream`] for details. /// /// # Spilling /// -/// This section is only for implementation notes, for background, see [`super::ordered_partial_stream::OrderedPartialAggregateStream`] +/// This section is only for implementation notes, for background, see [`super::clustered_partial_stream::ClusteredPartialAggregateStream`] /// -/// For partially sorted input, spilling works as follows: +/// For partial group completion, spilling works as follows: /// /// - Reserve the table footprint plus one `u32` sort index per buffered group. The /// extra index array is used in later sorting before spilling. @@ -55,13 +57,13 @@ use crate::{InputOrderMode, SendableRecordBatchStream}; /// batch and full index remain live until the run is written. /// - After input ends, merge the sorted runs and replay them through a fully /// ordered final aggregate stream. -pub(crate) struct OrderedFinalAggregateStream { +pub(crate) struct ClusteredFinalAggregateStream { reservation: MemoryReservation, - context: OrderedFinalAggregateContext, + context: ClusteredFinalAggregateContext, stage: ExecutionStage, } -/// Execution stages described in [`OrderedFinalAggregateStream::into_stream`]. +/// Execution stages described in [`ClusteredFinalAggregateStream::into_stream`]. enum ExecutionStage { Aggregating(Aggregating), Outputting(Outputting), @@ -70,8 +72,8 @@ enum ExecutionStage { struct Aggregating { input: SendableRecordBatchStream, - table: OrderedAggregateTable, - /// None when temporary files are disabled or all group keys are ordered. + table: ClusteredAggregateTable, + /// None when temporary files are disabled or group completion is full. spill_context: Option>, } @@ -83,13 +85,13 @@ struct Outputting { } /// Immutable execution context shared by aggregation and output emission. -struct OrderedFinalAggregateContext { +struct ClusteredFinalAggregateContext { schema: SchemaRef, batch_size: usize, baseline_metrics: BaselineMetrics, } -impl OrderedFinalAggregateStream { +impl ClusteredFinalAggregateStream { pub fn new( agg: &AggregateExec, context: &Arc, @@ -99,10 +101,10 @@ impl OrderedFinalAggregateStream { agg.mode, AggregateMode::Final | AggregateMode::FinalPartitioned )); - debug_assert_ne!(agg.input_order_mode, InputOrderMode::Linear); + debug_assert_ne!(agg.group_completion_mode, GroupCompletionMode::None); let input = agg.input.execute(partition, Arc::clone(context))?; - Self::new_with_input(agg, context, partition, input, &agg.input_order_mode) + Self::new_with_input(agg, context, partition, input, &agg.group_completion_mode) } pub(in crate::aggregates) fn new_with_input( @@ -110,15 +112,15 @@ impl OrderedFinalAggregateStream { context: &Arc, partition: usize, input: SendableRecordBatchStream, - input_order_mode: &InputOrderMode, + group_completion_mode: &GroupCompletionMode, ) -> Result { let baseline_metrics = BaselineMetrics::new(&agg.metrics, partition); - let metrics = OrderedAggregateTableMetrics::new(agg, partition); + let metrics = ClusteredAggregateTableMetrics::new(agg, partition); let spill_metrics = SpillMetrics::new(&agg.metrics, partition); let reservation = - MemoryConsumer::new(format!("OrderedFinalAggregateStream[{partition}]")) - // HACK: Technically, fully ordered aggregate is a non-spillable - // consumer, since it uses bounded memory. There is a known race + MemoryConsumer::new(format!("ClusteredFinalAggregateStream[{partition}]")) + // HACK: Full group completion uses a non-spilling execution + // path. There is a known race // condition bug, and we set it to spillable to let it have larger // memory budget to suppress the bug. // Bug issue: https://github.com/apache/datafusion/issues/17334 @@ -129,7 +131,7 @@ impl OrderedFinalAggregateStream { context, partition, input, - input_order_mode, + group_completion_mode, baseline_metrics, metrics, Some(spill_metrics), @@ -149,9 +151,9 @@ impl OrderedFinalAggregateStream { context: &Arc, partition: usize, input: SendableRecordBatchStream, - input_order_mode: &InputOrderMode, + group_completion_mode: &GroupCompletionMode, baseline_metrics: BaselineMetrics, - metrics: OrderedAggregateTableMetrics, + metrics: ClusteredAggregateTableMetrics, spill_metrics: Option, reservation: MemoryReservation, ) -> Result { @@ -159,25 +161,27 @@ impl OrderedFinalAggregateStream { agg.mode, AggregateMode::Final | AggregateMode::FinalPartitioned )); - debug_assert_ne!(*input_order_mode, InputOrderMode::Linear); + debug_assert_ne!(*group_completion_mode, GroupCompletionMode::None); let schema = Arc::clone(&agg.schema); let input_schema = input.schema(); let batch_size = context.session_config().batch_size(); - let can_spill = matches!(input_order_mode, InputOrderMode::PartiallySorted(_)) + let can_spill = matches!(group_completion_mode, GroupCompletionMode::Partial(_)) && context.runtime_env().disk_manager.tmp_files_enabled(); let spill_context = if can_spill { let Some(spill_metrics) = spill_metrics else { - return internal_err!("Spillable ordered final stream requires metrics"); + return internal_err!( + "Spillable clustered final stream requires metrics" + ); }; Some(Box::new(AggregateSpill::try_new( - "OrderedFinalAggregateSpill", + "ClusteredFinalAggregateSpill", agg, context, partition, batch_size, - input_order_mode, + group_completion_mode, &input_schema, spill_metrics, )?)) @@ -185,11 +189,11 @@ impl OrderedFinalAggregateStream { None }; - let table = OrderedAggregateTable::::new_with_input_order( + let table = ClusteredAggregateTable::::new_with_group_completion( agg, &input_schema, Arc::clone(&schema), - input_order_mode, + group_completion_mode, metrics, )?; @@ -198,7 +202,7 @@ impl OrderedFinalAggregateStream { Ok(Self { reservation, - context: OrderedFinalAggregateContext { + context: ClusteredFinalAggregateContext { schema, batch_size, baseline_metrics, @@ -211,9 +215,9 @@ impl OrderedFinalAggregateStream { }) } - /// Entry point for the ordered final aggregate execution stages. + /// Entry point for the clustered final aggregate execution stages. /// - /// See [`OrderedFinalAggregateStream`] for high-level ideas. + /// See [`ClusteredFinalAggregateStream`] for high-level ideas. /// /// # Stage transition graph: /// @@ -248,9 +252,9 @@ impl OrderedFinalAggregateStream { /// /// ### Incremental output /// - /// See the [ordered partial aggregate notes] for details. + /// See the [clustered partial aggregate notes] for details. /// - /// [ordered partial aggregate notes]: super::ordered_partial_stream::OrderedPartialAggregateStream::into_stream + /// [clustered partial aggregate notes]: super::clustered_partial_stream::ClusteredPartialAggregateStream::into_stream /// /// ## Transition Edges /// @@ -259,7 +263,7 @@ impl OrderedFinalAggregateStream { /// - If memory fits and no groups are complete, continue reading input. /// - If OOM, spill. /// 3. Prepare output: - /// - Before any spill, ordering proves a prefix complete: materialize the + /// - Before any spill, a prefix of groups is complete: materialize the /// entire prefix once, retaining the input and active groups to resume /// aggregation. /// - At EOF without spills, materialize all remaining results and prepare @@ -312,7 +316,7 @@ impl Aggregating { fn reservation_size(&self) -> usize { let table_size = self.table.memory_size(); if self.spill_context.is_some() { - // See `OrderedFinalAggregateStream` for the spill memory estimate. + // See `ClusteredFinalAggregateStream` for the spill memory estimate. table_size .saturating_add(self.table.num_groups().saturating_mul(size_of::())) } else { @@ -323,7 +327,7 @@ impl Aggregating { /// Merges partial states until final results are ready or spill replay begins. async fn handle_stage( mut self, - context: &OrderedFinalAggregateContext, + context: &ClusteredFinalAggregateContext, reservation: &MemoryReservation, ) -> Result> { let elapsed_compute = context.baseline_metrics.elapsed_compute(); @@ -406,7 +410,7 @@ impl Outputting { /// Emits slices of one materialized batch without touching the aggregate table. async fn handle_stage( self, - context: &OrderedFinalAggregateContext, + context: &ClusteredFinalAggregateContext, reservation: &MemoryReservation, emitter: &mut TryEmitter, ) -> Result> { @@ -454,6 +458,7 @@ impl Outputting { mod tests { use super::*; use crate::ExecutionPlan; + use crate::aggregates::GroupCompletionMode; use crate::aggregates::PhysicalGroupBy; use crate::common::collect; use crate::test::TestMemoryExec; @@ -556,8 +561,8 @@ mod tests { schema, )?; assert_eq!( - aggregate.input_order_mode(), - &InputOrderMode::PartiallySorted(vec![0]) + aggregate.group_completion_mode(), + &GroupCompletionMode::Partial(vec![0]) ); let pool: Arc = Arc::new(GreedyMemoryPool::new(limit)); @@ -582,12 +587,12 @@ mod tests { Arc::clone(&partial_schema), receiver, )); - let stream = OrderedFinalAggregateStream::new_with_input( + let stream = ClusteredFinalAggregateStream::new_with_input( &aggregate, &context, partition, input, - aggregate.input_order_mode(), + &aggregate.group_completion_mode, )?; Ok((sender, stream.into_stream())) }; diff --git a/datafusion/physical-plan/src/aggregates/ordered_partial_stream.rs b/datafusion/physical-plan/src/aggregates/clustered_partial_stream.rs similarity index 81% rename from datafusion/physical-plan/src/aggregates/ordered_partial_stream.rs rename to datafusion/physical-plan/src/aggregates/clustered_partial_stream.rs index 83a25620cd328..577ae86d18dba 100644 --- a/datafusion/physical-plan/src/aggregates/ordered_partial_stream.rs +++ b/datafusion/physical-plan/src/aggregates/clustered_partial_stream.rs @@ -15,7 +15,7 @@ // specific language governing permissions and limitations // under the License. -//! Partial aggregate stream for ordered group input. +//! Partial aggregate stream for input with group-completion guarantees. use std::sync::Arc; @@ -27,27 +27,27 @@ use datafusion_execution::{TaskContext, TryEmitter, async_try_stream}; use futures::stream::StreamExt; use super::AggregateExec; -use super::aggregate_hash_table::{OrderedAggregateTable, PartialMarker}; +use super::aggregate_hash_table::{ClusteredAggregateTable, PartialMarker}; use crate::aggregates::AggregateMode; -use crate::aggregates::order::GroupOrdering; +use crate::aggregates::order::{GroupCompletion, GroupCompletionMode}; use crate::metrics::{BaselineMetrics, MetricBuilder, SpillMetrics}; use crate::stream::{ObservedStream, RecordBatchStreamAdapter}; -use crate::{InputOrderMode, SendableRecordBatchStream, metrics}; +use crate::{SendableRecordBatchStream, metrics}; -/// Partial aggregate stream for `InputOrderMode::Sorted` and -/// `InputOrderMode::PartiallySorted`. +/// Partial aggregate stream for [`GroupCompletionMode::Partial`] and +/// [`GroupCompletionMode::Full`]. /// /// # Example /// /// SELECT k, AVG(v) FROM t GROUP BY k; /// -/// If the input is ordered by `k`, the aggregate can use ordered partial and +/// If the input is ordered by `k`, the aggregate can use clustered partial and /// final stages: /// /// ## Plan -/// AggregateExec(stage=final, ordered) +/// AggregateExec(stage=final, clustered) /// -- RepartitionExec(hash(k), preserves_order=true) -/// ---- AggregateExec(stage=partial, ordered) +/// ---- AggregateExec(stage=partial, clustered) /// /// ## Partial Stage Behavior /// Input: raw rows @@ -59,22 +59,24 @@ use crate::{InputOrderMode, SendableRecordBatchStream, metrics}; /// Output: results for all groups (for example, `AVG(x)` calculated from the /// state) /// -/// # Order-based Optimization +/// # Group Completion Optimization /// /// For the aggregation work, the hash aggregation implementation is reused. /// -/// After each input batch, check whether any groups can be emitted eagerly to -/// improve memory efficiency. For example, if the last group key seen is -/// `k = 100`, it is safe to emit all groups with keys less than 100 because the -/// input is ordered. Materialize that entire completed prefix once, then emit -/// slices of it before reading more input. This avoids repeatedly removing small -/// batches of groups and shifting the remaining hash table and accumulator state. +/// After each input batch, the group-completion mode determines whether any +/// groups can be emitted eagerly to improve memory efficiency. For example, if +/// the input is ordered by `k` and the last group key seen is `k = 100`, all +/// groups with keys less than 100 are complete. Materialize that entire completed +/// prefix once, then emit slices of it before reading more input. This avoids +/// repeatedly removing small batches of groups and shifting the remaining hash +/// table and accumulator state. /// /// # Memory Pressure and Spilling /// -/// ## Fully ordered case +/// ## Full group completion /// -/// If the input is ordered by every group key, for example: +/// Every complete grouping tuple is contiguous. Ordering by every group key is +/// one way to establish this mode, for example: /// /// - Input order: `a, b` /// - `GROUP BY`: `a, b` @@ -86,9 +88,10 @@ use crate::{InputOrderMode, SendableRecordBatchStream, metrics}; /// If a memory reservation nevertheless fails, the stream returns the error /// directly, indicating an unexpected behavior. /// -/// ## Partially ordered case +/// ## Partial group completion /// -/// If the input is ordered by only a subset of the group keys, for example: +/// Rows are contiguous for a subset of the group keys. Ordering by that subset +/// is one way to establish this mode, for example: /// /// - Input order: `a` /// - `GROUP BY`: `a, b` @@ -96,21 +99,21 @@ use crate::{InputOrderMode, SendableRecordBatchStream, metrics}; /// If one `a` value contains many distinct `b` values, the table may accumulate /// enough groups to exceed the memory limit. /// -/// - `OrderedPartialAggregateStream`: On reservation failure, it emits all current +/// - `ClusteredPartialAggregateStream`: On reservation failure, it emits all current /// intermediate states downstream and resets the table. The final stage can /// merge repeated `(a, b)` state rows, so no disk spill is required. -/// - `OrderedFinalAggregateStream`: It cannot emit incomplete final results. On +/// - `ClusteredFinalAggregateStream`: It cannot emit incomplete final results. On /// reservation failure, it sorts the current intermediate states by the complete /// group key and spills them as one run. After the input ends, it spills any /// remaining states, performs a sort-preserving merge of all runs, and feeds the -/// merged input into a fully ordered final aggregate stream. -pub(crate) struct OrderedPartialAggregateStream { +/// merged input into a fully clustered final aggregate stream. +pub(crate) struct ClusteredPartialAggregateStream { reservation: MemoryReservation, - context: OrderedPartialAggregateContext, + context: ClusteredPartialAggregateContext, stage: ExecutionStage, } -/// Execution stages described in [`OrderedPartialAggregateStream::into_stream`]. +/// Execution stages described in [`ClusteredPartialAggregateStream::into_stream`]. enum ExecutionStage { Aggregating(Aggregating), Outputting(Outputting), @@ -118,7 +121,7 @@ enum ExecutionStage { struct Aggregating { input: SendableRecordBatchStream, - table: OrderedAggregateTable, + table: ClusteredAggregateTable, } struct Outputting { @@ -130,21 +133,21 @@ struct Outputting { } /// Immutable execution context shared by aggregation and output emission. -struct OrderedPartialAggregateContext { +struct ClusteredPartialAggregateContext { schema: SchemaRef, batch_size: usize, baseline_metrics: BaselineMetrics, reduction_factor: metrics::RatioMetrics, } -impl OrderedPartialAggregateStream { +impl ClusteredPartialAggregateStream { pub fn new( agg: &AggregateExec, context: &Arc, partition: usize, ) -> Result { debug_assert_eq!(agg.mode, AggregateMode::Partial); - debug_assert_ne!(agg.input_order_mode, InputOrderMode::Linear); + debug_assert_ne!(agg.group_completion_mode, GroupCompletionMode::None); let schema = Arc::clone(&agg.schema); let input = agg.input.execute(partition, Arc::clone(context))?; @@ -157,16 +160,16 @@ impl OrderedPartialAggregateStream { .with_type(metrics::MetricType::Summary) .ratio_metrics("reduction_factor", partition); - let table = OrderedAggregateTable::::new( + let table = ClusteredAggregateTable::::new( agg, partition, Arc::clone(&schema), )?; let reservation = - MemoryConsumer::new(format!("OrderedPartialAggregateStream[{partition}]")) + MemoryConsumer::new(format!("ClusteredPartialAggregateStream[{partition}]")) .with_can_spill(matches!( - table.group_ordering(), - GroupOrdering::Partial(_) + table.group_completion(), + GroupCompletion::Partial(_) )) .register(context.memory_pool()); @@ -175,7 +178,7 @@ impl OrderedPartialAggregateStream { Ok(Self { reservation, - context: OrderedPartialAggregateContext { + context: ClusteredPartialAggregateContext { schema, batch_size, baseline_metrics, @@ -185,9 +188,9 @@ impl OrderedPartialAggregateStream { }) } - /// Entry point for the ordered partial aggregate execution stages. + /// Entry point for the clustered partial aggregate execution stages. /// - /// See [`OrderedPartialAggregateStream`] for high-level ideas. + /// See [`ClusteredPartialAggregateStream`] for high-level ideas. /// /// # Stage transition graph: /// @@ -256,10 +259,10 @@ impl OrderedPartialAggregateStream { /// 2. Aggregate one input batch. If memory fits and no groups are complete, /// continue reading input. /// 3. Prepare output: - /// - Ordering proves a prefix complete: materialize the entire prefix once, + /// - A prefix of groups is complete: materialize the entire prefix once, /// retaining the input and active groups to resume aggregation. - /// - On memory pressure with partial ordering, materialize all current - /// states instead, including incomplete groups, and reset the table. + /// - On memory pressure with partial group completion, materialize all + /// current states, including incomplete groups, and reset the table. /// - At EOF, materialize all remaining states and prepare to output. /// 4. Input was exhausted with no remaining groups, directly end. /// 5. Yield one slice without materializing the table again. Keep the shared @@ -299,10 +302,10 @@ impl OrderedPartialAggregateStream { impl Aggregating { /// Aggregates raw input and materializes one batch of partial states. /// - /// See [`OrderedPartialAggregateStream::into_stream`] for stage transitions. + /// See [`ClusteredPartialAggregateStream::into_stream`] for stage transitions. async fn handle_stage( mut self, - context: &OrderedPartialAggregateContext, + context: &ClusteredPartialAggregateContext, reservation: &MemoryReservation, ) -> Result> { let elapsed_compute = context.baseline_metrics.elapsed_compute(); @@ -315,9 +318,9 @@ impl Aggregating { let output = match reservation.try_resize(self.table.memory_size()) { Ok(()) => self.table.take_completed_state_batch()?, Err(oom @ DataFusionError::ResourcesExhausted(_)) => { - // Partial ordering may have an unbounded active key range. + // Partial group completion may have an unbounded active key range. // The final stage can merge incomplete states emitted here. - if matches!(self.table.group_ordering(), GroupOrdering::Full(_)) { + if matches!(self.table.group_completion(), GroupCompletion::Full(_)) { return Err(oom); } let Some(batch) = self.table.take_state_batch()? else { @@ -363,11 +366,11 @@ impl Aggregating { impl Outputting { /// Emits slices of one materialized batch without touching the hash table. /// - /// See [`OrderedPartialAggregateStream::into_stream`] for stage transitions + /// See [`ClusteredPartialAggregateStream::into_stream`] for stage transitions /// and output memory accounting. async fn handle_stage( self, - context: &OrderedPartialAggregateContext, + context: &ClusteredPartialAggregateContext, reservation: &MemoryReservation, emitter: &mut TryEmitter, ) -> Result> { diff --git a/datafusion/physical-plan/src/aggregates/ordered_single_stream.rs b/datafusion/physical-plan/src/aggregates/clustered_single_stream.rs similarity index 77% rename from datafusion/physical-plan/src/aggregates/ordered_single_stream.rs rename to datafusion/physical-plan/src/aggregates/clustered_single_stream.rs index 57744dec57f12..e41c89277645d 100644 --- a/datafusion/physical-plan/src/aggregates/ordered_single_stream.rs +++ b/datafusion/physical-plan/src/aggregates/clustered_single_stream.rs @@ -15,7 +15,7 @@ // specific language governing permissions and limitations // under the License. -//! Single-stage aggregate stream for ordered raw input. +//! Single-stage aggregate stream for raw input with group-completion guarantees. use std::ops::ControlFlow; use std::sync::Arc; @@ -28,51 +28,52 @@ use datafusion_execution::TaskContext; use datafusion_execution::memory_pool::{MemoryConsumer, MemoryReservation}; use futures::stream::{Stream, StreamExt}; -use super::aggregate_hash_table::{OrderedAggregateTable, SingleMarker}; +use super::aggregate_hash_table::{ClusteredAggregateTable, SingleMarker}; +use super::order::GroupCompletionMode; use super::spill::AggregateSpill; use super::{AggregateExec, create_schema}; use crate::aggregates::AggregateMode; use crate::metrics::{BaselineMetrics, RecordOutput, SpillMetrics}; use crate::stream::EmptyRecordBatchStream; -use crate::{InputOrderMode, RecordBatchStream, SendableRecordBatchStream}; +use crate::{RecordBatchStream, SendableRecordBatchStream}; -/// Single aggregate stream for `InputOrderMode::Sorted` and -/// `InputOrderMode::PartiallySorted`. +/// Single aggregate stream for [`GroupCompletionMode::Partial`] and +/// [`GroupCompletionMode::Full`]. /// /// # Example /// /// SELECT k, AVG(v) FROM t GROUP BY k; /// -/// If the input is ordered by `k`, and there are existing key partitioning on group -/// by keys, the single mode aggregation with ordering optimization can be used: +/// If the input is ordered by `k` and already key-partitioned on the group-by +/// keys, clustered single aggregation can be used: /// /// ## Plan -/// AggregateExec(stage=single, ordered) +/// AggregateExec(stage=single, clustered) /// -- DataSourceExec(t) /// /// ## Single Stage Behavior /// Input: raw rows /// Output: final results for all groups (for example, `AVG(x)`) /// -/// # Order-based Optimization +/// # Group Completion Optimization /// /// For the aggregation work, the hash aggregation implementation is reused. /// -/// After each input batch, check whether any groups can be emitted eagerly to -/// improve memory efficiency. For example, if the last group key seen is -/// `k = 100`, it is safe to emit all groups with keys less than 100 because the -/// input is ordered. Materialize that entire completed prefix once, then emit -/// slices of it before reading more input. See -/// [`OrderedPartialAggregateStream::into_stream`] for why this avoids +/// After each input batch, the group-completion mode determines whether any +/// groups can be emitted eagerly to improve memory efficiency. Materialize +/// that entire completed prefix once, then emit slices of it before reading +/// more input. See +/// [`ClusteredPartialAggregateStream::into_stream`] for why this avoids /// repeatedly removing small batches of groups from the table. /// -/// [`OrderedPartialAggregateStream::into_stream`]: super::ordered_partial_stream::OrderedPartialAggregateStream::into_stream +/// [`ClusteredPartialAggregateStream::into_stream`]: super::clustered_partial_stream::ClusteredPartialAggregateStream::into_stream /// /// # Memory Pressure and Spilling /// -/// ## Fully ordered case +/// ## Full group completion /// -/// If the input is ordered by every group key, for example: +/// Every complete grouping tuple is contiguous. Ordering by every group key is +/// one way to establish this mode, for example: /// /// - Input order: `a, b` /// - `GROUP BY`: `a, b` @@ -84,9 +85,10 @@ use crate::{InputOrderMode, RecordBatchStream, SendableRecordBatchStream}; /// If a memory reservation nevertheless fails, the stream returns the error /// directly, indicating an unexpected behavior. /// -/// ## Partially ordered case +/// ## Partial group completion /// -/// If the input is ordered by only a subset of the group keys, for example: +/// Rows are contiguous for a subset of the group keys. Ordering by that subset +/// is one way to establish this mode, for example: /// /// - Input order: `a` /// - `GROUP BY`: `a, b` @@ -97,27 +99,27 @@ use crate::{InputOrderMode, RecordBatchStream, SendableRecordBatchStream}; /// On reservation failure, the stream sorts the current intermediate states by /// the complete group key and spills them as one run. After the input ends, it /// spills any remaining states, performs a sort-preserving merge of all runs, -/// and feeds the merged input into a fully ordered final aggregate stream. -pub(crate) struct OrderedSingleAggregateStream { +/// and feeds the merged input into a fully clustered final aggregate stream. +pub(crate) struct ClusteredSingleAggregateStream { schema: SchemaRef, input: SendableRecordBatchStream, reservation: MemoryReservation, baseline_metrics: BaselineMetrics, batch_size: usize, - state: Option, + state: Option, } /// See comments at `poll_next()` for details. -enum OrderedSingleAggregateState { +enum ClusteredSingleAggregateState { ReadingInput { - table: OrderedAggregateTable, + table: ClusteredAggregateTable, /// None if either /// - Disk Manager doesn't enable temporary file creation - /// - The group keys are fully ordered, it's expected to use bounded memory + /// - Full group completion is used, so completed groups can be released spill_context: Option>, }, Spilling { - table: OrderedAggregateTable, + table: ClusteredAggregateTable, spill_context: Box, }, /// Emits one materialized batch in `batch_size` slices, then continues @@ -126,10 +128,10 @@ enum OrderedSingleAggregateState { batch: RecordBatch, /// Reserved memory of `batch`, released when handing off the last slice. batch_memory: usize, - next_state: Box, + next_state: Box, }, PreparingMergeInput { - table: OrderedAggregateTable, + table: ClusteredAggregateTable, spill_context: Box, }, MergingSpills { @@ -142,13 +144,13 @@ enum OrderedSingleAggregateState { Error, } -type OrderedSingleAggregatePoll = Poll>>; -type OrderedSingleAggregateStateTransition = ControlFlow< - (OrderedSingleAggregatePoll, OrderedSingleAggregateState), - OrderedSingleAggregateState, +type ClusteredSingleAggregatePoll = Poll>>; +type ClusteredSingleAggregateStateTransition = ControlFlow< + (ClusteredSingleAggregatePoll, ClusteredSingleAggregateState), + ClusteredSingleAggregateState, >; -impl OrderedSingleAggregateStream { +impl ClusteredSingleAggregateStream { pub fn new( agg: &AggregateExec, context: &Arc, @@ -158,7 +160,7 @@ impl OrderedSingleAggregateStream { agg.mode, AggregateMode::Single | AggregateMode::SinglePartitioned )); - debug_assert_ne!(agg.input_order_mode, InputOrderMode::Linear); + debug_assert_ne!(agg.group_completion_mode, GroupCompletionMode::None); let schema = Arc::clone(&agg.schema); let input = agg.input.execute(partition, Arc::clone(context))?; @@ -173,7 +175,7 @@ impl OrderedSingleAggregateStream { AggregateMode::Partial, )?); - let table = OrderedAggregateTable::::new( + let table = ClusteredAggregateTable::::new( agg, partition, Arc::clone(&schema), @@ -181,16 +183,16 @@ impl OrderedSingleAggregateStream { )?; let can_spill = - matches!(agg.input_order_mode, InputOrderMode::PartiallySorted(_)) + matches!(agg.group_completion_mode, GroupCompletionMode::Partial(_)) && context.runtime_env().disk_manager.tmp_files_enabled(); let spill_context = if can_spill { Some(Box::new(AggregateSpill::try_new( - "OrderedSingleAggregateSpill", + "ClusteredSingleAggregateSpill", agg, context, partition, batch_size, - &agg.input_order_mode, + &agg.group_completion_mode, &state_schema, spill_metrics, )?)) @@ -199,7 +201,7 @@ impl OrderedSingleAggregateStream { }; let reservation = - MemoryConsumer::new(format!("OrderedSingleAggregateStream[{partition}]")) + MemoryConsumer::new(format!("ClusteredSingleAggregateStream[{partition}]")) .with_can_spill(can_spill) .register(context.memory_pool()); @@ -212,7 +214,7 @@ impl OrderedSingleAggregateStream { reservation, baseline_metrics, batch_size, - state: Some(OrderedSingleAggregateState::ReadingInput { + state: Some(ClusteredSingleAggregateState::ReadingInput { table, spill_context, }), @@ -224,25 +226,25 @@ impl OrderedSingleAggregateStream { self.input = Box::pin(EmptyRecordBatchStream::new(input_schema)); } - fn break_with_err(error: DataFusionError) -> OrderedSingleAggregateStateTransition { + fn break_with_err(error: DataFusionError) -> ClusteredSingleAggregateStateTransition { ControlFlow::Break(( Poll::Ready(Some(Err(error))), - OrderedSingleAggregateState::Error, + ClusteredSingleAggregateState::Error, )) } - fn break_with_internal_err(message: &str) -> OrderedSingleAggregateStateTransition { + fn break_with_internal_err(message: &str) -> ClusteredSingleAggregateStateTransition { Self::break_with_err(internal_datafusion_err!("{message}")) } /// Reserve memory for the current aggregate table. fn reservation_size_for_table( - table: &OrderedAggregateTable, + table: &ClusteredAggregateTable, spill_context: Option<&AggregateSpill>, ) -> usize { let table_size = table.memory_size(); if spill_context.is_some() { - // See `OrderedSingleAggregateStream` comments for how is it estimated + // See `ClusteredSingleAggregateStream` comments for how is it estimated table_size.saturating_add(table.num_groups().saturating_mul(size_of::())) } else { table_size @@ -256,11 +258,11 @@ impl OrderedSingleAggregateStream { &mut self, batch: RecordBatch, table_memory: usize, - next_state: OrderedSingleAggregateState, - ) -> OrderedSingleAggregateStateTransition { + next_state: ClusteredSingleAggregateState, + ) -> ClusteredSingleAggregateStateTransition { let batch_memory = batch.get_array_memory_size(); match self.reservation.try_resize(table_memory + batch_memory) { - Ok(()) => ControlFlow::Continue(OrderedSingleAggregateState::Outputting { + Ok(()) => ControlFlow::Continue(ClusteredSingleAggregateState::Outputting { batch, batch_memory, next_state: Box::new(next_state), @@ -279,8 +281,8 @@ impl OrderedSingleAggregateStream { } } - /// Consumes one ordered raw input batch, then materializes all finalized - /// groups if the ordering proves any group is ready. + /// Consumes one clustered raw input batch, then materializes all finalized + /// groups if any are complete. /// /// See comments at `poll_next()` for details. /// @@ -288,22 +290,22 @@ impl OrderedSingleAggregateStream { fn handle_reading_input( &mut self, cx: &mut Context<'_>, - original_state: OrderedSingleAggregateState, - ) -> OrderedSingleAggregateStateTransition { - let OrderedSingleAggregateState::ReadingInput { + original_state: ClusteredSingleAggregateState, + ) -> ClusteredSingleAggregateStateTransition { + let ClusteredSingleAggregateState::ReadingInput { mut table, spill_context, } = original_state else { return Self::break_with_internal_err( - "Ordered single aggregate stream expected ReadingInput state", + "Clustered single aggregate stream expected ReadingInput state", ); }; match self.input.poll_next_unpin(cx) { Poll::Pending => ControlFlow::Break(( Poll::Pending, - OrderedSingleAggregateState::ReadingInput { + ClusteredSingleAggregateState::ReadingInput { table, spill_context, }, @@ -332,16 +334,16 @@ impl OrderedSingleAggregateStream { Err(e @ DataFusionError::ResourcesExhausted(_)) => { let Some(spill_context) = spill_context else { // `None` means spilling is not supported, see comments - // at `OrderedSingleAggregateState` for details. + // at `ClusteredSingleAggregateState` for details. return Self::break_with_err(e); }; if table.is_empty() { return Self::break_with_internal_err( - "Ordered single aggregate ran out of memory with no aggregated groups", + "Clustered single aggregate ran out of memory with no aggregated groups", ); } return ControlFlow::Continue( - OrderedSingleAggregateState::Spilling { + ClusteredSingleAggregateState::Spilling { table, spill_context, }, @@ -377,19 +379,19 @@ impl OrderedSingleAggregateStream { self.start_outputting( batch, table_memory, - OrderedSingleAggregateState::ReadingInput { + ClusteredSingleAggregateState::ReadingInput { table, spill_context, }, ) } // Can't do early emit, continue aggregating. - Ok(None) => { - ControlFlow::Continue(OrderedSingleAggregateState::ReadingInput { + Ok(None) => ControlFlow::Continue( + ClusteredSingleAggregateState::ReadingInput { table, spill_context, - }) - } + }, + ), Err(e) => Self::break_with_err(e), } } @@ -399,7 +401,7 @@ impl OrderedSingleAggregateStream { match spill_context { Some(spill_context) if spill_context.has_spills() => { ControlFlow::Continue( - OrderedSingleAggregateState::PreparingMergeInput { + ClusteredSingleAggregateState::PreparingMergeInput { table, spill_context, }, @@ -417,10 +419,10 @@ impl OrderedSingleAggregateStream { Ok(Some(batch)) => self.start_outputting( batch, 0, - OrderedSingleAggregateState::Done, + ClusteredSingleAggregateState::Done, ), Ok(None) => { - ControlFlow::Continue(OrderedSingleAggregateState::Done) + ControlFlow::Continue(ClusteredSingleAggregateState::Done) } Err(e) => Self::break_with_err(e), } @@ -437,22 +439,22 @@ impl OrderedSingleAggregateStream { /// Returns the next operator state with control flow decision. fn handle_spilling( &mut self, - original_state: OrderedSingleAggregateState, - ) -> OrderedSingleAggregateStateTransition { - let OrderedSingleAggregateState::Spilling { + original_state: ClusteredSingleAggregateState, + ) -> ClusteredSingleAggregateStateTransition { + let ClusteredSingleAggregateState::Spilling { mut table, mut spill_context, } = original_state else { return Self::break_with_internal_err( - "Ordered single aggregate stream expected Spilling state", + "Clustered single aggregate stream expected Spilling state", ); }; // Sanity check: it's impossible to OOM when the table is empty if table.is_empty() { return Self::break_with_internal_err( - "Ordered single aggregation entered Spilling with an empty table", + "Clustered single aggregation entered Spilling with an empty table", ); } @@ -472,10 +474,12 @@ impl OrderedSingleAggregateStream { match result { // Finished spilling the aggregate table, continue aggregating from input - Ok(()) => ControlFlow::Continue(OrderedSingleAggregateState::ReadingInput { - table, - spill_context: Some(spill_context), - }), + Ok(()) => { + ControlFlow::Continue(ClusteredSingleAggregateState::ReadingInput { + table, + spill_context: Some(spill_context), + }) + } Err(e) => Self::break_with_err(e), } } @@ -483,7 +487,7 @@ impl OrderedSingleAggregateStream { /// 1. Spills the last in-memory run. /// 2. Constructs a globally ordered input stream by applying a sort-preserving /// merge to all spills. - /// 3. Constructs a replay stream: an ordered aggregate stream over the fully + /// 3. Constructs a replay stream: a clustered aggregate stream over the fully /// ordered input constructed from the spills. /// /// See comments at `poll_next()` for details. @@ -491,15 +495,15 @@ impl OrderedSingleAggregateStream { /// Returns the next operator state with control flow decision. fn handle_preparing_merge_input( &mut self, - original_state: OrderedSingleAggregateState, - ) -> OrderedSingleAggregateStateTransition { - let OrderedSingleAggregateState::PreparingMergeInput { + original_state: ClusteredSingleAggregateState, + ) -> ClusteredSingleAggregateStateTransition { + let ClusteredSingleAggregateState::PreparingMergeInput { mut table, mut spill_context, } = original_state else { return Self::break_with_internal_err( - "Ordered single aggregate stream expected PreparingMergeInput state", + "Clustered single aggregate stream expected PreparingMergeInput state", ); }; @@ -527,7 +531,7 @@ impl OrderedSingleAggregateStream { match replay { Ok(stream) => { - ControlFlow::Continue(OrderedSingleAggregateState::MergingSpills { + ControlFlow::Continue(ClusteredSingleAggregateState::MergingSpills { stream, }) } @@ -544,26 +548,28 @@ impl OrderedSingleAggregateStream { fn handle_merging_spills( &mut self, cx: &mut Context<'_>, - original_state: OrderedSingleAggregateState, - ) -> OrderedSingleAggregateStateTransition { - let OrderedSingleAggregateState::MergingSpills { mut stream } = original_state + original_state: ClusteredSingleAggregateState, + ) -> ClusteredSingleAggregateStateTransition { + let ClusteredSingleAggregateState::MergingSpills { mut stream } = original_state else { return Self::break_with_internal_err( - "Ordered single aggregate stream expected MergingSpills state", + "Clustered single aggregate stream expected MergingSpills state", ); }; match stream.poll_next_unpin(cx) { Poll::Pending => ControlFlow::Break(( Poll::Pending, - OrderedSingleAggregateState::MergingSpills { stream }, + ClusteredSingleAggregateState::MergingSpills { stream }, )), Poll::Ready(Some(Ok(batch))) => ControlFlow::Break(( Poll::Ready(Some(Ok(batch))), - OrderedSingleAggregateState::MergingSpills { stream }, + ClusteredSingleAggregateState::MergingSpills { stream }, )), Poll::Ready(Some(Err(e))) => Self::break_with_err(e), - Poll::Ready(None) => ControlFlow::Continue(OrderedSingleAggregateState::Done), + Poll::Ready(None) => { + ControlFlow::Continue(ClusteredSingleAggregateState::Done) + } } } @@ -574,16 +580,16 @@ impl OrderedSingleAggregateStream { /// Returns the next operator state with control flow decision. fn handle_outputting( &mut self, - original_state: OrderedSingleAggregateState, - ) -> OrderedSingleAggregateStateTransition { - let OrderedSingleAggregateState::Outputting { + original_state: ClusteredSingleAggregateState, + ) -> ClusteredSingleAggregateStateTransition { + let ClusteredSingleAggregateState::Outputting { batch, batch_memory, next_state, } = original_state else { return Self::break_with_internal_err( - "Ordered single aggregate stream expected Outputting state", + "Clustered single aggregate stream expected Outputting state", ); }; @@ -592,7 +598,7 @@ impl OrderedSingleAggregateStream { let batch = batch.slice(self.batch_size, batch.num_rows() - self.batch_size); return ControlFlow::Break(( Poll::Ready(Some(Ok(output.record_output(&self.baseline_metrics)))), - OrderedSingleAggregateState::Outputting { + ClusteredSingleAggregateState::Outputting { batch, batch_memory, next_state, @@ -611,20 +617,20 @@ impl OrderedSingleAggregateStream { } } -impl Stream for OrderedSingleAggregateStream { +impl Stream for ClusteredSingleAggregateStream { type Item = Result; - /// Entry point for the ordered single aggregate state machine. + /// Entry point for the clustered single aggregate state machine. /// - /// See comments in [`OrderedSingleAggregateStream`] for high-level ideas. + /// See comments in [`ClusteredSingleAggregateStream`] for high-level ideas. /// /// State transition graph: /// /// ```text /// (start) /// -> ReadingInput - /// The stream starts by polling ordered raw input and updating the - /// ordered single aggregate table. + /// The stream starts by polling clustered raw input and updating the + /// clustered single aggregate table. /// /// ReadingInput /// -> ReadingInput @@ -634,7 +640,7 @@ impl Stream for OrderedSingleAggregateStream { /// The table cannot reserve enough memory. Move all current states into /// one fully group-key-sorted spill run. /// -> Outputting - /// Either the input ordering proves some groups complete, or input was + /// Either some groups are complete, or input was /// exhausted without spilling and every remaining group is complete. /// Materialize all of them once into one batch. If the batch cannot be /// reserved, yield it whole and go to the state after `Outputting`. @@ -689,31 +695,31 @@ impl Stream for OrderedSingleAggregateStream { let cur_state = self .state .take() - .expect("OrderedSingleAggregateStream state should not be None"); + .expect("ClusteredSingleAggregateStream state should not be None"); let next_state = match cur_state { - state @ OrderedSingleAggregateState::ReadingInput { .. } => { + state @ ClusteredSingleAggregateState::ReadingInput { .. } => { self.handle_reading_input(cx, state) } - state @ OrderedSingleAggregateState::Spilling { .. } => { + state @ ClusteredSingleAggregateState::Spilling { .. } => { self.handle_spilling(state) } - state @ OrderedSingleAggregateState::PreparingMergeInput { .. } => { + state @ ClusteredSingleAggregateState::PreparingMergeInput { .. } => { self.handle_preparing_merge_input(state) } - state @ OrderedSingleAggregateState::MergingSpills { .. } => { + state @ ClusteredSingleAggregateState::MergingSpills { .. } => { self.handle_merging_spills(cx, state) } - state @ OrderedSingleAggregateState::Outputting { .. } => { + state @ ClusteredSingleAggregateState::Outputting { .. } => { self.handle_outputting(state) } - state @ OrderedSingleAggregateState::Error => { + state @ ClusteredSingleAggregateState::Error => { self.close_input(); self.reservation.free(); self.state = Some(state); return Poll::Ready(None); } - state @ OrderedSingleAggregateState::Done => { + state @ ClusteredSingleAggregateState::Done => { let _ = self.reservation.try_resize(0); self.state = Some(state); return Poll::Ready(None); @@ -727,14 +733,14 @@ impl Stream for OrderedSingleAggregateStream { ControlFlow::Break((Poll::Ready(Some(Err(e))), next_state)) => { debug_assert!(matches!( next_state, - OrderedSingleAggregateState::Error + ClusteredSingleAggregateState::Error )); // The handler has already discarded its state-owned resources. // Release the remaining stream-owned resources before returning. self.close_input(); self.reservation.free(); - self.state = Some(OrderedSingleAggregateState::Error); + self.state = Some(ClusteredSingleAggregateState::Error); return Poll::Ready(Some(Err(e))); } ControlFlow::Break((poll, next_state)) => { @@ -746,7 +752,7 @@ impl Stream for OrderedSingleAggregateStream { } } -impl RecordBatchStream for OrderedSingleAggregateStream { +impl RecordBatchStream for ClusteredSingleAggregateStream { fn schema(&self) -> SchemaRef { Arc::clone(&self.schema) } diff --git a/datafusion/physical-plan/src/aggregates/group_values/mod.rs b/datafusion/physical-plan/src/aggregates/group_values/mod.rs index 32eabfe8dc26c..8843fe54afea8 100644 --- a/datafusion/physical-plan/src/aggregates/group_values/mod.rs +++ b/datafusion/physical-plan/src/aggregates/group_values/mod.rs @@ -29,7 +29,7 @@ mod row; pub use row::GroupValuesRows; mod single_group_by; use datafusion_physical_expr::binary_map::OutputType; -use multi_group_by::{GroupValuesColumn, GroupValuesOrdered}; +use multi_group_by::{GroupValuesClustered, GroupValuesColumn}; pub(crate) use single_group_by::primitive::HashValue; @@ -38,7 +38,7 @@ use crate::aggregates::{ boolean::GroupValuesBoolean, bytes::GroupValuesBytes, bytes_view::GroupValuesBytesView, primitive::GroupValuesPrimitive, }, - order::GroupOrdering, + order::GroupCompletion, }; mod metrics; @@ -139,7 +139,7 @@ pub trait GroupValues: Send { /// /// [`GroupValues`] implementations choosing logic: /// -/// - Fully ordered multi-column keys with supported scalar types use adjacent +/// - Fully clustered multi-column keys with supported scalar types use adjacent /// comparisons and column builders, without a hash table. /// /// - If group by single column, and type of this column has @@ -156,12 +156,12 @@ pub trait GroupValues: Send { /// `GroupValuesRows`: crate::aggregates::group_values::GroupValuesRows pub fn new_group_values( schema: SchemaRef, - group_ordering: &GroupOrdering, + group_completion: &GroupCompletion, ) -> Result> { - if matches!(group_ordering, GroupOrdering::Full(_)) - && GroupValuesOrdered::supports_schema(&schema) + if matches!(group_completion, GroupCompletion::Full(_)) + && GroupValuesClustered::supports_schema(&schema) { - return Ok(Box::new(GroupValuesOrdered::try_new(schema)?)); + return Ok(Box::new(GroupValuesClustered::try_new(schema)?)); } if schema.fields.len() == 1 { let d = schema.fields[0].data_type(); @@ -200,7 +200,7 @@ pub fn new_group_values( } if multi_group_by::supported_schema(schema.as_ref()) { - if matches!(group_ordering, GroupOrdering::None) { + if matches!(group_completion, GroupCompletion::None) { Ok(Box::new(GroupValuesColumn::::try_new(schema)?)) } else { Ok(Box::new(GroupValuesColumn::::try_new(schema)?)) @@ -221,7 +221,7 @@ mod tests { use datafusion_expr::{EmitTo, GroupSelection}; use super::new_group_values; - use crate::aggregates::order::GroupOrdering; + use crate::aggregates::order::GroupCompletion; #[test] fn preserving_values_keep_group_indices_valid() { @@ -230,7 +230,7 @@ mod tests { DataType::Int32, true, )])); - let mut group_values = new_group_values(schema, &GroupOrdering::None).unwrap(); + let mut group_values = new_group_values(schema, &GroupCompletion::None).unwrap(); assert!(group_values.supports_values_preserving()); let input = Arc::new(Int32Array::from(vec![ @@ -287,7 +287,7 @@ mod tests { Field::new("primitive", DataType::Int32, false), Field::new("boolean", DataType::Boolean, false), ])); - let mut group_values = new_group_values(schema, &GroupOrdering::None).unwrap(); + let mut group_values = new_group_values(schema, &GroupCompletion::None).unwrap(); let input = vec![ Arc::new(Int32Array::from(vec![10, 20, 10])) as ArrayRef, Arc::new(BooleanArray::from(vec![true, false, true])) as ArrayRef, @@ -318,7 +318,7 @@ mod tests { true, )])); let mut group_values = - new_group_values(schema, &GroupOrdering::None).unwrap(); + new_group_values(schema, &GroupCompletion::None).unwrap(); let input: ArrayRef = match data_type { DataType::Utf8 => Arc::new(StringArray::from(vec![ Some("a"), diff --git a/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/ordered.rs b/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/clustered.rs similarity index 87% rename from datafusion/physical-plan/src/aggregates/group_values/multi_group_by/ordered.rs rename to datafusion/physical-plan/src/aggregates/group_values/multi_group_by/clustered.rs index 9e34ac7aeac17..233bdef086594 100644 --- a/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/ordered.rs +++ b/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/clustered.rs @@ -27,7 +27,7 @@ use datafusion_expr::{EmitTo, GroupSelection}; use super::{GroupColumn, GroupValuesColumn}; use crate::aggregates::group_values::GroupValues; -/// Columnar group keys for input fully ordered by all grouping expressions. +/// Columnar group keys for input clustered by the complete grouping tuple. /// /// Equal keys must be contiguous across input batches. Within a batch Arrow's /// partition kernel finds those runs. Only the first run can continue the last @@ -36,14 +36,15 @@ use crate::aggregates::group_values::GroupValues; /// /// This removes hashing and hash-table storage, but does not change the dense /// group-id or emission contracts. In particular, `First(n)` shifts the remaining -/// keys and their ids together. Partially ordered inputs must use a hash table. -pub(crate) struct GroupValuesOrdered { +/// keys and their ids together. Inputs clustered by only a subset of the keys +/// must use a hash table. +pub(crate) struct GroupValuesClustered { schema: SchemaRef, columns: Vec>, new_groups: Vec, } -impl GroupValuesOrdered { +impl GroupValuesClustered { /// Types for which adjacent Arrow equality agrees with GROUP BY equality. /// Keep single-column specializations and floats/nested/encoded keys on the /// established path until their semantics and performance are validated. @@ -74,7 +75,7 @@ impl GroupValuesOrdered { pub(crate) fn try_new(schema: SchemaRef) -> Result { if !Self::supports_schema(&schema) { return not_impl_err!( - "Unsupported schema for fully ordered group values: {schema}" + "Unsupported schema for fully clustered group values: {schema}" ); } let columns = GroupValuesColumn::::build_group_columns(&schema)?; @@ -86,7 +87,7 @@ impl GroupValuesOrdered { } } -impl GroupValues for GroupValuesOrdered { +impl GroupValues for GroupValuesClustered { fn intern(&mut self, cols: &[ArrayRef], groups: &mut Vec) -> Result<()> { groups.clear(); let ranges = partition(cols)?.ranges(); @@ -170,7 +171,7 @@ impl GroupValues for GroupValuesOrdered { fn clear_shrink(&mut self, num_rows: usize) { self.columns = GroupValuesColumn::::build_group_columns(&self.schema) - .expect("schema validated by GroupValuesOrdered::try_new"); + .expect("schema validated by GroupValuesClustered::try_new"); self.new_groups.clear(); self.new_groups.shrink_to(num_rows); } @@ -191,8 +192,12 @@ mod tests { // Compare with the existing hash implementation, changing actual input // boundaries and removing completed groups between batches. This checks the // semantic contract rather than duplicating the adjacent-run algorithm. - #[test] - fn ordered_keys_match_hash_grouping_across_batches_and_emits() -> Result<()> { + #[rstest::rstest] + #[case::sorted(false)] + #[case::unsorted(true)] + fn clustered_keys_match_hash_grouping_across_batches_and_emits( + #[case] unsorted: bool, + ) -> Result<()> { for key_type in [ DataType::Boolean, DataType::Int8, @@ -213,7 +218,9 @@ mod tests { let rows = 1025; let first: ArrayRef = Arc::new(Int32Array::from_iter((0..rows).map(|row| { let group = row / 7; - (group >= 4).then_some(group / 4) + let key = group / 4; + // Swap neighboring key values while keeping each run contiguous. + (group >= 4).then_some(if unsorted { key ^ 1 } else { key }) }))); let second: ArrayRef = match key_type { DataType::Boolean => { @@ -262,8 +269,8 @@ mod tests { ])); for batch_size in [1, 2, 7, 8, 63, 1024] { for emit_limit in [0, 1, 17, usize::MAX] { - let mut ordered = - GroupValuesOrdered::try_new(Arc::clone(&schema))?; + let mut clustered = + GroupValuesClustered::try_new(Arc::clone(&schema))?; let mut hashed = GroupValuesColumn::::try_new(Arc::clone(&schema))?; let mut actual = Vec::new(); @@ -274,16 +281,16 @@ mod tests { .iter() .map(|array| array.slice(offset, length)) .collect::>(); - ordered.intern(&batch, &mut actual)?; + clustered.intern(&batch, &mut actual)?; hashed.intern(&batch, &mut expected)?; assert_eq!( actual, expected, "{key_type:?}, batch={batch_size}, offset={offset}, descending={descending}" ); - assert_eq!(ordered.len(), hashed.len()); - let selection = GroupSelection::all(ordered.len()); + assert_eq!(clustered.len(), hashed.len()); + let selection = GroupSelection::all(clustered.len()); assert_eq!( - ordered.values_preserving(selection)?, + clustered.values_preserving(selection)?, hashed.values_preserving(selection)? ); // A zero-row batch must neither forget the boundary @@ -292,23 +299,29 @@ mod tests { .iter() .map(|array| array.slice(0, 0)) .collect::>(); - ordered.intern(&empty, &mut actual)?; + clustered.intern(&empty, &mut actual)?; assert!(actual.is_empty()); - let emit = emit_limit.min(ordered.len().saturating_sub(1)); + let emit = emit_limit.min(clustered.len().saturating_sub(1)); assert_eq!( - ordered.emit(EmitTo::First(emit))?, + clustered.emit(EmitTo::First(emit))?, hashed.emit(EmitTo::First(emit))? ); } - assert_eq!(ordered.emit(EmitTo::All)?, hashed.emit(EmitTo::All)?); - assert!(ordered.is_empty()); + assert_eq!( + clustered.emit(EmitTo::All)?, + hashed.emit(EmitTo::All)? + ); + assert!(clustered.is_empty()); // All and clear_shrink reset the cross-batch state. - ordered.clear_shrink(0); + clustered.clear_shrink(0); hashed.clear_shrink(0); - ordered.intern(&input, &mut actual)?; + clustered.intern(&input, &mut actual)?; hashed.intern(&input, &mut expected)?; assert_eq!(actual, expected); - assert_eq!(ordered.emit(EmitTo::All)?, hashed.emit(EmitTo::All)?); + assert_eq!( + clustered.emit(EmitTo::All)?, + hashed.emit(EmitTo::All)? + ); } } } @@ -317,7 +330,7 @@ mod tests { } #[test] - fn ordered_schema_gate_keeps_unvalidated_types_on_existing_paths() { + fn clustered_schema_gate_keeps_unvalidated_types_on_existing_paths() { for data_type in [ DataType::Float32, DataType::Float64, @@ -329,18 +342,20 @@ mod tests { Field::new("a", DataType::Int32, false), Field::new("b", data_type, true), ]); - assert!(!GroupValuesOrdered::supports_schema(&schema)); + assert!(!GroupValuesClustered::supports_schema(&schema)); } - assert!(!GroupValuesOrdered::supports_schema(&Schema::new(vec![ + assert!(!GroupValuesClustered::supports_schema(&Schema::new(vec![ Field::new("a", DataType::Int32, false) ]))); } #[tokio::test] async fn ordered_single_and_partial_final_match_unordered_execution() -> Result<()> { - use crate::aggregates::{AggregateExec, AggregateMode, PhysicalGroupBy}; + use crate::aggregates::{ + AggregateExec, AggregateMode, GroupCompletionMode, PhysicalGroupBy, + }; use crate::test::TestMemoryExec; - use crate::{ExecutionPlan, InputOrderMode, collect}; + use crate::{ExecutionPlan, collect}; use arrow::compute::{SortOptions, take_record_batch}; use arrow::record_batch::RecordBatch; use arrow::row::{RowConverter, SortField}; @@ -437,7 +452,10 @@ mod tests { Arc::clone(&schema), )?; if sorted { - assert_eq!(plan.input_order_mode(), &InputOrderMode::Sorted); + assert_eq!( + plan.group_completion_mode(), + &GroupCompletionMode::Full + ); } let plan: Arc = if two_stage { let plan = AggregateExec::try_new( @@ -450,8 +468,8 @@ mod tests { )?; if sorted { assert_eq!( - plan.input_order_mode(), - &InputOrderMode::Sorted + plan.group_completion_mode(), + &GroupCompletionMode::Full ); } Arc::new(plan) diff --git a/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/list.rs b/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/list.rs index b5dbdb38763f4..96cd4dde16ba4 100644 --- a/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/list.rs +++ b/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/list.rs @@ -445,7 +445,7 @@ mod tests { // List/Struct together correctly and that intern/emit round-trips // through `GroupValuesColumn` rather than `GroupValuesRows`. use crate::aggregates::group_values::new_group_values; - use crate::aggregates::order::GroupOrdering; + use crate::aggregates::order::GroupCompletion; use arrow::array::{ Int32Array, LargeListArray, StringArray, StructArray, builder::Int32Builder, builder::LargeListBuilder, builder::StringBuilder, builder::StructBuilder, @@ -507,7 +507,7 @@ mod tests { Arc::new(list_builder.finish()) }; - let mut gv = new_group_values(schema, &GroupOrdering::None).unwrap(); + let mut gv = new_group_values(schema, &GroupCompletion::None).unwrap(); // Batch 1: a mix of duplicate / distinct / null lists. let batch1 = notes_v(&[ @@ -584,7 +584,7 @@ mod tests { #[test] fn list_dispatcher_round_trip_through_new_group_values() { use crate::aggregates::group_values::new_group_values; - use crate::aggregates::order::GroupOrdering; + use crate::aggregates::order::GroupCompletion; use arrow::datatypes::Schema; use datafusion_expr::EmitTo; @@ -593,7 +593,7 @@ mod tests { DataType::List(child_field()), true, )])); - let mut gv = new_group_values(schema, &GroupOrdering::None).unwrap(); + let mut gv = new_group_values(schema, &GroupCompletion::None).unwrap(); // Batch 1. let batch1: ArrayRef = list_array(&[ @@ -869,7 +869,7 @@ mod tests { // ints. Exercises a recursive child GroupColumn built via the // dispatcher. use crate::aggregates::group_values::new_group_values; - use crate::aggregates::order::GroupOrdering; + use crate::aggregates::order::GroupCompletion; use arrow::array::{ListArray, builder::Int32Builder, builder::ListBuilder}; use arrow::datatypes::Schema; use datafusion_expr::EmitTo; @@ -917,7 +917,7 @@ mod tests { Arc::new(outer.finish()) }; - let mut gv = new_group_values(schema, &GroupOrdering::None).unwrap(); + let mut gv = new_group_values(schema, &GroupCompletion::None).unwrap(); // Three groups: [[1,2],[3]], its duplicate, a distinct value, and a null. let batch = mk(&[ diff --git a/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/mod.rs b/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/mod.rs index 9497d521e6321..136a40d263c8f 100644 --- a/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/mod.rs +++ b/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/mod.rs @@ -20,11 +20,11 @@ mod boolean; mod bytes; pub mod bytes_view; +mod clustered; mod dictionary; mod fixed_size_binary; mod list; -mod ordered; -pub(super) use ordered::GroupValuesOrdered; +pub(super) use clustered::GroupValuesClustered; pub mod primitive; pub mod row_backed; diff --git a/datafusion/physical-plan/src/aggregates/grouped_hash_stream.rs b/datafusion/physical-plan/src/aggregates/grouped_hash_stream.rs index b9a6d2833ab7b..a874c523ec9a2 100644 --- a/datafusion/physical-plan/src/aggregates/grouped_hash_stream.rs +++ b/datafusion/physical-plan/src/aggregates/grouped_hash_stream.rs @@ -35,14 +35,14 @@ use std::task::{Context, Poll}; use std::vec; use super::aggregate_hash_table::{accumulator_phases, create_group_accumulator}; -use super::order::GroupOrdering; +use super::order::GroupCompletion; use super::skip_partial::SkipAggregationProbe; use super::{AggregateExec, format_human_display}; use crate::aggregates::group_values::{ AccumulatorPhase, AggregateAccumulatorMetrics, AggregateArgumentMetrics, GroupByMetrics, GroupValues, aggregate_sub_metrics, new_group_values, }; -use crate::aggregates::order::GroupOrderingFull; +use crate::aggregates::order::GroupCompletionFull; use crate::aggregates::{ AggregateInputMode, AggregateMode, AggregateOutputMode, PhysicalGroupBy, aggregate_metric_label, create_schema, evaluate_group_by, evaluate_optional, @@ -353,10 +353,8 @@ pub(crate) struct GroupedHashAggregateStream { // TASK-SPECIFIC STATES: // Inner states groups together properties, states for a specific task. // ======================================================================== - /// Optional ordering information, that might allow groups to be - /// emitted from the hash table prior to seeing the end of the - /// input - group_ordering: GroupOrdering, + /// Tracks groups that can be emitted from the hash table before the input ends. + group_completion: GroupCompletion, /// The spill state object spill_state: SpillState, @@ -528,20 +526,20 @@ impl GroupedHashAggregateStream { .collect::>() .join(", "); let name = format!("GroupedHashAggregateStream[{partition}] ({agg_fn_names})"); - let group_ordering = GroupOrdering::try_new(&agg.input_order_mode)?; - let oom_mode = match (agg.mode, &group_ordering) { + let group_completion = GroupCompletion::try_new(&agg.group_completion_mode)?; + let oom_mode = match (agg.mode, &group_completion) { // In partial aggregation mode, always prefer to emit incomplete results early. (AggregateMode::Partial, _) => OutOfMemoryMode::EmitEarly, // For non-partial aggregation modes, emitting incomplete results is not an option. // Instead, use disk spilling to store sorted, incomplete results, and merge them // afterwards. - (_, GroupOrdering::None | GroupOrdering::Partial(_)) + (_, GroupCompletion::None | GroupCompletion::Partial(_)) if context.runtime_env().disk_manager.tmp_files_enabled() => { OutOfMemoryMode::Spill } - // For `GroupOrdering::Full`, the incoming stream is already sorted. This ensures the - // number of incomplete groups can be kept small at all times. If we still hit + // For `GroupCompletion::Full`, each group is contiguous in the input. This keeps + // the number of incomplete groups small at all times. If we still hit // an out-of-memory condition, spilling to disk would not be beneficial since the same // situation is likely to reoccur when reading back the spilled data. // Therefore, we fall back to simply reporting the error immediately. @@ -550,7 +548,7 @@ impl GroupedHashAggregateStream { _ => OutOfMemoryMode::ReportError, }; - let group_values = new_group_values(group_schema, &group_ordering)?; + let group_values = new_group_values(group_schema, &group_completion)?; let reservation = MemoryConsumer::new(name) // We interpret 'can spill' as 'can handle memory back pressure'. // This value needs to be set to true for the default memory pool implementations @@ -586,7 +584,7 @@ impl GroupedHashAggregateStream { // since Final mode expects unique group values as its input // - there is only one GROUP BY expressions set let skip_aggregation_probe = if agg.mode == AggregateMode::Partial - && matches!(group_ordering, GroupOrdering::None) + && matches!(group_completion, GroupCompletion::None) && agg_group_by.is_single() { let options = &context.session_config().options().execution; @@ -641,7 +639,7 @@ impl GroupedHashAggregateStream { aggregate_argument_metrics, aggregate_accumulator_metrics, batch_size, - group_ordering, + group_completion, input_done: false, spill_state, group_values_soft_limit: agg.limit_options().map(|config| config.limit()), @@ -698,7 +696,7 @@ impl Stream for GroupedHashAggregateStream { // this might lead to incorrect output ordering if (self.spill_state.spills.is_empty() || self.spill_state.is_stream_merging) - && let Some(to_emit) = self.group_ordering.emit_to() + && let Some(to_emit) = self.group_completion.emit_to() { timer.done(); if let Some(batch) = self.emit(to_emit, false)? { @@ -899,7 +897,7 @@ impl GroupedHashAggregateStream { })?; for group_values in &group_by_values { - // Calculate group indices and update ordering information. + // Calculate group indices and update the completion tracker. let total_num_groups = self.group_by_metrics.time_group_key_preparation(|| { let starting_num_groups = self.group_values.len(); @@ -908,7 +906,7 @@ impl GroupedHashAggregateStream { let group_indices = &self.current_group_indices; let total_num_groups = self.group_values.len(); if total_num_groups > starting_num_groups { - self.group_ordering.new_groups( + self.group_completion.new_groups( group_values, group_indices, total_num_groups, @@ -997,7 +995,7 @@ impl GroupedHashAggregateStream { self.group_values.len() }; - if let Some(emit_to) = self.group_ordering.oom_emit_to(n) + if let Some(emit_to) = self.group_completion.oom_emit_to(n) && let Some(batch) = self.emit(emit_to, false)? { return Ok(Some(ExecutionState::ProducingOutput(batch))); @@ -1014,7 +1012,7 @@ impl GroupedHashAggregateStream { let acc = self.accumulators.iter().map(|x| x.size()).sum::(); let groups_and_acc_size = acc + self.group_values.size() - + self.group_ordering.size() + + self.group_completion.size() + self.current_group_indices.allocated_size(); // Reserve extra headroom for sorting during potential spill. @@ -1060,7 +1058,7 @@ impl GroupedHashAggregateStream { let output = group_by_metrics.time_emitting(|| { let mut output = self.group_values.emit(emit_to)?; if let EmitTo::First(n) = emit_to { - self.group_ordering.remove_groups(n); + self.group_completion.remove_groups(n); } // Next output each aggregate value. @@ -1140,7 +1138,7 @@ impl GroupedHashAggregateStream { .intern(&cols, &mut self.current_group_indices)?; let total_groups = self.group_values.len(); if total_groups > starting_groups { - self.group_ordering.new_groups( + self.group_completion.new_groups( &cols, &self.current_group_indices, total_groups, @@ -1314,7 +1312,7 @@ impl GroupedHashAggregateStream { /// in case of disk spilling, the SPM stream have been drained. fn set_input_done_and_produce_output(&mut self) -> Result<()> { self.input_done = true; - self.group_ordering.input_done(); + self.group_completion.input_done(); // Release the original input pipeline's resources now that we're done // reading from it. In the spill branch below, `self.input` is replaced // again with a stream that merges spill files. @@ -1359,12 +1357,12 @@ impl GroupedHashAggregateStream { // Reset the group values collectors. self.clear_all(); - // We can now use `GroupOrdering::Full` since the spill files are sorted + // We can now use `GroupCompletion::Full` since the spill files are sorted // on the grouping columns. - self.group_ordering = GroupOrdering::Full(GroupOrderingFull::new()); + self.group_completion = GroupCompletion::Full(GroupCompletionFull::new()); // Recreate `group_values` for streaming merge so group ids are assigned - // in first-seen order, as required by `GroupOrderingFull`. + // in first-seen order, as required by `GroupCompletionFull`. // The pre-spill collector may use `vectorized_intern`, which can assign // new group ids out of input order under hash collisions. That is the // multi-column collector, which also serves a single group column @@ -1374,7 +1372,7 @@ impl GroupedHashAggregateStream { .spill_state .merging_group_by .group_schema(&self.spill_state.spill_schema)?; - self.group_values = new_group_values(group_schema, &self.group_ordering)?; + self.group_values = new_group_values(group_schema, &self.group_completion)?; // Use `OutOfMemoryMode::ReportError` from this point on // to ensure we don't spill the spilled data to disk again. @@ -1483,7 +1481,7 @@ impl GroupedHashAggregateStream { mod tests { use super::*; use crate::ExecutionPlan; - use crate::InputOrderMode; + use crate::aggregates::GroupCompletionMode; use crate::test::TestMemoryExec; use arrow::array::{Int32Array, Int64Array, UInt32Array}; use arrow::datatypes::{DataType, Field, Schema}; @@ -1770,7 +1768,7 @@ mod tests { Ok(()) } - // Migrated to OrderedPartialAggregateStream coverage in aggregates/mod.rs; + // Migrated to ClusteredPartialAggregateStream coverage in aggregates/mod.rs; // kept here for the legacy GroupedHashAggregateStream implementation. #[tokio::test] async fn test_emit_early_with_partially_sorted() -> Result<()> { @@ -1838,11 +1836,11 @@ mod tests { Arc::clone(&schema), )?; assert!(matches!( - aggregate_exec.input_order_mode(), - InputOrderMode::PartiallySorted(_) + aggregate_exec.group_completion_mode(), + GroupCompletionMode::Partial(_) )); - // Must not panic with "assertion failed: *current_sort >= n" + // Must not panic with "assertion failed: *current_run_start >= n" let mut stream = GroupedHashAggregateStream::new(&aggregate_exec, &task_ctx, 0)?; while let Some(result) = stream.next().await { if let Err(e) = result { diff --git a/datafusion/physical-plan/src/aggregates/hash_stream.rs b/datafusion/physical-plan/src/aggregates/hash_stream.rs index 2f1bb9b638a4c..93992041f1d8d 100644 --- a/datafusion/physical-plan/src/aggregates/hash_stream.rs +++ b/datafusion/physical-plan/src/aggregates/hash_stream.rs @@ -34,16 +34,17 @@ use futures::stream::{Stream, StreamExt}; use super::AggregateExec; use super::aggregate_hash_table::{ - AggregateHashTable, FinalMarker, OrderedAggregateTableMetrics, PartialMarker, + AggregateHashTable, ClusteredAggregateTableMetrics, FinalMarker, PartialMarker, PartialSkipMarker, }; +use super::order::GroupCompletionMode; use super::skip_partial::SkipAggregationProbe; use super::spill::AggregateSpill; use crate::metrics::{ BaselineMetrics, MetricBuilder, MetricCategory, RecordOutput, SpillMetrics, }; use crate::stream::{EmptyRecordBatchStream, RecordBatchStreamAdapter}; -use crate::{InputOrderMode, SendableRecordBatchStream, metrics}; +use crate::{SendableRecordBatchStream, metrics}; /// Hash aggregation is implemented in two stages: partial and final. This /// stream implements the partial stage. @@ -136,7 +137,7 @@ use crate::{InputOrderMode, SendableRecordBatchStream, metrics}; /// 3. Perform a sort-preserving merge of all spill files and feed the merged output /// into an ordered streaming aggregation, which ensures bounded memory usage and /// evaluates the final result. -/// - [`OrderedFinalAggregateStream`](super::ordered_final_stream::OrderedFinalAggregateStream) is reused for the streaming aggregation. +/// - [`ClusteredFinalAggregateStream`](super::clustered_final_stream::ClusteredFinalAggregateStream) is reused for the streaming aggregation. pub(crate) struct PartialHashAggregateStream { /// Output schema: group columns followed by partial aggregate state columns. schema: SchemaRef, @@ -217,7 +218,7 @@ impl PartialHashAggregateStream { partition: usize, ) -> Result { debug_assert_eq!(agg.mode, super::AggregateMode::Partial); - debug_assert_eq!(agg.input_order_mode, InputOrderMode::Linear); + debug_assert_eq!(agg.group_completion_mode, GroupCompletionMode::None); let schema = Arc::clone(&agg.schema); let input = agg.input.execute(partition, Arc::clone(context))?; @@ -573,7 +574,7 @@ impl FinalHashAggregateStream { agg.mode, super::AggregateMode::Final | super::AggregateMode::FinalPartitioned )); - debug_assert_eq!(agg.input_order_mode, InputOrderMode::Linear); + debug_assert_eq!(agg.group_completion_mode, GroupCompletionMode::None); let input = agg.input.execute(partition, Arc::clone(context))?; Self::new_with_input(agg, context, partition, input) @@ -607,7 +608,7 @@ impl FinalHashAggregateStream { context, partition, batch_size, - &InputOrderMode::Linear, + &GroupCompletionMode::None, &input_schema, spill_metrics, )?)) @@ -783,7 +784,7 @@ impl FinalHashAggregateStream { /// Produce output from spills /// 1. Spill in progress in-memory hash table - /// 2. Switch to ordered final stream + /// 2. Switch to clustered final stream /// 3. passthrough stream output async fn produce_output_from_spills( &mut self, @@ -801,7 +802,7 @@ impl FinalHashAggregateStream { // Construct the ordered input used to merge all spill files. let mut output_stream = - self.switch_to_ordered_final_stream(hash_table, spill_context)?; + self.switch_to_clustered_final_stream(hash_table, spill_context)?; timer.done(); @@ -819,16 +820,16 @@ impl FinalHashAggregateStream { /// 1. Constructs a globally ordered input stream by applying a sort-preserving /// merge to all spills. - /// 2. Constructs a replay stream: an ordered final aggregate stream over the + /// 2. Constructs a replay stream: a clustered final aggregate stream over the /// fully ordered input constructed from the spills. /// /// Returns the replay stream - fn switch_to_ordered_final_stream( + fn switch_to_clustered_final_stream( &mut self, hash_table: AggregateHashTable, spill_context: Box, ) -> Result { - let metrics = OrderedAggregateTableMetrics::from_hash_table(&hash_table); + let metrics = ClusteredAggregateTableMetrics::from_hash_table(&hash_table); drop(hash_table); self.reservation.try_resize(0)?; spill_context.into_replay_stream( @@ -879,6 +880,7 @@ mod tests { use std::time::Duration; use super::*; + use crate::aggregates::GroupCompletionMode; use crate::aggregates::{AggregateMode, PhysicalGroupBy}; use crate::common::collect; use crate::execution_plan::ExecutionPlan; @@ -1567,7 +1569,10 @@ mod tests { input, schema, )?; - assert_eq!(aggregate.input_order_mode(), &InputOrderMode::Linear); + assert_eq!( + aggregate.group_completion_mode(), + &GroupCompletionMode::None + ); let pool: Arc = Arc::new(GreedyMemoryPool::new(limit)); let context = Arc::new( diff --git a/datafusion/physical-plan/src/aggregates/mod.rs b/datafusion/physical-plan/src/aggregates/mod.rs index 6552bd95e06a5..101be7f8f4929 100644 --- a/datafusion/physical-plan/src/aggregates/mod.rs +++ b/datafusion/physical-plan/src/aggregates/mod.rs @@ -48,19 +48,19 @@ //! //! See [`PartialHashAggregateStream`] and [`FinalHashAggregateStream`] for details. //! -//! ### Ordering optimization +//! ### Group completion optimization //! -//! When the input is ordered by the group key, an ordered fast path is used. It -//! uses a similar two-stage hash aggregation with an early-emission optimization. +//! When the input is ordered by group keys, rows are clustered by those keys. +//! The clustered paths use this guarantee to emit completed groups early. //! //! ```text -//! AggregateExec (final, ordered) +//! AggregateExec (final, clustered) //! RepartitionExec (hash by group keys, order-preserving) -//! AggregateExec (partial, ordered) +//! AggregateExec (partial, clustered) //! ``` //! -//! See [`OrderedPartialAggregateStream`], [`OrderedFinalAggregateStream`], and -//! [`OrderedSingleAggregateStream`] for details. +//! See [`ClusteredPartialAggregateStream`], [`ClusteredFinalAggregateStream`], and +//! [`ClusteredSingleAggregateStream`] for details. //! //! Related configuration: //! @@ -81,7 +81,7 @@ //! input //! ``` //! -//! See [`SingleHashAggregateStream`] and [`OrderedSingleAggregateStream`] for +//! See [`SingleHashAggregateStream`] and [`ClusteredSingleAggregateStream`] for //! details. //! //! Related configuration: @@ -153,12 +153,12 @@ use std::sync::Arc; use super::{DisplayAs, ExecutionPlanProperties, PlanProperties}; use crate::aggregates::{ aggregate_stream::AggregateStream, + clustered_final_stream::ClusteredFinalAggregateStream, + clustered_partial_stream::ClusteredPartialAggregateStream, + clustered_single_stream::ClusteredSingleAggregateStream, grouped_hash_stream::GroupedHashAggregateStream, grouped_topk_stream::GroupedTopKAggregateStream, hash_stream::{FinalHashAggregateStream, PartialHashAggregateStream}, - ordered_final_stream::OrderedFinalAggregateStream, - ordered_partial_stream::OrderedPartialAggregateStream, - ordered_single_stream::OrderedSingleAggregateStream, partial_reduce_stream::PartialReduceHashAggregateStream, single_stream::SingleHashAggregateStream, }; @@ -174,7 +174,7 @@ use crate::statistics::{ChildStats, StatisticsArgs}; use crate::{ChildrenPropertiesMode, ReplaceChildrenOptions, validate_child_count}; use crate::{ DisplayFormatType, Distribution, ExecutionPlan, InputDistributionRequirements, - InputOrderMode, Partitioning, SendableRecordBatchStream, Statistics, + Partitioning, SendableRecordBatchStream, Statistics, }; use datafusion_common::config::ConfigOptions; use parking_lot::Mutex; @@ -207,19 +207,20 @@ use datafusion_physical_expr_common::sort_expr::{ use datafusion_expr::utils::AggregateOrderSensitivity; use datafusion_physical_expr_common::utils::evaluate_expressions_to_arrays; use itertools::Itertools; +pub use order::GroupCompletionMode; use topk::hash_table::is_supported_hash_key_type; use topk::heap::is_supported_heap_type; mod aggregate_hash_table; mod aggregate_stream; +mod clustered_final_stream; +mod clustered_partial_stream; +mod clustered_single_stream; pub mod group_values; mod grouped_hash_stream; mod grouped_topk_stream; mod hash_stream; pub mod order; -mod ordered_final_stream; -mod ordered_partial_stream; -mod ordered_single_stream; mod partial_reduce_stream; mod single_stream; mod skip_partial; @@ -691,12 +692,12 @@ enum StreamType { /// Single stage of the hash aggregation /// Input output scheme: initial input -> final result SingleHash(SingleHashAggregateStream), - /// Partial stage of aggregation for ordered input. - OrderedPartialAggregate(OrderedPartialAggregateStream), - /// Final stage of aggregation for ordered input. - OrderedFinalAggregate(OrderedFinalAggregateStream), - /// Single stage of aggregation for ordered input. - OrderedSingleAggregate(OrderedSingleAggregateStream), + /// Partial stage of aggregation for clustered input. + ClusteredPartialAggregate(ClusteredPartialAggregateStream), + /// Final stage of aggregation for clustered input. + ClusteredFinalAggregate(ClusteredFinalAggregateStream), + /// Single stage of aggregation for clustered input. + ClusteredSingleAggregate(ClusteredSingleAggregateStream), /// Legacy hash aggregation reused for multiple stages /// /// Every path it handles now has a dedicated stream, so this variant is only @@ -719,9 +720,9 @@ impl From for SendableRecordBatchStream { StreamType::PartialReduceHash(stream) => Box::pin(stream), StreamType::FinalHash(stream) => stream.into_stream(), StreamType::SingleHash(stream) => stream.into_stream(), - StreamType::OrderedPartialAggregate(stream) => stream.into_stream(), - StreamType::OrderedFinalAggregate(stream) => stream.into_stream(), - StreamType::OrderedSingleAggregate(stream) => Box::pin(stream), + StreamType::ClusteredPartialAggregate(stream) => stream.into_stream(), + StreamType::ClusteredFinalAggregate(stream) => stream.into_stream(), + StreamType::ClusteredSingleAggregate(stream) => Box::pin(stream), StreamType::GroupedHash(stream) => Box::pin(stream), StreamType::GroupedPriorityQueue(stream) => Box::pin(stream), } @@ -910,12 +911,12 @@ pub struct AggregateExec { /// Execution metrics metrics: ExecutionPlanMetricsSet, required_input_ordering: Option, - /// Describes how the input is ordered relative to the group by columns + /// Describes when the executor can determine that groups are complete. /// - /// This field is also overloaded to mean "the output MUST preserve this - /// input order". When that is not possible, the constructor overwrites it - /// with the unordered variant [`InputOrderMode::Linear`]. - input_order_mode: InputOrderMode, + /// Input ordering describes a subset of the cases in which groups can be + /// safely emitted before the input ends. Full group completion requires only + /// that rows for each complete grouping tuple are contiguous. + group_completion_mode: GroupCompletionMode, cache: Arc, /// During initialization, if the plan supports dynamic filtering (see [`AggrDynFilter`]), /// it is set to `Some(..)` regardless of whether it can be pushed down to a child node. @@ -1106,7 +1107,7 @@ impl AggregateExec { // Commit the kind and properties together: heap output is unordered and final. self.kind = kind; - self.input_order_mode = InputOrderMode::Linear; + self.group_completion_mode = GroupCompletionMode::None; self.required_input_ordering = None; // Keep unchanged properties so parent aggregates do not need rebuilding. if !self.cache.eq_properties.oeq_class().is_empty() @@ -1331,21 +1332,21 @@ impl AggregateExec { .iter() .filter(|expr| input_eq_properties.is_expr_constant(expr).is_none()) .count(); - let mut input_order_mode = if indices.len() == num_non_constant_groupby_exprs + let mut group_completion_mode = if indices.len() == num_non_constant_groupby_exprs && !indices.is_empty() && group_by.groups.len() == 1 { - InputOrderMode::Sorted + GroupCompletionMode::Full } else if !indices.is_empty() { - InputOrderMode::PartiallySorted(indices) + GroupCompletionMode::Partial(indices) } else { - InputOrderMode::Linear + GroupCompletionMode::None }; - // Input order mode is also used to advertise plan output ordering, grouping - // sets handling, and partial reduce aggregation can't promise that. + // Grouping sets can change group keys. PartialReduce combines intermediate + // states without using group boundaries to recognize completed groups. if group_by.has_grouping_set() || mode == AggregateMode::PartialReduce { - input_order_mode = InputOrderMode::Linear; + group_completion_mode = GroupCompletionMode::None; } // construct a map from the input expression to the output expression of the Aggregation group by @@ -1361,7 +1362,7 @@ impl AggregateExec { &group_expr_mapping, group_by.is_true_no_grouping(), &mode, - &input_order_mode, + &group_completion_mode, aggr_expr.as_ref(), )? }; @@ -1378,7 +1379,7 @@ impl AggregateExec { input_schema, metrics: ExecutionPlanMetricsSet::new(), required_input_ordering, - input_order_mode, + group_completion_mode, cache: Arc::new(cache), dynamic_filter: None, }; @@ -1683,47 +1684,51 @@ impl AggregateExec { ); } - // Choose the execution path based on (aggregation mode, ordering). - // - // Note that `self.input_order_mode` represents both input ordering and output - // order promise. See its comment for details. + // Choose the execution path based on aggregation mode and when groups + // are known to be complete. use AggregateMode::*; - use InputOrderMode::*; - let stream = match (self.mode, &self.input_order_mode) { - (Partial, Linear) => StreamType::PartialHash( + let stream = match (self.mode, &self.group_completion_mode) { + (Partial, GroupCompletionMode::None) => StreamType::PartialHash( PartialHashAggregateStream::new(self, context, partition)?, ), - (Partial, Sorted | PartiallySorted(_)) => { - StreamType::OrderedPartialAggregate(OrderedPartialAggregateStream::new( - self, context, partition, - )?) + (Partial, GroupCompletionMode::Partial(_) | GroupCompletionMode::Full) => { + StreamType::ClusteredPartialAggregate( + ClusteredPartialAggregateStream::new(self, context, partition)?, + ) } - (PartialReduce, Linear) => StreamType::PartialReduceHash( + (PartialReduce, GroupCompletionMode::None) => StreamType::PartialReduceHash( PartialReduceHashAggregateStream::new(self, context, partition)?, ), - (PartialReduce, Sorted | PartiallySorted(_)) => { - // See the comment above: the builder enforces `Linear` order for - // `PartialReduce` mode. + ( + PartialReduce, + GroupCompletionMode::Partial(_) | GroupCompletionMode::Full, + ) => { return internal_err!( - "PartialReduce aggregation must use InputOrderMode::Linear" + "PartialReduce aggregation must use GroupCompletionMode::None" ); } - (Final | FinalPartitioned, Linear) => StreamType::FinalHash( - FinalHashAggregateStream::new(self, context, partition)?, - ), - (Final | FinalPartitioned, Sorted | PartiallySorted(_)) => { - StreamType::OrderedFinalAggregate(OrderedFinalAggregateStream::new( + (Final | FinalPartitioned, GroupCompletionMode::None) => { + StreamType::FinalHash(FinalHashAggregateStream::new( self, context, partition, )?) } - (Single | SinglePartitioned, Linear) => StreamType::SingleHash( - SingleHashAggregateStream::new(self, context, partition)?, - ), - (Single | SinglePartitioned, Sorted | PartiallySorted(_)) => { - StreamType::OrderedSingleAggregate(OrderedSingleAggregateStream::new( + ( + Final | FinalPartitioned, + GroupCompletionMode::Partial(_) | GroupCompletionMode::Full, + ) => StreamType::ClusteredFinalAggregate(ClusteredFinalAggregateStream::new( + self, context, partition, + )?), + (Single | SinglePartitioned, GroupCompletionMode::None) => { + StreamType::SingleHash(SingleHashAggregateStream::new( self, context, partition, )?) } + ( + Single | SinglePartitioned, + GroupCompletionMode::Partial(_) | GroupCompletionMode::Full, + ) => StreamType::ClusteredSingleAggregate( + ClusteredSingleAggregateStream::new(self, context, partition)?, + ), }; Ok(stream) } @@ -1781,7 +1786,7 @@ impl AggregateExec { group_expr_mapping: &ProjectionMapping, is_true_no_grouping: bool, mode: &AggregateMode, - input_order_mode: &InputOrderMode, + group_completion_mode: &GroupCompletionMode, aggr_exprs: &[Arc], ) -> Result { // Construct equivalence properties: @@ -1789,9 +1794,11 @@ impl AggregateExec { .equivalence_properties() .project(group_expr_mapping, schema); - // An aggregation that does not maintain its input order must not - // propegrate the input's ordering either, match `maintains_input_order` value - if *input_order_mode == InputOrderMode::Linear { + // Only the clustered paths preserve existing ordering on group keys. + // Project the input's actual sort expressions; completion alone does + // not establish an output ordering. Keep this consistent with + // `maintains_input_order`. + if *group_completion_mode == GroupCompletionMode::None { eq_properties.clear_orderings(); } @@ -1839,7 +1846,7 @@ impl AggregateExec { }; // TODO: Emission type and boundedness information can be enhanced here - let emission_type = if *input_order_mode == InputOrderMode::Linear { + let emission_type = if *group_completion_mode == GroupCompletionMode::None { EmissionType::Final } else { input.pipeline_behavior() @@ -1869,8 +1876,12 @@ impl AggregateExec { ) } - pub fn input_order_mode(&self) -> &InputOrderMode { - &self.input_order_mode + /// Describes when groups can be completed before the input ends. + /// + /// This does not imply a sort order. See [`ExecutionPlanProperties::output_ordering`] + /// for the ordering of the aggregate's output. + pub fn group_completion_mode(&self) -> &GroupCompletionMode { + &self.group_completion_mode } /// Estimates output statistics for this aggregate node. @@ -2354,8 +2365,12 @@ impl DisplayAs for AggregateExec { write!(f, ", lim=[{}]", config.limit)?; } - if self.input_order_mode != InputOrderMode::Linear { - write!(f, ", ordering_mode={:?}", self.input_order_mode)?; + if self.group_completion_mode != GroupCompletionMode::None { + write!( + f, + ", group_completion_mode={:?}", + self.group_completion_mode + )?; } } DisplayFormatType::TreeRender => { @@ -2481,17 +2496,10 @@ impl ExecutionPlan for AggregateExec { vec![self.required_input_ordering.clone()] } - /// The output ordering of [`AggregateExec`] is determined by its `group_by` - /// columns. Although this method is not explicitly used by any optimizer - /// rules yet, overriding the default implementation ensures that it - /// accurately reflects the actual behavior. - /// - /// If the [`InputOrderMode`] is `Linear`, the `group_by` columns don't have - /// an ordering, which means the results do not either. However, in the - /// `Ordered` and `PartiallyOrdered` cases, the `group_by` columns do have - /// an ordering, which is preserved in the output. + /// Clustered aggregation preserves existing ordering on the group-by + /// columns. Aggregate result columns do not inherit input ordering. fn maintains_input_order(&self) -> Vec { - vec![self.input_order_mode != InputOrderMode::Linear] + vec![self.group_completion_mode != GroupCompletionMode::None] } fn children(&self) -> Vec<&Arc> { @@ -2753,7 +2761,7 @@ impl ExecutionPlan for AggregateExec { // Derived at construction from the input ordering and `group_by`. required_input_ordering: _, // Derived at construction from the input ordering and `group_by`. - input_order_mode: _, + group_completion_mode: _, // Derived at construction by `Self::compute_properties`. cache: _, dynamic_filter, @@ -4163,11 +4171,11 @@ mod tests { match (mode, ordered, &stream) { (AggregateMode::Final, false, StreamType::FinalHash(_)) | (AggregateMode::Single, false, StreamType::SingleHash(_)) => {} - (AggregateMode::Final, true, StreamType::OrderedFinalAggregate(_)) - | (AggregateMode::Single, true, StreamType::OrderedSingleAggregate(_)) => { + (AggregateMode::Final, true, StreamType::ClusteredFinalAggregate(_)) + | (AggregateMode::Single, true, StreamType::ClusteredSingleAggregate(_)) => { assert_eq!( - aggregate.input_order_mode(), - &InputOrderMode::PartiallySorted(vec![0]) + aggregate.group_completion_mode(), + &GroupCompletionMode::Partial(vec![0]) ); } _ => panic!("unexpected stream for {mode:?}, ordered={ordered}"), @@ -5109,12 +5117,12 @@ mod tests { let stream = aggregate.execute_typed(0, &fits)?; match (mode, ordered, &stream) { (AggregateMode::Partial, false, StreamType::PartialHash(_)) - | (AggregateMode::Partial, true, StreamType::OrderedPartialAggregate(_)) + | (AggregateMode::Partial, true, StreamType::ClusteredPartialAggregate(_)) | (AggregateMode::PartialReduce, false, StreamType::PartialReduceHash(_)) | (AggregateMode::Final, false, StreamType::FinalHash(_)) - | (AggregateMode::Final, true, StreamType::OrderedFinalAggregate(_)) + | (AggregateMode::Final, true, StreamType::ClusteredFinalAggregate(_)) | (AggregateMode::Single, false, StreamType::SingleHash(_)) - | (AggregateMode::Single, true, StreamType::OrderedSingleAggregate(_)) => {} + | (AggregateMode::Single, true, StreamType::ClusteredSingleAggregate(_)) => {} _ => panic!("unexpected stream for {mode:?}, ordered={ordered}"), } let reserved = fits.memory_pool().reserved(); @@ -5602,7 +5610,116 @@ mod tests { Ok(()) } - /// Ensures `OrderedSingleAggregateStream` is used for ordered raw input. + #[rstest::rstest] + #[case::full(true)] + #[case::partial(false)] + fn group_completion_preserves_sort_options( + #[case] full: bool, + #[values(AggregateMode::Partial, AggregateMode::Single)] mode: AggregateMode, + #[values(false, true)] descending: bool, + #[values(false, true)] nulls_first: bool, + ) -> Result<()> { + let schema = Arc::new(Schema::new(vec![ + Field::new("a", DataType::Int32, true), + Field::new("b", DataType::Int32, true), + Field::new("value", DataType::Int64, true), + ])); + let options = SortOptions::new(descending, nulls_first); + let mut input_ordering = vec![PhysicalSortExpr::new(col("a", &schema)?, options)]; + let mut output_ordering = vec![PhysicalSortExpr::new( + Arc::new(Column::new("group_a", 1)), + options, + )]; + if full { + input_ordering.push(PhysicalSortExpr::new(col("b", &schema)?, options)); + output_ordering.push(PhysicalSortExpr::new( + Arc::new(Column::new("group_b", 0)), + options, + )); + } + let input_ordering = LexOrdering::new(input_ordering).unwrap(); + let input = TestMemoryExec::try_new(&[vec![]], Arc::clone(&schema), None)? + .try_with_sort_information(vec![input_ordering.clone()])?; + let aggregate = AggregateExec::try_new( + mode, + PhysicalGroupBy::new_single(vec![ + (col("b", &schema)?, "group_b".to_string()), + (col("a", &schema)?, "group_a".to_string()), + ]), + vec![Arc::new( + AggregateExprBuilder::new(sum_udaf(), vec![col("value", &schema)?]) + .schema(Arc::clone(&schema)) + .alias("SUM(value)") + .build()?, + )], + vec![None], + Arc::new(TestMemoryExec::update_cache(&Arc::new(input))), + schema, + )?; + + let expected_mode = if full { + GroupCompletionMode::Full + } else { + GroupCompletionMode::Partial(vec![1]) + }; + assert_eq!(aggregate.group_completion_mode(), &expected_mode); + assert_eq!(aggregate.maintains_input_order(), vec![true]); + assert_eq!( + aggregate.properties().emission_type, + EmissionType::Incremental + ); + assert_eq!( + aggregate.properties().output_ordering(), + LexOrdering::new(output_ordering).as_ref() + ); + assert_eq!( + aggregate.required_input_ordering(), + vec![Some(OrderingRequirements::new_soft(input_ordering.into()))] + ); + Ok(()) + } + + #[test] + fn group_completion_is_recomputed_with_new_children() -> Result<()> { + let aggregate = Arc::new(single_test_aggregate()?); + let original_input = Arc::clone(aggregate.input()); + let schema = original_input.schema(); + let ordering = + LexOrdering::new([PhysicalSortExpr::new_default(col("a", &schema)?)]) + .unwrap(); + let sorted_input = TestMemoryExec::try_new(&[vec![]], schema, None)? + .try_with_sort_information(vec![ordering.clone()])?; + let sorted_input = + Arc::new(TestMemoryExec::update_cache(&Arc::new(sorted_input))); + + let sorted = aggregate.replace_children( + vec![sorted_input], + ReplaceChildrenOptions::new(ChildrenPropertiesMode::Recompute), + )?; + let sorted_aggregate = sorted.downcast_ref::().unwrap(); + assert_eq!( + sorted_aggregate.group_completion_mode(), + &GroupCompletionMode::Full + ); + assert_eq!(sorted.output_ordering(), Some(&ordering)); + assert_eq!(sorted.pipeline_behavior(), EmissionType::Incremental); + + let unordered = sorted.replace_children( + vec![original_input], + ReplaceChildrenOptions::new(ChildrenPropertiesMode::Recompute), + )?; + let unordered_aggregate = unordered.downcast_ref::().unwrap(); + assert_eq!( + unordered_aggregate.group_completion_mode(), + &GroupCompletionMode::None + ); + assert_eq!(unordered.maintains_input_order(), vec![false]); + assert!(unordered.output_ordering().is_none()); + assert_eq!(unordered.pipeline_behavior(), EmissionType::Final); + Ok(()) + } + + /// Ensures `ClusteredSingleAggregateStream` is used for ordered raw input. #[tokio::test] async fn ordered_single_aggregate_planning() -> Result<()> { let schema = Arc::new(Schema::new(vec![ @@ -5651,14 +5768,14 @@ mod tests { input, Arc::clone(&schema), )?; - assert!(matches!( - aggregate.input_order_mode(), - InputOrderMode::PartiallySorted(_) - )); + assert_eq!( + aggregate.group_completion_mode(), + &GroupCompletionMode::Partial(vec![0]) + ); let task_ctx = new_migrated_hash_ctx(2); let stream = aggregate.execute_typed(0, &task_ctx)?; - assert!(matches!(stream, StreamType::OrderedSingleAggregate(_))); + assert!(matches!(stream, StreamType::ClusteredSingleAggregate(_))); let stream: SendableRecordBatchStream = stream.into(); let output = collect(stream).await?; assert_snapshot!(batches_to_sort_string(&output), @r" @@ -5674,7 +5791,7 @@ mod tests { let finite_memory_task_ctx = new_finite_memory_migrated_hash_ctx(2, 1024 * 1024)?; let stream = aggregate.execute_typed(0, &finite_memory_task_ctx)?; - assert!(matches!(stream, StreamType::OrderedSingleAggregate(_))); + assert!(matches!(stream, StreamType::ClusteredSingleAggregate(_))); Ok(()) } @@ -5711,7 +5828,10 @@ mod tests { Arc::clone(&schema), )?; - assert_eq!(partial_reduce.input_order_mode(), &InputOrderMode::Linear); + assert_eq!( + partial_reduce.group_completion_mode(), + &GroupCompletionMode::None + ); assert_eq!(partial_reduce.maintains_input_order(), vec![false]); assert!( partial_reduce.properties().output_ordering().is_none(), @@ -5766,7 +5886,10 @@ mod tests { None, )?; let aggregate = build_aggregate(unordered_input)?; - assert_eq!(aggregate.input_order_mode(), &InputOrderMode::Linear); + assert_eq!( + aggregate.group_completion_mode(), + &GroupCompletionMode::None + ); assert_eq!( aggregate.schema().as_ref(), &Schema::new(vec![ @@ -5798,7 +5921,10 @@ mod tests { let ordered_input = Arc::new(TestMemoryExec::update_cache(&Arc::new(ordered_input))); let aggregate = build_aggregate(ordered_input)?; - assert_eq!(aggregate.input_order_mode(), &InputOrderMode::Sorted); + assert_eq!( + aggregate.group_completion_mode(), + &GroupCompletionMode::Full + ); Ok(()) } @@ -6116,7 +6242,10 @@ mod tests { schema, )?; - assert_eq!(aggregate.input_order_mode(), &InputOrderMode::Linear); + assert_eq!( + aggregate.group_completion_mode(), + &GroupCompletionMode::None + ); // This captures the behavior before #24438. When the source can declare // `(key, time_bin)` group-contiguous, the corresponding case can use // `EmissionType::Incremental`. @@ -6141,7 +6270,7 @@ mod tests { Ok(()) } - /// Ensures for ordered input, `OrderedPartialAggregateStream` is used. + /// Ensures for ordered input, `ClusteredPartialAggregateStream` is used. #[tokio::test] async fn ordered_partial_aggregate_planning() -> Result<()> { let schema = Arc::new(Schema::new(vec![ @@ -6195,13 +6324,13 @@ mod tests { Arc::clone(&schema), )?; assert!(matches!( - aggregate.input_order_mode(), - InputOrderMode::PartiallySorted(_) + aggregate.group_completion_mode(), + GroupCompletionMode::Partial(_) )); let task_ctx = new_migrated_hash_ctx(2); let stream = aggregate.execute_typed(0, &task_ctx)?; - assert!(matches!(stream, StreamType::OrderedPartialAggregate(_))); + assert!(matches!(stream, StreamType::ClusteredPartialAggregate(_))); let stream: SendableRecordBatchStream = stream.into(); let output = collect(stream).await?; @@ -6217,15 +6346,15 @@ mod tests { +----------+-----------+-------------------------+ "); - // Ordered partial aggregation supports finite memory. + // Clustered partial aggregation supports finite memory. let finite_memory_task_ctx = new_finite_memory_migrated_hash_ctx(2, 1024 * 1024)?; let stream = aggregate.execute_typed(0, &finite_memory_task_ctx)?; - assert!(matches!(stream, StreamType::OrderedPartialAggregate(_))); + assert!(matches!(stream, StreamType::ClusteredPartialAggregate(_))); Ok(()) } - /// Ensures for ordered input, `OrderedFinalAggregateStream` is used. + /// Ensures for ordered input, `ClusteredFinalAggregateStream` is used. #[tokio::test] async fn ordered_final_aggregate_planning() -> Result<()> { let schema = Arc::new(Schema::new(vec![ @@ -6276,11 +6405,14 @@ mod tests { final_input, Arc::clone(&schema), )?; - assert_eq!(final_aggregate.input_order_mode(), &InputOrderMode::Sorted); + assert_eq!( + final_aggregate.group_completion_mode(), + &GroupCompletionMode::Full + ); let task_ctx = new_migrated_hash_ctx(2); let stream = final_aggregate.execute_typed(0, &task_ctx)?; - assert!(matches!(stream, StreamType::OrderedFinalAggregate(_))); + assert!(matches!(stream, StreamType::ClusteredFinalAggregate(_))); let stream: SendableRecordBatchStream = stream.into(); let output = collect(stream).await?; @@ -6295,10 +6427,10 @@ mod tests { +-----+--------------+ "); - // Ordered final aggregation supports finite memory. + // Clustered final aggregation supports finite memory. let finite_memory_task_ctx = new_finite_memory_migrated_hash_ctx(2, 1024 * 1024)?; let stream = final_aggregate.execute_typed(0, &finite_memory_task_ctx)?; - assert!(matches!(stream, StreamType::OrderedFinalAggregate(_))); + assert!(matches!(stream, StreamType::ClusteredFinalAggregate(_))); Ok(()) } @@ -6350,8 +6482,8 @@ mod tests { Arc::clone(&schema), )?; assert!(matches!( - aggregate.input_order_mode(), - InputOrderMode::PartiallySorted(_) + aggregate.group_completion_mode(), + GroupCompletionMode::Partial(_) )); let runtime = RuntimeEnvBuilder::default() @@ -6368,7 +6500,7 @@ mod tests { ); let mut stream: SendableRecordBatchStream = - OrderedPartialAggregateStream::new(&aggregate, &task_ctx, 0)?.into_stream(); + ClusteredPartialAggregateStream::new(&aggregate, &task_ctx, 0)?.into_stream(); while let Some(result) = stream.next().await { if let Err(e) = result { @@ -6649,7 +6781,7 @@ mod tests { // // "AggregateExec: mode=Final, gby=[a@0 as a], aggr=[FIRST_VALUE(b)]", // " CoalescePartitionsExec", - // " AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[FIRST_VALUE(b)], ordering_mode=None", + // " AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[FIRST_VALUE(b)], group_completion_mode=None", // " DataSourceExec: partitions=4, partition_sizes=[1, 1, 1, 1]", // // and checks whether the function `merge_batch` works correctly for diff --git a/datafusion/physical-plan/src/aggregates/order/full.rs b/datafusion/physical-plan/src/aggregates/order/full.rs index ca818d6a2d598..30909f482f390 100644 --- a/datafusion/physical-plan/src/aggregates/order/full.rs +++ b/datafusion/physical-plan/src/aggregates/order/full.rs @@ -18,10 +18,10 @@ use datafusion_expr::EmitTo; use std::mem::size_of; -/// Tracks grouping state when the data is ordered entirely by its -/// group keys +/// Tracks group completion when rows are contiguous for the complete +/// grouping tuple. /// -/// When the group values are sorted, as soon as we see group `n+1` we +/// When groups are contiguous, as soon as we see group `n+1` we /// know we will never see any rows for group `n` again and thus they /// can be emitted. /// @@ -55,7 +55,7 @@ use std::mem::size_of; /// `0..12` can be emitted. Note that `13` can not yet be emitted as /// there may be more values in the next batch with the same group_id. #[derive(Debug)] -pub struct GroupOrderingFull { +pub struct GroupCompletionFull { state: State, } @@ -72,7 +72,7 @@ enum State { Complete, } -impl GroupOrderingFull { +impl GroupCompletionFull { pub fn new() -> Self { Self { state: State::Start, @@ -115,13 +115,13 @@ impl GroupOrderingFull { self.state = State::Complete; } - /// Starts tracking a new fully ordered input segment. + /// Starts tracking a new input segment with contiguous groups. pub fn reset(&mut self) { self.state = State::Start; } /// Called when new groups are added in a batch. See documentation - /// on [`super::GroupOrdering::new_groups`] + /// on [`super::GroupCompletion::new_groups`] pub fn new_groups(&mut self, total_num_groups: usize) { assert_ne!(total_num_groups, 0); @@ -149,7 +149,7 @@ impl GroupOrderingFull { } } -impl Default for GroupOrderingFull { +impl Default for GroupCompletionFull { fn default() -> Self { Self::new() } diff --git a/datafusion/physical-plan/src/aggregates/order/mod.rs b/datafusion/physical-plan/src/aggregates/order/mod.rs index 147e4e0d6f18d..b3a756d7739ed 100644 --- a/datafusion/physical-plan/src/aggregates/order/mod.rs +++ b/datafusion/physical-plan/src/aggregates/order/mod.rs @@ -24,48 +24,85 @@ use datafusion_expr::EmitTo; mod full; mod partial; -use crate::InputOrderMode; -pub use full::GroupOrderingFull; -pub use partial::GroupOrderingPartial; +pub use full::GroupCompletionFull; +pub use partial::GroupCompletionPartial; -/// Ordering information for each group in the hash table +/// Describes how an aggregate can determine that groups are complete. +/// +/// Input ordering is one way to establish a group-completion mode, but the +/// execution machinery only needs to know when it can safely emit completed +/// groups. This mode does not describe the sort order of the input or output. +/// +/// For example, when grouping by `key`, both inputs have fully contiguous +/// groups within the input partition: +/// +/// ```text +/// sorted: A A B B C C +/// not sorted: C C A A B B +/// ``` +/// +/// In both cases, once the key changes, the previous key will not appear again, +/// so its group is complete and can be emitted. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum GroupCompletionMode { + /// No group can be known complete before the input ends. + None, + /// Rows with the same values at these grouping-expression indices form one + /// contiguous range. When those values change, every group in the previous + /// range is complete and can be emitted. + /// + /// For example, with `GROUP BY (a, b)`, `Partial(vec![0])` means all rows + /// for each value of `a` are contiguous, while an `(a, b)` tuple may recur + /// within that range. + Partial(Vec), + /// Rows with the same complete grouping tuple form one contiguous range. + /// When the tuple changes, the previous group can be emitted. + Full, +} + +/// Tracks when groups in the hash table are complete and can be emitted. #[derive(Debug)] -pub enum GroupOrdering { - /// Groups are not ordered +pub enum GroupCompletion { + /// No group can be known complete before the input ends. None, - /// Groups are ordered by some pre-set of the group keys - Partial(GroupOrderingPartial), - /// Groups are entirely contiguous, - Full(GroupOrderingFull), + /// Rows are contiguous for a subset of the grouping keys. + /// When those key values change, all groups in the previous run + /// are complete and can be emitted. + Partial(GroupCompletionPartial), + /// Rows are contiguous for the complete grouping tuple. + /// When the tuple changes, the previous group can be emitted. + Full(GroupCompletionFull), } -impl GroupOrdering { - /// Create a `GroupOrdering` for the specified ordering - pub fn try_new(mode: &InputOrderMode) -> Result { +impl GroupCompletion { + /// Create a `GroupCompletion` for the specified group-completion mode. + pub fn try_new(mode: &GroupCompletionMode) -> Result { match mode { - InputOrderMode::Linear => Ok(GroupOrdering::None), - InputOrderMode::PartiallySorted(order_indices) => { - GroupOrderingPartial::try_new(order_indices.clone()) - .map(GroupOrdering::Partial) + GroupCompletionMode::None => Ok(GroupCompletion::None), + GroupCompletionMode::Partial(grouping_indices) => { + GroupCompletionPartial::try_new(grouping_indices.clone()) + .map(GroupCompletion::Partial) + } + GroupCompletionMode::Full => { + Ok(GroupCompletion::Full(GroupCompletionFull::new())) } - InputOrderMode::Sorted => Ok(GroupOrdering::Full(GroupOrderingFull::new())), } } - /// Returns how many groups can be emitted while respecting the current - /// ordering guarantees, or `None` if no data can be emitted. + /// Returns how many completed groups can be emitted, or `None` if no data + /// can be emitted. pub fn emit_to(&self) -> Option { match self { - GroupOrdering::None => None, - GroupOrdering::Partial(partial) => partial.emit_to(), - GroupOrdering::Full(full) => full.emit_to(), + GroupCompletion::None => None, + GroupCompletion::Partial(partial) => partial.emit_to(), + GroupCompletion::Full(full) => full.emit_to(), } } /// Returns the emit strategy to use under memory pressure (OOM). /// /// Returns the strategy that must be used when emitting up to `n` groups - /// while respecting the current ordering guarantees. + /// while respecting the configured group-completion mode. /// /// Returns `None` if no data can be emitted. pub fn oom_emit_to(&self, n: usize) -> Option { @@ -74,8 +111,8 @@ impl GroupOrdering { } match self { - GroupOrdering::None => Some(EmitTo::First(n)), - GroupOrdering::Partial(_) | GroupOrdering::Full(_) => { + GroupCompletion::None => Some(EmitTo::First(n)), + GroupCompletion::Partial(_) | GroupCompletion::Full(_) => { self.emit_to().map(|emit_to| match emit_to { EmitTo::First(max) => EmitTo::First(n.min(max)), EmitTo::All => EmitTo::First(n), @@ -87,23 +124,23 @@ impl GroupOrdering { /// Updates the state to indicate that the input is complete. pub fn input_done(&mut self) { match self { - GroupOrdering::None => {} - GroupOrdering::Partial(partial) => partial.input_done(), - GroupOrdering::Full(full) => full.input_done(), + GroupCompletion::None => {} + GroupCompletion::Partial(partial) => partial.input_done(), + GroupCompletion::Full(full) => full.input_done(), } } - /// Resets the ordering state while preserving the configured ordering mode. + /// Resets the completion state while preserving the configured mode. /// - /// Ordered partial aggregation uses this after passing intermediate states - /// downstream, and ordered final aggregation uses it after spilling a run. + /// Clustered partial aggregation uses this after passing intermediate states + /// downstream, and clustered final aggregation uses it after spilling a run. /// In both cases the hash table is empty and can start tracking the next - /// input batch from a fresh ordering state. + /// input batch from a fresh completion state. pub fn reset(&mut self) { match self { - GroupOrdering::None => {} - GroupOrdering::Partial(partial) => partial.reset(), - GroupOrdering::Full(full) => full.reset(), + GroupCompletion::None => {} + GroupCompletion::Partial(partial) => partial.reset(), + GroupCompletion::Full(full) => full.reset(), } } @@ -111,9 +148,9 @@ impl GroupOrdering { /// existing indexes down by `n`. pub fn remove_groups(&mut self, n: usize) { match self { - GroupOrdering::None => {} - GroupOrdering::Partial(partial) => partial.remove_groups(n), - GroupOrdering::Full(full) => full.remove_groups(n), + GroupCompletion::None => {} + GroupCompletion::Partial(partial) => partial.remove_groups(n), + GroupCompletion::Full(full) => full.remove_groups(n), } } @@ -132,28 +169,28 @@ impl GroupOrdering { total_num_groups: usize, ) -> Result<()> { match self { - GroupOrdering::None => {} - GroupOrdering::Partial(partial) => { + GroupCompletion::None => {} + GroupCompletion::Partial(partial) => { partial.new_groups( batch_group_values, group_indices, total_num_groups, )?; } - GroupOrdering::Full(full) => { + GroupCompletion::Full(full) => { full.new_groups(total_num_groups); } } Ok(()) } - /// Returns the size of memory used by the ordering state, in bytes. + /// Returns the size of memory used by the completion state, in bytes. pub fn size(&self) -> usize { size_of::() + match self { - GroupOrdering::None => 0, - GroupOrdering::Partial(partial) => partial.size(), - GroupOrdering::Full(full) => full.size(), + GroupCompletion::None => 0, + GroupCompletion::Partial(partial) => partial.size(), + GroupCompletion::Full(full) => full.size(), } } } @@ -167,52 +204,52 @@ mod tests { use arrow::array::Int32Array; #[test] - fn test_oom_emit_to_none_ordering() { - let group_ordering = GroupOrdering::None; + fn test_oom_emit_to_none_completion() { + let group_completion = GroupCompletion::None; - assert_eq!(group_ordering.oom_emit_to(0), None); - assert_eq!(group_ordering.oom_emit_to(5), Some(EmitTo::First(5))); + assert_eq!(group_completion.oom_emit_to(0), None); + assert_eq!(group_completion.oom_emit_to(5), Some(EmitTo::First(5))); } - /// Creates a partially ordered grouping state with three groups. + /// Creates a partial group-completion tracker with three groups. /// - /// `sort_key_values` controls whether a sort boundary exists in the batch: + /// `group_key_values` controls whether a run boundary exists in the batch: /// distinct values such as `[1, 2, 3]` create boundaries, while repeated /// values such as `[1, 1, 1]` do not. - fn partial_ordering(sort_key_values: Vec) -> Result { - let mut group_ordering = - GroupOrdering::Partial(GroupOrderingPartial::try_new(vec![0])?); + fn partial_completion(group_key_values: Vec) -> Result { + let mut group_completion = + GroupCompletion::Partial(GroupCompletionPartial::try_new(vec![0])?); let batch_group_values: Vec = vec![ - Arc::new(Int32Array::from(sort_key_values)), + Arc::new(Int32Array::from(group_key_values)), Arc::new(Int32Array::from(vec![10, 20, 30])), ]; let group_indices = vec![0, 1, 2]; - group_ordering.new_groups(&batch_group_values, &group_indices, 3)?; + group_completion.new_groups(&batch_group_values, &group_indices, 3)?; - Ok(group_ordering) + Ok(group_completion) } #[test] fn test_oom_emit_to_partial_clamps_to_boundary() -> Result<()> { - let group_ordering = partial_ordering(vec![1, 2, 3])?; + let group_completion = partial_completion(vec![1, 2, 3])?; // Can emit both `1` and `2` groups because we have seen `3` - assert_eq!(group_ordering.emit_to(), Some(EmitTo::First(2))); - assert_eq!(group_ordering.oom_emit_to(1), Some(EmitTo::First(1))); - assert_eq!(group_ordering.oom_emit_to(3), Some(EmitTo::First(2))); + assert_eq!(group_completion.emit_to(), Some(EmitTo::First(2))); + assert_eq!(group_completion.oom_emit_to(1), Some(EmitTo::First(1))); + assert_eq!(group_completion.oom_emit_to(3), Some(EmitTo::First(2))); Ok(()) } #[test] fn test_oom_emit_to_partial_without_boundary() -> Result<()> { - let group_ordering = partial_ordering(vec![1, 1, 1])?; + let group_completion = partial_completion(vec![1, 1, 1])?; // Can't emit the last `1` group as it may have more values - assert_eq!(group_ordering.emit_to(), None); - assert_eq!(group_ordering.oom_emit_to(3), None); + assert_eq!(group_completion.emit_to(), None); + assert_eq!(group_completion.oom_emit_to(3), None); Ok(()) } diff --git a/datafusion/physical-plan/src/aggregates/order/partial.rs b/datafusion/physical-plan/src/aggregates/order/partial.rs index 1603bb6d079be..df1637bd77785 100644 --- a/datafusion/physical-plan/src/aggregates/order/partial.rs +++ b/datafusion/physical-plan/src/aggregates/order/partial.rs @@ -27,12 +27,11 @@ use datafusion_common::{Result, ScalarValue}; use datafusion_execution::memory_pool::proxy::VecAllocExt; use datafusion_expr::EmitTo; -/// Tracks grouping state when the data is ordered by some subset of +/// Tracks group completion when rows are contiguous for a subset of /// the group keys. /// -/// Once the next *sort key* value is seen, never see groups with that -/// sort key again, so we can emit all groups with the previous sort -/// key and earlier. +/// Once those key values change, they will not appear again, so all groups +/// in the previous run are complete and can be emitted. /// /// For example, given `SUM(amt) GROUP BY id, state` if the input is /// sorted by `state`, when a new value of `state` is seen, all groups @@ -44,11 +43,11 @@ use datafusion_expr::EmitTo; /// ┏━━━━━━━━━━━━━━━━━┓ ┏━━━━━━━┓ /// ┌─────┐ ┌───────────────────┐ ┌─────┃ 9 ┃ ┃ "MD" ┃ /// │┌───┐│ │ ┌──────────────┐ │ │ ┗━━━━━━━━━━━━━━━━━┛ ┗━━━━━━━┛ -/// ││ 0 ││ │ │ 123, "MA" │ │ │ current_sort sort_key +/// ││ 0 ││ │ │ 123, "MA" │ │ │ current_run_start group_key /// │└───┘│ │ └──────────────┘ │ │ -/// │ ... │ │ ... │ │ current_sort tracks the +/// │ ... │ │ ... │ │ current_run_start tracks the /// │┌───┐│ │ ┌──────────────┐ │ │ smallest group index that had -/// ││ 8 ││ │ │ 765, "MA" │ │ │ the same sort_key as current +/// ││ 8 ││ │ │ 765, "MA" │ │ │ the same group_key as current /// │├───┤│ │ ├──────────────┤ │ │ /// ││ 9 ││ │ │ 923, "MD" │◀─┼─┘ /// │├───┤│ │ ├──────────────┤ │ ┏━━━━━━━━━━━━━━┓ @@ -63,22 +62,22 @@ use datafusion_expr::EmitTo; /// order) recent group index /// ``` #[derive(Debug)] -pub struct GroupOrderingPartial { +pub struct GroupCompletionPartial { /// State machine state: State, - /// The indexes of the group by columns that form the sort key. - /// For example if grouping by `id, state` and ordered by `state` + /// The indexes of the group by columns whose values form contiguous runs. + /// For example if grouping by `id, state` and contiguous on `state` /// this would be `[1]`. - order_indices: Vec, + grouping_indices: Vec, } #[derive(Debug, Default, PartialEq)] enum State { - /// The ordering was temporarily taken. `Self::Taken` is left + /// The state was temporarily taken. `Self::Taken` is left /// when state must be temporarily taken to satisfy the borrow /// checker. If an error happens before the state can be restored, - /// the ordering information is lost and execution can not + /// the completion information is lost and execution can not /// proceed, but there is no undefined behavior. #[default] Taken, @@ -88,10 +87,10 @@ enum State { /// Data is in progress. InProgress { - /// Smallest group index with the sort_key - current_sort: usize, - /// The sort key of group_index `current_sort` - sort_key: Vec, + /// Smallest group index in the current run. + current_run_start: usize, + /// The key values of the current run. + group_key: Vec, /// index of the current group for which values are being /// generated current: usize, @@ -106,7 +105,7 @@ impl State { match self { State::Taken => 0, State::Start => 0, - State::InProgress { sort_key, .. } => sort_key + State::InProgress { group_key, .. } => group_key .iter() .map(|scalar_value| scalar_value.size()) .sum(), @@ -115,24 +114,23 @@ impl State { } } -impl GroupOrderingPartial { - /// TODO: Remove unnecessary `input_schema` parameter. - pub fn try_new(order_indices: Vec) -> Result { - debug_assert!(!order_indices.is_empty()); +impl GroupCompletionPartial { + /// Creates a tracker for runs defined by the specified grouping columns. + pub fn try_new(grouping_indices: Vec) -> Result { + debug_assert!(!grouping_indices.is_empty()); Ok(Self { state: State::Start, - order_indices, + grouping_indices, }) } - /// Select sort keys from the group values + /// Select the keys that define contiguous runs from the group values. /// - /// For example, if group_values had `A, B, C` but the input was - /// only sorted on `B` and `C` this should return rows for (`B`, - /// `C`) - fn compute_sort_keys(&mut self, group_values: &[ArrayRef]) -> Vec { - // Take only the columns that are in the sort key - self.order_indices + /// For example, if `group_values` contains `A, B, C` but the input is + /// contiguous on `(B, C)`, this returns the arrays for `B` and `C`. + fn compute_group_keys(&mut self, group_values: &[ArrayRef]) -> Vec { + // Take only the columns that define contiguous runs. + self.grouping_indices .iter() .map(|&idx| Arc::clone(&group_values[idx])) .collect() @@ -143,14 +141,14 @@ impl GroupOrderingPartial { match &self.state { State::Taken => unreachable!("State previously taken"), State::Start => None, - State::InProgress { current_sort, .. } => { - // Can not emit if we are still on the first row sort - // row otherwise we can emit all groups that had earlier sort keys - // - if *current_sort == 0 { + State::InProgress { + current_run_start, .. + } => { + // The current run is incomplete; only groups from earlier runs can be emitted. + if *current_run_start == 0 { None } else { - Some(EmitTo::First(*current_sort)) + Some(EmitTo::First(*current_run_start)) } } State::Complete => Some(EmitTo::All), @@ -164,15 +162,15 @@ impl GroupOrderingPartial { State::Taken => unreachable!("State previously taken"), State::Start => panic!("invalid state: start"), State::InProgress { - current_sort, + current_run_start, current, - sort_key: _, + group_key: _, } => { // shift indexes down by n assert!(*current >= n); *current -= n; - assert!(*current_sort >= n); - *current_sort -= n; + assert!(*current_run_start >= n); + *current_run_start -= n; } State::Complete => panic!("invalid state: complete"), } @@ -186,31 +184,31 @@ impl GroupOrderingPartial { }; } - /// Starts tracking a new ordered input segment with the same sort-key + /// Starts tracking a new input segment with the same contiguous-key /// columns. pub fn reset(&mut self) { self.state = State::Start; } - fn updated_sort_key( - current_sort: usize, - sort_key: Option>, - range_current_sort: usize, - range_sort_key: Vec, + fn updated_group_key( + current_run_start: usize, + group_key: Option>, + range_current_run_start: usize, + range_group_key: Vec, ) -> Result<(usize, Vec)> { - if let Some(sort_key) = sort_key { - let sort_options = vec![SortOptions::new(false, false); sort_key.len()]; - let ordering = compare_rows(&sort_key, &range_sort_key, &sort_options)?; + if let Some(group_key) = group_key { + let sort_options = vec![SortOptions::new(false, false); group_key.len()]; + let ordering = compare_rows(&group_key, &range_group_key, &sort_options)?; if ordering == Ordering::Equal { - return Ok((current_sort, sort_key)); + return Ok((current_run_start, group_key)); } } - Ok((range_current_sort, range_sort_key)) + Ok((range_current_run_start, range_group_key)) } /// Called when new groups are added in a batch. See documentation - /// on [`super::GroupOrdering::new_groups`] + /// on [`super::GroupCompletion::new_groups`] pub fn new_groups( &mut self, batch_group_values: &[ArrayRef], @@ -222,46 +220,46 @@ impl GroupOrderingPartial { let max_group_index = total_num_groups - 1; - let (current_sort, sort_key) = match std::mem::take(&mut self.state) { + let (current_run_start, group_key) = match std::mem::take(&mut self.state) { State::Taken => unreachable!("State previously taken"), State::Start => (0, None), State::InProgress { - current_sort, - sort_key, + current_run_start, + group_key, .. - } => (current_sort, Some(sort_key)), + } => (current_run_start, Some(group_key)), State::Complete => { panic!("Saw new group after the end of input"); } }; - // Select the sort key columns - let sort_keys = self.compute_sort_keys(batch_group_values); + // Select the columns that define contiguous runs. + let group_keys = self.compute_group_keys(batch_group_values); - // Check if the sort keys indicate a boundary inside the batch - let ranges = partition(&sort_keys)?.ranges(); + // Check if the key values indicate a boundary inside the batch. + let ranges = partition(&group_keys)?.ranges(); let last_range = ranges.last().unwrap(); - let range_current_sort = group_indices[last_range.start]; - let range_sort_key = get_row_at_idx(&sort_keys, last_range.start)?; + let range_current_run_start = group_indices[last_range.start]; + let range_group_key = get_row_at_idx(&group_keys, last_range.start)?; - let (current_sort, sort_key) = if last_range.start == 0 { - // There was no boundary in the batch. Compare with the previous sort_key (if present) + let (current_run_start, group_key) = if last_range.start == 0 { + // There was no boundary in the batch. Compare with the previous group_key (if present) // to check if there was a boundary between the current batch and the previous one. - Self::updated_sort_key( - current_sort, - sort_key, - range_current_sort, - range_sort_key, + Self::updated_group_key( + current_run_start, + group_key, + range_current_run_start, + range_group_key, )? } else { - (range_current_sort, range_sort_key) + (range_current_run_start, range_group_key) }; self.state = State::InProgress { - current_sort, + current_run_start, current: max_group_index, - sort_key, + group_key, }; Ok(()) @@ -269,7 +267,7 @@ impl GroupOrderingPartial { /// Return the size of memory allocated by this structure pub(crate) fn size(&self) -> usize { - size_of::() + self.order_indices.allocated_size() + self.state.size() + size_of::() + self.grouping_indices.allocated_size() + self.state.size() } } @@ -279,79 +277,85 @@ mod tests { use arrow::array::Int32Array; - #[test] - fn test_group_ordering_partial() -> Result<()> { - // Ordered on column a - let order_indices = vec![0]; - let mut group_ordering = GroupOrderingPartial::try_new(order_indices)?; + #[rstest::rstest] + #[case::sorted([1, 2, 3, 4])] + #[case::clustered([3, 1, 4, 2])] + fn test_group_completion_partial(#[case] keys: [i32; 4]) -> Result<()> { + let [first, second, third, fourth] = keys; + // Contiguous on column a. + let grouping_indices = vec![0]; + let mut group_completion = GroupCompletionPartial::try_new(grouping_indices)?; let batch_group_values: Vec = vec![ - Arc::new(Int32Array::from(vec![1, 2, 3])), + Arc::new(Int32Array::from(vec![first, second, third])), Arc::new(Int32Array::from(vec![2, 1, 3])), ]; let group_indices = vec![0, 1, 2]; let total_num_groups = 3; - group_ordering.new_groups( + group_completion.new_groups( &batch_group_values, &group_indices, total_num_groups, )?; assert_eq!( - group_ordering.state, + group_completion.state, State::InProgress { - current_sort: 2, - sort_key: vec![ScalarValue::Int32(Some(3))], + current_run_start: 2, + group_key: vec![ScalarValue::Int32(Some(third))], current: 2 } ); + assert_eq!(group_completion.emit_to(), Some(EmitTo::First(2))); // push without a boundary let batch_group_values: Vec = vec![ - Arc::new(Int32Array::from(vec![3, 3, 3])), + Arc::new(Int32Array::from(vec![third, third, third])), Arc::new(Int32Array::from(vec![2, 1, 7])), ]; let group_indices = vec![3, 4, 5]; let total_num_groups = 6; - group_ordering.new_groups( + group_completion.new_groups( &batch_group_values, &group_indices, total_num_groups, )?; assert_eq!( - group_ordering.state, + group_completion.state, State::InProgress { - current_sort: 2, - sort_key: vec![ScalarValue::Int32(Some(3))], + current_run_start: 2, + group_key: vec![ScalarValue::Int32(Some(third))], current: 5 } ); + assert_eq!(group_completion.emit_to(), Some(EmitTo::First(2))); // push with only a boundary to previous batch let batch_group_values: Vec = vec![ - Arc::new(Int32Array::from(vec![4, 4, 4])), - Arc::new(Int32Array::from(vec![1, 1, 1])), + Arc::new(Int32Array::from(vec![fourth, fourth, fourth])), + Arc::new(Int32Array::from(vec![1, 2, 3])), ]; let group_indices = vec![6, 7, 8]; let total_num_groups = 9; - group_ordering.new_groups( + group_completion.new_groups( &batch_group_values, &group_indices, total_num_groups, )?; assert_eq!( - group_ordering.state, + group_completion.state, State::InProgress { - current_sort: 6, - sort_key: vec![ScalarValue::Int32(Some(4))], + current_run_start: 6, + group_key: vec![ScalarValue::Int32(Some(fourth))], current: 8 } ); + assert_eq!(group_completion.emit_to(), Some(EmitTo::First(6))); Ok(()) } diff --git a/datafusion/physical-plan/src/aggregates/partial_reduce_stream.rs b/datafusion/physical-plan/src/aggregates/partial_reduce_stream.rs index d5057d53cf19d..5e930f1b72239 100644 --- a/datafusion/physical-plan/src/aggregates/partial_reduce_stream.rs +++ b/datafusion/physical-plan/src/aggregates/partial_reduce_stream.rs @@ -29,10 +29,11 @@ use datafusion_execution::memory_pool::{MemoryConsumer, MemoryReservation}; use futures::stream::{Stream, StreamExt}; use super::AggregateExec; +use super::GroupCompletionMode; use super::aggregate_hash_table::{AggregateHashTable, PartialReduceMarker}; use crate::metrics::{BaselineMetrics, Count, MetricBuilder, RecordOutput, SpillMetrics}; use crate::stream::EmptyRecordBatchStream; -use crate::{InputOrderMode, RecordBatchStream, SendableRecordBatchStream}; +use crate::{RecordBatchStream, SendableRecordBatchStream}; /// Hash aggregation can combine multiple partial stages before final /// evaluation. This stream implements the partial-reduce stage. @@ -182,7 +183,7 @@ impl PartialReduceHashAggregateStream { partition: usize, ) -> Result { debug_assert_eq!(agg.mode, super::AggregateMode::PartialReduce); - debug_assert_eq!(agg.input_order_mode, InputOrderMode::Linear); + debug_assert_eq!(agg.group_completion_mode, GroupCompletionMode::None); let schema = Arc::clone(&agg.schema); let input = agg.input.execute(partition, Arc::clone(context))?; diff --git a/datafusion/physical-plan/src/aggregates/single_stream.rs b/datafusion/physical-plan/src/aggregates/single_stream.rs index f27017784bfd3..c04d93f9d6b15 100644 --- a/datafusion/physical-plan/src/aggregates/single_stream.rs +++ b/datafusion/physical-plan/src/aggregates/single_stream.rs @@ -30,14 +30,15 @@ use datafusion_execution::{TaskContext, TryEmitter, async_try_stream}; use futures::stream::StreamExt; use super::aggregate_hash_table::{ - AggregateHashTable, OrderedAggregateTableMetrics, SingleMarker, + AggregateHashTable, ClusteredAggregateTableMetrics, SingleMarker, }; +use super::order::GroupCompletionMode; use super::spill::AggregateSpill; use super::{AggregateExec, create_schema}; +use crate::SendableRecordBatchStream; use crate::aggregates::AggregateMode; use crate::metrics::{BaselineMetrics, SpillMetrics}; use crate::stream::{ObservedStream, RecordBatchStreamAdapter}; -use crate::{InputOrderMode, SendableRecordBatchStream}; /// Hash aggregation can run the full logical aggregation in one operator. This /// stream implements the single stage for grouped hash aggregation. @@ -82,7 +83,7 @@ use crate::{InputOrderMode, SendableRecordBatchStream}; /// 3. Perform a sort-preserving merge of all spill files and feed the merged output /// into an ordered streaming aggregation, which ensures bounded memory usage and /// evaluates the final result. -/// - [`OrderedFinalAggregateStream`](super::ordered_final_stream::OrderedFinalAggregateStream) is reused for the streaming aggregation. +/// - [`ClusteredFinalAggregateStream`](super::clustered_final_stream::ClusteredFinalAggregateStream) is reused for the streaming aggregation. /// /// # Optimization: DISTINCT LIMIT Soft Limit /// @@ -162,7 +163,7 @@ impl SingleHashAggregateStream { agg.mode, AggregateMode::Single | AggregateMode::SinglePartitioned )); - debug_assert_eq!(agg.input_order_mode, InputOrderMode::Linear); + debug_assert_eq!(agg.group_completion_mode, GroupCompletionMode::None); let schema = Arc::clone(&agg.schema); let input = agg.input.execute(partition, Arc::clone(context))?; @@ -193,7 +194,7 @@ impl SingleHashAggregateStream { context, partition, batch_size, - &InputOrderMode::Linear, + &GroupCompletionMode::None, &state_schema, spill_metrics, )?)) @@ -398,7 +399,7 @@ impl Aggregating { // Construct the replay stream: an ordered final aggregate stream // over the sort-preserving merge of all spill runs. - let metrics = OrderedAggregateTableMetrics::from_hash_table(&hash_table); + let metrics = ClusteredAggregateTableMetrics::from_hash_table(&hash_table); drop(hash_table); reservation.try_resize(0)?; // The outer ObservedStream counts output; replay only shares compute time. diff --git a/datafusion/physical-plan/src/aggregates/spill.rs b/datafusion/physical-plan/src/aggregates/spill.rs index 887be41855996..1bce4fd8d23bd 100644 --- a/datafusion/physical-plan/src/aggregates/spill.rs +++ b/datafusion/physical-plan/src/aggregates/spill.rs @@ -28,14 +28,15 @@ use datafusion_physical_expr::PhysicalSortExpr; use datafusion_physical_expr::expressions::Column; use datafusion_physical_expr_common::sort_expr::LexOrdering; -use super::aggregate_hash_table::OrderedAggregateTableMetrics; -use super::ordered_final_stream::OrderedFinalAggregateStream; +use super::aggregate_hash_table::ClusteredAggregateTableMetrics; +use super::clustered_final_stream::ClusteredFinalAggregateStream; +use super::order::GroupCompletionMode; use super::{AggregateExec, AggregateMode}; +use crate::SendableRecordBatchStream; use crate::metrics::{BaselineMetrics, SpillMetrics}; use crate::sorts::IncrementalSortIterator; use crate::sorts::streaming_merge::{SortedSpillFile, StreamingMergeBuilder}; use crate::spill::spill_manager::SpillManager; -use crate::{InputOrderMode, SendableRecordBatchStream}; /// Spill configuration and accumulated runs of one grouped aggregation stream. /// @@ -43,7 +44,7 @@ use crate::{InputOrderMode, SendableRecordBatchStream}; /// drains all currently buffered groups as intermediate state (see /// `take_state_batch` on the aggregate tables), sorts them by the full group /// key, and writes them to one spill file. After the original input ends, all -/// files are merged and replayed through an [`OrderedFinalAggregateStream`], +/// files are merged and replayed through a [`ClusteredFinalAggregateStream`], /// which merges the states and evaluates the final aggregate values. pub(super) struct AggregateSpill { /// Aggregate configuration used to construct the replay stream. @@ -92,7 +93,7 @@ pub(super) struct AggregateSpill { /// using the two previously sorted spill files. /// 2. Build a final aggregation stream: /// - The input is the SPM stream. - /// - It reuses `OrderedFinalAggregateStream` for processing. + /// - It reuses `ClusteredFinalAggregateStream` for processing. /// - It returns the final aggregation result directly. /// /// SPM output Final aggregate output @@ -126,10 +127,11 @@ impl AggregateSpill { /// Creates the spill context of a stream, whose spill requests are described /// as `label`. /// - /// `input_order_mode` is the order of the stream's input: spill files are - /// sorted by the already ordered group columns first, followed by the - /// remaining ones, so that replay keeps the ordering the stream promised. - /// Fully sorted input aggregates in bounded memory and never spills. + /// `group_completion_mode` determines which group columns are already + /// contiguous. Spill files are sorted by those columns first, followed by + /// the remaining ones. Existing output sort options are retained so replay + /// preserves any advertised ordering as well as group completion. + /// Full group completion aggregates in bounded memory and never spills. /// /// `spill_schema` is the schema of the intermediate state batches. #[expect(clippy::too_many_arguments)] @@ -139,12 +141,12 @@ impl AggregateSpill { context: &Arc, partition: usize, batch_size: usize, - input_order_mode: &InputOrderMode, + group_completion_mode: &GroupCompletionMode, spill_schema: &SchemaRef, spill_metrics: SpillMetrics, ) -> Result { let mut replay_agg = agg.clone(); - replay_agg.input_order_mode = InputOrderMode::Sorted; + replay_agg.group_completion_mode = GroupCompletionMode::Full; let group_schema = match agg.mode { AggregateMode::Final | AggregateMode::FinalPartitioned => { agg.group_by().group_schema(spill_schema)? @@ -164,17 +166,16 @@ impl AggregateSpill { }; let num_group_columns = group_schema.fields().len(); - let ordered_indices: &[usize] = match input_order_mode { - InputOrderMode::Linear => &[], - InputOrderMode::PartiallySorted(ordered_indices) => ordered_indices, - InputOrderMode::Sorted => { - return internal_err!("{label}: fully ordered input does not spill"); + let contiguous_indices: &[usize] = match group_completion_mode { + GroupCompletionMode::None => &[], + GroupCompletionMode::Partial(contiguous_indices) => contiguous_indices, + GroupCompletionMode::Full => { + return internal_err!("{label}: fully contiguous groups do not spill"); } }; - let spill_indices = ordered_indices - .iter() - .copied() - .chain((0..num_group_columns).filter(|idx| !ordered_indices.contains(idx))); + let spill_indices = contiguous_indices.iter().copied().chain( + (0..num_group_columns).filter(|idx| !contiguous_indices.contains(idx)), + ); let output_ordering = agg.cache.output_ordering(); let spill_sort_exprs = spill_indices.map(|idx| { let output_expr = Column::new(group_schema.field(idx).name(), idx); @@ -249,11 +250,11 @@ impl AggregateSpill { } /// Merges every sorted run, and does the aggregate evaluation with - /// [`OrderedFinalAggregateStream`]. + /// [`ClusteredFinalAggregateStream`]. pub(super) fn into_replay_stream( self, baseline_metrics: &BaselineMetrics, - metrics: OrderedAggregateTableMetrics, + metrics: ClusteredAggregateTableMetrics, reservation: MemoryReservation, ) -> Result { let Self { @@ -284,12 +285,12 @@ impl AggregateSpill { .with_replay_headroom() .with_intermediate_merge_sizing(Some(min_spill_batch_rows)) .build()?; - let replay = OrderedFinalAggregateStream::new_with_input_and_metrics( + let replay = ClusteredFinalAggregateStream::new_with_input_and_metrics( &replay_agg, &context, partition, merged, - &InputOrderMode::Sorted, + &GroupCompletionMode::Full, baseline_metrics.clone(), metrics, None, diff --git a/datafusion/physical-plan/src/recursive_query.rs b/datafusion/physical-plan/src/recursive_query.rs index 0a56488de84dd..270d03ccafca2 100644 --- a/datafusion/physical-plan/src/recursive_query.rs +++ b/datafusion/physical-plan/src/recursive_query.rs @@ -23,7 +23,7 @@ use std::task::{Context, Poll}; use super::work_table::{ReservedBatches, WorkTable}; use crate::aggregates::group_values::{GroupValues, new_group_values}; -use crate::aggregates::order::GroupOrdering; +use crate::aggregates::order::GroupCompletion; use crate::common::project_plan_to_schema; use crate::execution_plan::{Boundedness, EmissionType, reset_plan_states}; use crate::metrics::{ @@ -464,7 +464,7 @@ struct DistinctDeduplicator { impl DistinctDeduplicator { fn new(schema: SchemaRef, task_context: &TaskContext) -> Result { - let group_values = new_group_values(schema, &GroupOrdering::None)?; + let group_values = new_group_values(schema, &GroupCompletion::None)?; let reservation = MemoryConsumer::new("RecursiveQueryHashTable") .register(task_context.memory_pool()); Ok(Self { diff --git a/datafusion/sqllogictest/test_files/agg_func_substitute.slt b/datafusion/sqllogictest/test_files/agg_func_substitute.slt index be9749bc9543e..8402644f8ed90 100644 --- a/datafusion/sqllogictest/test_files/agg_func_substitute.slt +++ b/datafusion/sqllogictest/test_files/agg_func_substitute.slt @@ -44,9 +44,9 @@ logical_plan 03)----TableScan: multiple_ordered_table projection=[a, c] physical_plan 01)ProjectionExec: expr=[a@0 as a, nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]@1 as result] -02)--AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], ordering_mode=Sorted +02)--AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], group_completion_mode=Full 03)----RepartitionExec: partitioning=Hash([a@0], 4), input_partitions=4, preserve_order=true, sort_exprs=a@0 ASC NULLS LAST -04)------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], ordering_mode=Sorted +04)------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], group_completion_mode=Full 05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, maintains_sort_order=true 06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, c], output_orderings=[[a@0 ASC NULLS LAST], [c@1 ASC NULLS LAST]], file_type=csv, has_header=true @@ -62,9 +62,9 @@ logical_plan 03)----TableScan: multiple_ordered_table projection=[a, c] physical_plan 01)ProjectionExec: expr=[a@0 as a, nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]@1 as result] -02)--AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], ordering_mode=Sorted +02)--AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], group_completion_mode=Full 03)----RepartitionExec: partitioning=Hash([a@0], 4), input_partitions=4, preserve_order=true, sort_exprs=a@0 ASC NULLS LAST -04)------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], ordering_mode=Sorted +04)------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], group_completion_mode=Full 05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, maintains_sort_order=true 06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, c], output_orderings=[[a@0 ASC NULLS LAST], [c@1 ASC NULLS LAST]], file_type=csv, has_header=true @@ -79,9 +79,9 @@ logical_plan 03)----TableScan: multiple_ordered_table projection=[a, c] physical_plan 01)ProjectionExec: expr=[a@0 as a, nth_value(multiple_ordered_table.c,Int64(1) + Int64(100)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]@1 as result] -02)--AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c, 101) ORDER BY [multiple_ordered_table.c ASC NULLS LAST] as nth_value(multiple_ordered_table.c,Int64(1) + Int64(100)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], ordering_mode=Sorted +02)--AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c, 101) ORDER BY [multiple_ordered_table.c ASC NULLS LAST] as nth_value(multiple_ordered_table.c,Int64(1) + Int64(100)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], group_completion_mode=Full 03)----RepartitionExec: partitioning=Hash([a@0], 4), input_partitions=4, preserve_order=true, sort_exprs=a@0 ASC NULLS LAST -04)------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c, 101) ORDER BY [multiple_ordered_table.c ASC NULLS LAST] as nth_value(multiple_ordered_table.c,Int64(1) + Int64(100)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], ordering_mode=Sorted +04)------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c, 101) ORDER BY [multiple_ordered_table.c ASC NULLS LAST] as nth_value(multiple_ordered_table.c,Int64(1) + Int64(100)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], group_completion_mode=Full 05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, maintains_sort_order=true 06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, c], output_orderings=[[a@0 ASC NULLS LAST], [c@1 ASC NULLS LAST]], file_type=csv, has_header=true diff --git a/datafusion/sqllogictest/test_files/aggregate.slt b/datafusion/sqllogictest/test_files/aggregate.slt index d72372dee47d3..f0a2385ab4f57 100644 --- a/datafusion/sqllogictest/test_files/aggregate.slt +++ b/datafusion/sqllogictest/test_files/aggregate.slt @@ -9540,7 +9540,7 @@ CREATE TABLE stream_test ( (3, 1.0, 1.0, 7, false, 'e'), (3, 2.0, 2.0, 8, false, 'f'); # Test comprehensive aggregates with streaming -# This verifies that CORR and other aggregates work together in a streaming plan (ordering_mode=Sorted) +# This verifies that CORR and other aggregates work together in a streaming plan (group_completion_mode=Full) # Basic Aggregates query TT @@ -9579,7 +9579,7 @@ logical_plan 05)--------TableScan: stream_test projection=[g, x, y, i, b] physical_plan 01)ProjectionExec: expr=[g@0 as g, count(Int64(1))@1 as count(*), sum(stream_test.x)@2 as sum(stream_test.x), avg(stream_test.x)@3 as avg(stream_test.x), avg(stream_test.x)@3 as mean(stream_test.x), min(stream_test.x)@4 as min(stream_test.x), max(stream_test.y)@5 as max(stream_test.y), bit_and(stream_test.i)@6 as bit_and(stream_test.i), bit_or(stream_test.i)@7 as bit_or(stream_test.i), bit_xor(stream_test.i)@8 as bit_xor(stream_test.i), bool_and(stream_test.b)@9 as bool_and(stream_test.b), bool_or(stream_test.b)@10 as bool_or(stream_test.b), median(stream_test.x)@11 as median(stream_test.x), 0 as grouping(stream_test.g), var(stream_test.x)@12 as var(stream_test.x), var(stream_test.x)@12 as var_samp(stream_test.x), var_pop(stream_test.x)@13 as var_pop(stream_test.x), var(stream_test.x)@12 as var_sample(stream_test.x), var_pop(stream_test.x)@13 as var_population(stream_test.x), stddev(stream_test.x)@14 as stddev(stream_test.x), stddev(stream_test.x)@14 as stddev_samp(stream_test.x), stddev_pop(stream_test.x)@15 as stddev_pop(stream_test.x)] -02)--AggregateExec: mode=Single, gby=[g@0 as g], aggr=[count(Int64(1)), sum(stream_test.x), avg(stream_test.x), min(stream_test.x), max(stream_test.y), bit_and(stream_test.i), bit_or(stream_test.i), bit_xor(stream_test.i), bool_and(stream_test.b), bool_or(stream_test.b), median(stream_test.x), var(stream_test.x), var_pop(stream_test.x), stddev(stream_test.x), stddev_pop(stream_test.x)], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[g@0 as g], aggr=[count(Int64(1)), sum(stream_test.x), avg(stream_test.x), min(stream_test.x), max(stream_test.y), bit_and(stream_test.i), bit_or(stream_test.i), bit_xor(stream_test.i), bool_and(stream_test.b), bool_or(stream_test.b), median(stream_test.x), var(stream_test.x), var_pop(stream_test.x), stddev(stream_test.x), stddev_pop(stream_test.x)], group_completion_mode=Full 03)----SortExec: TopK(fetch=10000), expr=[g@0 ASC NULLS LAST], preserve_partitioning=[false] 04)------DataSourceExec: partitions=1, partition_sizes=[1] @@ -9634,7 +9634,7 @@ logical_plan 03)----Sort: stream_test.g ASC NULLS LAST, fetch=10000 04)------TableScan: stream_test projection=[g, x] physical_plan -01)AggregateExec: mode=Single, gby=[g@0 as g], aggr=[array_agg(stream_test.x) ORDER BY [stream_test.x ASC NULLS LAST], array_agg(DISTINCT stream_test.x) ORDER BY [stream_test.x ASC NULLS LAST], first_value(stream_test.x) ORDER BY [stream_test.x ASC NULLS LAST], last_value(stream_test.x) ORDER BY [stream_test.x ASC NULLS LAST], nth_value(stream_test.x,Int64(1)) ORDER BY [stream_test.x ASC NULLS LAST]], ordering_mode=Sorted +01)AggregateExec: mode=Single, gby=[g@0 as g], aggr=[array_agg(stream_test.x) ORDER BY [stream_test.x ASC NULLS LAST], array_agg(DISTINCT stream_test.x) ORDER BY [stream_test.x ASC NULLS LAST], first_value(stream_test.x) ORDER BY [stream_test.x ASC NULLS LAST], last_value(stream_test.x) ORDER BY [stream_test.x ASC NULLS LAST], nth_value(stream_test.x,Int64(1)) ORDER BY [stream_test.x ASC NULLS LAST]], group_completion_mode=Full 02)--SortExec: TopK(fetch=10000), expr=[g@0 ASC NULLS LAST, x@1 ASC NULLS LAST], preserve_partitioning=[false] 03)----DataSourceExec: partitions=1, partition_sizes=[1] @@ -9671,7 +9671,7 @@ logical_plan 03)----Sort: stream_test.g ASC NULLS LAST, fetch=10000 04)------TableScan: stream_test projection=[g, s] physical_plan -01)AggregateExec: mode=Single, gby=[g@0 as g], aggr=[array_agg(stream_test.s) ORDER BY [stream_test.s ASC NULLS LAST], string_agg(stream_test.s,Utf8("|")) ORDER BY [stream_test.s ASC NULLS LAST], string_agg(DISTINCT stream_test.s,Utf8("|")) ORDER BY [stream_test.s ASC NULLS LAST]], ordering_mode=Sorted +01)AggregateExec: mode=Single, gby=[g@0 as g], aggr=[array_agg(stream_test.s) ORDER BY [stream_test.s ASC NULLS LAST], string_agg(stream_test.s,Utf8("|")) ORDER BY [stream_test.s ASC NULLS LAST], string_agg(DISTINCT stream_test.s,Utf8("|")) ORDER BY [stream_test.s ASC NULLS LAST]], group_completion_mode=Full 02)--SortExec: TopK(fetch=10000), expr=[g@0 ASC NULLS LAST, s@1 ASC NULLS LAST], preserve_partitioning=[false] 03)----DataSourceExec: partitions=1, partition_sizes=[1] @@ -9718,7 +9718,7 @@ logical_plan 05)--------TableScan: stream_test projection=[g, x, y] physical_plan 01)ProjectionExec: expr=[g@0 as g, corr(stream_test.x,stream_test.y)@1 as corr(stream_test.x,stream_test.y), covar_samp(stream_test.x,stream_test.y)@2 as covar(stream_test.x,stream_test.y), covar_samp(stream_test.x,stream_test.y)@2 as covar_samp(stream_test.x,stream_test.y), covar_pop(stream_test.x,stream_test.y)@3 as covar_pop(stream_test.x,stream_test.y), regr_sxx(stream_test.x,stream_test.y)@4 as regr_sxx(stream_test.x,stream_test.y), regr_sxy(stream_test.x,stream_test.y)@5 as regr_sxy(stream_test.x,stream_test.y), regr_syy(stream_test.x,stream_test.y)@6 as regr_syy(stream_test.x,stream_test.y), regr_avgx(stream_test.x,stream_test.y)@7 as regr_avgx(stream_test.x,stream_test.y), regr_avgy(stream_test.x,stream_test.y)@8 as regr_avgy(stream_test.x,stream_test.y), regr_count(stream_test.x,stream_test.y)@9 as regr_count(stream_test.x,stream_test.y), regr_slope(stream_test.x,stream_test.y)@10 as regr_slope(stream_test.x,stream_test.y), regr_intercept(stream_test.x,stream_test.y)@11 as regr_intercept(stream_test.x,stream_test.y), regr_r2(stream_test.x,stream_test.y)@12 as regr_r2(stream_test.x,stream_test.y)] -02)--AggregateExec: mode=Single, gby=[g@0 as g], aggr=[corr(stream_test.x,stream_test.y), covar_samp(stream_test.x,stream_test.y), covar_pop(stream_test.x,stream_test.y), regr_sxx(stream_test.x,stream_test.y), regr_sxy(stream_test.x,stream_test.y), regr_syy(stream_test.x,stream_test.y), regr_avgx(stream_test.x,stream_test.y), regr_avgy(stream_test.x,stream_test.y), regr_count(stream_test.x,stream_test.y), regr_slope(stream_test.x,stream_test.y), regr_intercept(stream_test.x,stream_test.y), regr_r2(stream_test.x,stream_test.y)], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[g@0 as g], aggr=[corr(stream_test.x,stream_test.y), covar_samp(stream_test.x,stream_test.y), covar_pop(stream_test.x,stream_test.y), regr_sxx(stream_test.x,stream_test.y), regr_sxy(stream_test.x,stream_test.y), regr_syy(stream_test.x,stream_test.y), regr_avgx(stream_test.x,stream_test.y), regr_avgy(stream_test.x,stream_test.y), regr_count(stream_test.x,stream_test.y), regr_slope(stream_test.x,stream_test.y), regr_intercept(stream_test.x,stream_test.y), regr_r2(stream_test.x,stream_test.y)], group_completion_mode=Full 03)----SortExec: TopK(fetch=10000), expr=[g@0 ASC NULLS LAST], preserve_partitioning=[false] 04)------DataSourceExec: partitions=1, partition_sizes=[1] @@ -9771,7 +9771,7 @@ logical_plan 05)--------TableScan: stream_test projection=[g, x, i] physical_plan 01)ProjectionExec: expr=[g@0 as g, approx_distinct(stream_test.i)@1 as approx_distinct(stream_test.i), approx_median(stream_test.x)@2 as approx_median(stream_test.x), percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST]@3 as percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST]@3 as quantile_cont(stream_test.x,Float64(0.5)), approx_percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST]@4 as approx_percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], approx_percentile_cont_with_weight(Float64(1),Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST]@5 as approx_percentile_cont_with_weight(Float64(1),Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], percentile_cont(stream_test.x,Float64(0.5))@6 as percentile_cont(stream_test.x,Float64(0.5)), approx_percentile_cont(stream_test.x,Float64(0.5))@7 as approx_percentile_cont(stream_test.x,Float64(0.5)), approx_percentile_cont_with_weight(stream_test.x,Float64(1),Float64(0.5))@8 as approx_percentile_cont_with_weight(stream_test.x,Float64(1),Float64(0.5))] -02)--AggregateExec: mode=Single, gby=[g@0 as g], aggr=[approx_distinct(stream_test.i), approx_median(stream_test.x), percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], approx_percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], approx_percentile_cont_with_weight(Float64(1),Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], percentile_cont(stream_test.x,Float64(0.5)), approx_percentile_cont(stream_test.x,Float64(0.5)), approx_percentile_cont_with_weight(stream_test.x,Float64(1),Float64(0.5))], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[g@0 as g], aggr=[approx_distinct(stream_test.i), approx_median(stream_test.x), percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], approx_percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], approx_percentile_cont_with_weight(Float64(1),Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], percentile_cont(stream_test.x,Float64(0.5)), approx_percentile_cont(stream_test.x,Float64(0.5)), approx_percentile_cont_with_weight(stream_test.x,Float64(1),Float64(0.5))], group_completion_mode=Full 03)----SortExec: TopK(fetch=10000), expr=[g@0 ASC NULLS LAST], preserve_partitioning=[false] 04)------DataSourceExec: partitions=1, partition_sizes=[1] diff --git a/datafusion/sqllogictest/test_files/group_by.slt b/datafusion/sqllogictest/test_files/group_by.slt index 9eb1865ecf107..ec1970a1101b7 100644 --- a/datafusion/sqllogictest/test_files/group_by.slt +++ b/datafusion/sqllogictest/test_files/group_by.slt @@ -2112,7 +2112,7 @@ logical_plan 03)----TableScan: annotated_data_infinite2 projection=[a, b, c] physical_plan 01)ProjectionExec: expr=[a@1 as a, b@0 as b, sum(annotated_data_infinite2.c)@2 as summation1] -02)--AggregateExec: mode=Single, gby=[b@1 as b, a@0 as a], aggr=[sum(annotated_data_infinite2.c)], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[b@1 as b, a@0 as a], aggr=[sum(annotated_data_infinite2.c)], group_completion_mode=Full 03)----StreamingTableExec: partition_sizes=1, projection=[a, b, c], infinite_source=true, output_ordering=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, c@2 ASC NULLS LAST] @@ -2143,7 +2143,7 @@ logical_plan 03)----TableScan: annotated_data_infinite2 projection=[a, c, d] physical_plan 01)ProjectionExec: expr=[a@1 as a, d@0 as d, sum(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]@2 as summation1] -02)--AggregateExec: mode=Single, gby=[d@2 as d, a@0 as a], aggr=[sum(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]], ordering_mode=PartiallySorted([1]) +02)--AggregateExec: mode=Single, gby=[d@2 as d, a@0 as a], aggr=[sum(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]], group_completion_mode=Partial([1]) 03)----StreamingTableExec: partition_sizes=1, projection=[a, c, d], infinite_source=true, output_ordering=[a@0 ASC NULLS LAST] query III @@ -2176,7 +2176,7 @@ logical_plan 03)----TableScan: annotated_data_infinite2 projection=[a, b, c] physical_plan 01)ProjectionExec: expr=[a@0 as a, b@1 as b, first_value(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]@2 as first_c] -02)--AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[first_value(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[first_value(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]], group_completion_mode=Full 03)----StreamingTableExec: partition_sizes=1, projection=[a, b, c], infinite_source=true, output_ordering=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, c@2 ASC NULLS LAST] query III @@ -2202,7 +2202,7 @@ logical_plan 03)----TableScan: annotated_data_infinite2 projection=[a, b, c] physical_plan 01)ProjectionExec: expr=[a@0 as a, b@1 as b, last_value(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]@2 as last_c] -02)--AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[last_value(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[last_value(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]], group_completion_mode=Full 03)----StreamingTableExec: partition_sizes=1, projection=[a, b, c], infinite_source=true, output_ordering=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, c@2 ASC NULLS LAST] query III @@ -2229,7 +2229,7 @@ logical_plan 03)----TableScan: annotated_data_infinite2 projection=[a, b, c] physical_plan 01)ProjectionExec: expr=[a@0 as a, b@1 as b, last_value(annotated_data_infinite2.c)@2 as last_c] -02)--AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[last_value(annotated_data_infinite2.c)], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[last_value(annotated_data_infinite2.c)], group_completion_mode=Full 03)----StreamingTableExec: partition_sizes=1, projection=[a, b, c], infinite_source=true, output_ordering=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, c@2 ASC NULLS LAST] query III @@ -2288,7 +2288,7 @@ logical_plan 01)Aggregate: groupBy=[[annotated_data_infinite2.a, annotated_data_infinite2.b]], aggr=[[array_agg(annotated_data_infinite2.d) ORDER BY [annotated_data_infinite2.d ASC NULLS LAST]]] 02)--TableScan: annotated_data_infinite2 projection=[a, b, d] physical_plan -01)AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[array_agg(annotated_data_infinite2.d) ORDER BY [annotated_data_infinite2.d ASC NULLS LAST]], ordering_mode=Sorted +01)AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[array_agg(annotated_data_infinite2.d) ORDER BY [annotated_data_infinite2.d ASC NULLS LAST]], group_completion_mode=Full 02)--PartialSortExec: expr=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, d@2 ASC NULLS LAST], common_prefix_length=[2] 03)----StreamingTableExec: partition_sizes=1, projection=[a, b, d], infinite_source=true, output_ordering=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST] @@ -2612,7 +2612,7 @@ logical_plan 05)--------TableScan: sales_global projection=[country, amount] physical_plan 01)ProjectionExec: expr=[country@0 as country, array_agg(s.amount) ORDER BY [s.amount DESC NULLS FIRST]@1 as amounts, sum(s.amount)@2 as sum1] -02)--AggregateExec: mode=Single, gby=[country@0 as country], aggr=[array_agg(s.amount) ORDER BY [s.amount DESC NULLS FIRST], sum(s.amount)], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[country@0 as country], aggr=[array_agg(s.amount) ORDER BY [s.amount DESC NULLS FIRST], sum(s.amount)], group_completion_mode=Full 03)----SortExec: TopK(fetch=10), expr=[country@0 ASC NULLS LAST, amount@1 DESC], preserve_partitioning=[false] 04)------DataSourceExec: partitions=1, partition_sizes=[1] @@ -2650,7 +2650,7 @@ logical_plan 05)--------TableScan: sales_global projection=[zip_code, country, amount] physical_plan 01)ProjectionExec: expr=[country@0 as country, zip_code@1 as zip_code, array_agg(s.amount) ORDER BY [s.amount DESC NULLS FIRST]@2 as amounts, sum(s.amount)@3 as sum1] -02)--AggregateExec: mode=Single, gby=[country@1 as country, zip_code@0 as zip_code], aggr=[array_agg(s.amount) ORDER BY [s.amount DESC NULLS FIRST], sum(s.amount)], ordering_mode=PartiallySorted([0]) +02)--AggregateExec: mode=Single, gby=[country@1 as country, zip_code@0 as zip_code], aggr=[array_agg(s.amount) ORDER BY [s.amount DESC NULLS FIRST], sum(s.amount)], group_completion_mode=Partial([0]) 03)----SortExec: TopK(fetch=10), expr=[country@1 ASC NULLS LAST, amount@2 DESC], preserve_partitioning=[false] 04)------DataSourceExec: partitions=1, partition_sizes=[1] @@ -2687,7 +2687,7 @@ logical_plan 05)--------TableScan: sales_global projection=[country, amount] physical_plan 01)ProjectionExec: expr=[country@0 as country, array_agg(s.amount) ORDER BY [s.country DESC NULLS FIRST]@1 as amounts, sum(s.amount)@2 as sum1] -02)--AggregateExec: mode=Single, gby=[country@0 as country], aggr=[array_agg(s.amount) ORDER BY [s.country DESC NULLS FIRST], sum(s.amount)], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[country@0 as country], aggr=[array_agg(s.amount) ORDER BY [s.country DESC NULLS FIRST], sum(s.amount)], group_completion_mode=Full 03)----SortExec: TopK(fetch=10), expr=[country@0 ASC NULLS LAST], preserve_partitioning=[false] 04)------DataSourceExec: partitions=1, partition_sizes=[1] @@ -2723,7 +2723,7 @@ logical_plan 05)--------TableScan: sales_global projection=[country, amount] physical_plan 01)ProjectionExec: expr=[country@0 as country, array_agg(s.amount) ORDER BY [s.country DESC NULLS FIRST, s.amount DESC NULLS FIRST]@1 as amounts, sum(s.amount)@2 as sum1] -02)--AggregateExec: mode=Single, gby=[country@0 as country], aggr=[array_agg(s.amount) ORDER BY [s.country DESC NULLS FIRST, s.amount DESC NULLS FIRST], sum(s.amount)], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[country@0 as country], aggr=[array_agg(s.amount) ORDER BY [s.country DESC NULLS FIRST, s.amount DESC NULLS FIRST], sum(s.amount)], group_completion_mode=Full 03)----SortExec: TopK(fetch=10), expr=[country@0 ASC NULLS LAST, amount@1 DESC], preserve_partitioning=[false] 04)------DataSourceExec: partitions=1, partition_sizes=[1] @@ -4027,7 +4027,7 @@ logical_plan 12)------------------TableScan: multiple_ordered_table projection=[a, d] physical_plan 01)ProjectionExec: expr=[last_value(l.d) ORDER BY [l.a ASC NULLS LAST]@1 as amount_usd] -02)--AggregateExec: mode=Single, gby=[row_n@2 as row_n], aggr=[last_value(l.d) ORDER BY [l.a ASC NULLS LAST]], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[row_n@2 as row_n], aggr=[last_value(l.d) ORDER BY [l.a ASC NULLS LAST]], group_completion_mode=Full 03)----HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(d@1, d@1)], filter=CAST(a@0 AS Int64) >= CAST(a@1 AS Int64) - 10, projection=[a@0, d@1, row_n@4] 04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, d], output_ordering=[a@0 ASC NULLS LAST], file_type=csv, has_header=true 05)------ProjectionExec: expr=[a@0 as a, d@1 as d, row_number() ORDER BY [r.a ASC NULLS LAST] RANGE BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW@2 as row_n] @@ -4067,9 +4067,9 @@ logical_plan 01)Aggregate: groupBy=[[multiple_ordered_table_with_pk.c, multiple_ordered_table_with_pk.b]], aggr=[[sum(CAST(multiple_ordered_table_with_pk.d AS Int64))]] 02)--TableScan: multiple_ordered_table_with_pk projection=[b, c, d] physical_plan -01)AggregateExec: mode=FinalPartitioned, gby=[c@0 as c, b@1 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=PartiallySorted([0]) +01)AggregateExec: mode=FinalPartitioned, gby=[c@0 as c, b@1 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Partial([0]) 02)--RepartitionExec: partitioning=Hash([c@0, b@1], 8), input_partitions=8, preserve_order=true, sort_exprs=c@0 ASC NULLS LAST -03)----AggregateExec: mode=Partial, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=PartiallySorted([0]) +03)----AggregateExec: mode=Partial, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Partial([0]) 04)------RepartitionExec: partitioning=RoundRobinBatch(8), input_partitions=1, maintains_sort_order=true 05)--------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[b, c, d], output_ordering=[c@1 ASC NULLS LAST], constraints=[PrimaryKey([3])], file_type=csv, has_header=true @@ -4106,9 +4106,9 @@ logical_plan 01)Aggregate: groupBy=[[multiple_ordered_table_with_pk.c, multiple_ordered_table_with_pk.b]], aggr=[[sum(CAST(multiple_ordered_table_with_pk.d AS Int64))]] 02)--TableScan: multiple_ordered_table_with_pk projection=[b, c, d] physical_plan -01)AggregateExec: mode=FinalPartitioned, gby=[c@0 as c, b@1 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=PartiallySorted([0]) +01)AggregateExec: mode=FinalPartitioned, gby=[c@0 as c, b@1 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Partial([0]) 02)--RepartitionExec: partitioning=Hash([c@0, b@1], 8), input_partitions=8, preserve_order=true, sort_exprs=c@0 ASC NULLS LAST -03)----AggregateExec: mode=Partial, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=PartiallySorted([0]) +03)----AggregateExec: mode=Partial, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Partial([0]) 04)------RepartitionExec: partitioning=RoundRobinBatch(8), input_partitions=1, maintains_sort_order=true 05)--------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[b, c, d], output_ordering=[c@1 ASC NULLS LAST], constraints=[PrimaryKey([3])], file_type=csv, has_header=true @@ -4129,9 +4129,9 @@ logical_plan 03)----Aggregate: groupBy=[[multiple_ordered_table_with_pk.c]], aggr=[[sum(CAST(multiple_ordered_table_with_pk.d AS Int64))]] 04)------TableScan: multiple_ordered_table_with_pk projection=[c, d] physical_plan -01)AggregateExec: mode=Single, gby=[c@0 as c, sum1@1 as sum1], aggr=[], ordering_mode=PartiallySorted([0]) +01)AggregateExec: mode=Single, gby=[c@0 as c, sum1@1 as sum1], aggr=[], group_completion_mode=Partial([0]) 02)--ProjectionExec: expr=[c@0 as c, sum(multiple_ordered_table_with_pk.d)@1 as sum1] -03)----AggregateExec: mode=Single, gby=[c@0 as c], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=Sorted +03)----AggregateExec: mode=Single, gby=[c@0 as c], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Full 04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[c, d], output_ordering=[c@0 ASC NULLS LAST], constraints=[PrimaryKey([3])], file_type=csv, has_header=true query TT @@ -4151,7 +4151,7 @@ physical_plan 01)ProjectionExec: expr=[c@0 as c, sum1@2 as sum1, sum(multiple_ordered_table_with_pk.b) ROWS BETWEEN UNBOUNDED PRECEDING AND UNBOUNDED FOLLOWING@3 as sumb] 02)--WindowAggExec: wdw=[sum(multiple_ordered_table_with_pk.b) ROWS BETWEEN UNBOUNDED PRECEDING AND UNBOUNDED FOLLOWING: Ok(Field { name: "sum(multiple_ordered_table_with_pk.b) ROWS BETWEEN UNBOUNDED PRECEDING AND UNBOUNDED FOLLOWING", data_type: Int64, nullable: true }), frame: WindowFrame { units: Rows, start_bound: Preceding(UInt64(NULL)), end_bound: Following(UInt64(NULL)), is_causal: false }] 03)----ProjectionExec: expr=[c@0 as c, b@1 as b, sum(multiple_ordered_table_with_pk.d)@2 as sum1] -04)------AggregateExec: mode=Single, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=PartiallySorted([0]) +04)------AggregateExec: mode=Single, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Partial([0]) 05)--------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[b, c, d], output_ordering=[c@1 ASC NULLS LAST], constraints=[PrimaryKey([3])], file_type=csv, has_header=true query TT @@ -4180,10 +4180,10 @@ logical_plan physical_plan 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(b@1, b@1)], projection=[c@0, c@3, sum1@2, sum1@5] 02)--ProjectionExec: expr=[c@0 as c, b@1 as b, sum(multiple_ordered_table_with_pk.d)@2 as sum1] -03)----AggregateExec: mode=Single, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=PartiallySorted([0]) +03)----AggregateExec: mode=Single, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Partial([0]) 04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[b, c, d], output_ordering=[c@1 ASC NULLS LAST], constraints=[PrimaryKey([3])], file_type=csv, has_header=true 05)--ProjectionExec: expr=[c@0 as c, b@1 as b, sum(multiple_ordered_table_with_pk.d)@2 as sum1] -06)----AggregateExec: mode=Single, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=PartiallySorted([0]) +06)----AggregateExec: mode=Single, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Partial([0]) 07)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[b, c, d], output_ordering=[c@1 ASC NULLS LAST], constraints=[PrimaryKey([3])], file_type=csv, has_header=true query TT @@ -4212,10 +4212,10 @@ physical_plan 01)ProjectionExec: expr=[c@0 as c, c@2 as c, sum1@1 as sum1, sum1@3 as sum1] 02)--CrossJoinExec 03)----ProjectionExec: expr=[c@0 as c, sum(multiple_ordered_table_with_pk.d)@1 as sum1] -04)------AggregateExec: mode=Single, gby=[c@0 as c], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=Sorted +04)------AggregateExec: mode=Single, gby=[c@0 as c], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Full 05)--------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[c, d], output_ordering=[c@0 ASC NULLS LAST], constraints=[PrimaryKey([3])], file_type=csv, has_header=true 06)----ProjectionExec: expr=[c@0 as c, sum(multiple_ordered_table_with_pk.d)@1 as sum1] -07)------AggregateExec: mode=Single, gby=[c@0 as c], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=Sorted +07)------AggregateExec: mode=Single, gby=[c@0 as c], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Full 08)--------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[c, d], output_ordering=[c@0 ASC NULLS LAST], constraints=[PrimaryKey([3])], file_type=csv, has_header=true # we do not generate physical plan for Repartition yet (e.g Distribute By queries). @@ -4254,10 +4254,10 @@ logical_plan physical_plan 01)UnionExec 02)--ProjectionExec: expr=[c@0 as c, a@1 as a, sum(multiple_ordered_table_with_pk.d)@2 as sum1] -03)----AggregateExec: mode=Single, gby=[c@1 as c, a@0 as a], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=Sorted +03)----AggregateExec: mode=Single, gby=[c@1 as c, a@0 as a], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Full 04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, c, d], output_orderings=[[a@0 ASC NULLS LAST], [c@1 ASC NULLS LAST]], constraints=[PrimaryKey([3])], file_type=csv, has_header=true 05)--ProjectionExec: expr=[c@0 as c, a@1 as a, sum(multiple_ordered_table_with_pk.d)@2 as sum1] -06)----AggregateExec: mode=Single, gby=[c@1 as c, a@0 as a], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=Sorted +06)----AggregateExec: mode=Single, gby=[c@1 as c, a@0 as a], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Full 07)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, c, d], output_orderings=[[a@0 ASC NULLS LAST], [c@1 ASC NULLS LAST]], constraints=[PrimaryKey([3])], file_type=csv, has_header=true # table scan should be simplified. @@ -4272,7 +4272,7 @@ logical_plan 03)----TableScan: multiple_ordered_table_with_pk projection=[a, c, d] physical_plan 01)ProjectionExec: expr=[c@0 as c, a@1 as a, sum(multiple_ordered_table_with_pk.d)@2 as sum1] -02)--AggregateExec: mode=Single, gby=[c@1 as c, a@0 as a], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[c@1 as c, a@0 as a], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Full 03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, c, d], output_orderings=[[a@0 ASC NULLS LAST], [c@1 ASC NULLS LAST]], constraints=[PrimaryKey([3])], file_type=csv, has_header=true # limit should be simplified @@ -4291,7 +4291,7 @@ logical_plan physical_plan 01)ProjectionExec: expr=[c@0 as c, a@1 as a, sum(multiple_ordered_table_with_pk.d)@2 as sum1] 02)--GlobalLimitExec: skip=0, fetch=5 -03)----AggregateExec: mode=Single, gby=[c@1 as c, a@0 as a], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=Sorted +03)----AggregateExec: mode=Single, gby=[c@1 as c, a@0 as a], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Full 04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, c, d], output_orderings=[[a@0 ASC NULLS LAST], [c@1 ASC NULLS LAST]], constraints=[PrimaryKey([3])], file_type=csv, has_header=true statement ok @@ -4374,9 +4374,9 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [time_chunks@0 DESC], fetch=5 02)--ProjectionExec: expr=[date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)@0 as time_chunks] -03)----AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)@0 as date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)], aggr=[], ordering_mode=Sorted +03)----AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)@0 as date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)], aggr=[], group_completion_mode=Full 04)------RepartitionExec: partitioning=Hash([date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)@0], 8), input_partitions=8, preserve_order=true, sort_exprs=date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)@0 DESC -05)--------AggregateExec: mode=Partial, gby=[date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 900000000000 }, ts@0) as date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)], aggr=[], ordering_mode=Sorted +05)--------AggregateExec: mode=Partial, gby=[date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 900000000000 }, ts@0) as date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)], aggr=[], group_completion_mode=Full 06)----------RepartitionExec: partitioning=RoundRobinBatch(8), input_partitions=1, maintains_sort_order=true 07)------------StreamingTableExec: partition_sizes=1, projection=[ts], infinite_source=true, output_ordering=[ts@0 DESC] @@ -5099,7 +5099,7 @@ logical_plan 02)--Aggregate: groupBy=[[multiple_ordered_table.a, multiple_ordered_table.b]], aggr=[[array_agg(multiple_ordered_table.c) ORDER BY [multiple_ordered_table.c DESC NULLS FIRST]]] 03)----TableScan: multiple_ordered_table projection=[a, b, c] physical_plan -01)AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[array_agg(multiple_ordered_table.c) ORDER BY [multiple_ordered_table.c DESC NULLS FIRST]], ordering_mode=Sorted +01)AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[array_agg(multiple_ordered_table.c) ORDER BY [multiple_ordered_table.c DESC NULLS FIRST]], group_completion_mode=Full 02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, b, c], output_orderings=[[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST], [c@2 ASC NULLS LAST]], file_type=csv, has_header=true query II? diff --git a/datafusion/sqllogictest/test_files/joins.slt b/datafusion/sqllogictest/test_files/joins.slt index f4dbd212eb8ee..c929822263bd8 100644 --- a/datafusion/sqllogictest/test_files/joins.slt +++ b/datafusion/sqllogictest/test_files/joins.slt @@ -3510,7 +3510,7 @@ logical_plan 08)----------TableScan: annotated_data projection=[a, b] physical_plan 01)ProjectionExec: expr=[a@0 as a, last_value(r.b) ORDER BY [r.a ASC NULLS FIRST]@3 as last_col1] -02)--AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b, c@2 as c], aggr=[last_value(r.b) ORDER BY [r.a ASC NULLS FIRST]], ordering_mode=PartiallySorted([0]) +02)--AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b, c@2 as c], aggr=[last_value(r.b) ORDER BY [r.a ASC NULLS FIRST]], group_completion_mode=Partial([0]) 03)----HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(a@0, a@0)] 04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, b, c], output_ordering=[a@0 ASC, b@1 ASC NULLS LAST, c@2 ASC NULLS LAST], file_type=csv, has_header=true 05)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, b], output_ordering=[a@0 ASC, b@1 ASC NULLS LAST], file_type=csv, has_header=true @@ -3557,7 +3557,7 @@ logical_plan 12)------------------TableScan: multiple_ordered_table projection=[a, d] physical_plan 01)ProjectionExec: expr=[last_value(l.d) ORDER BY [l.a ASC NULLS LAST]@1 as amount_usd] -02)--AggregateExec: mode=Single, gby=[row_n@2 as row_n], aggr=[last_value(l.d) ORDER BY [l.a ASC NULLS LAST]], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[row_n@2 as row_n], aggr=[last_value(l.d) ORDER BY [l.a ASC NULLS LAST]], group_completion_mode=Full 03)----HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(d@1, d@1)], filter=CAST(a@0 AS Int64) >= CAST(a@1 AS Int64) - 10, projection=[a@0, d@1, row_n@4] 04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, d], output_ordering=[a@0 ASC NULLS LAST], file_type=csv, has_header=true 05)------ProjectionExec: expr=[a@0 as a, d@1 as d, row_number() ORDER BY [r.a ASC NULLS LAST] RANGE BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW@2 as row_n] @@ -3592,9 +3592,9 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [a@0 ASC] 02)--ProjectionExec: expr=[a@0 as a, last_value(r.b) ORDER BY [r.a ASC NULLS FIRST]@3 as last_col1] -03)----AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b, c@2 as c], aggr=[last_value(r.b) ORDER BY [r.a ASC NULLS FIRST]], ordering_mode=PartiallySorted([0]) +03)----AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b, c@2 as c], aggr=[last_value(r.b) ORDER BY [r.a ASC NULLS FIRST]], group_completion_mode=Partial([0]) 04)------RepartitionExec: partitioning=Hash([a@0, b@1, c@2], 2), input_partitions=2, preserve_order=true, sort_exprs=a@0 ASC -05)--------AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b, c@2 as c], aggr=[last_value(r.b) ORDER BY [r.a ASC NULLS FIRST]], ordering_mode=PartiallySorted([0]) +05)--------AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b, c@2 as c], aggr=[last_value(r.b) ORDER BY [r.a ASC NULLS FIRST]], group_completion_mode=Partial([0]) 06)----------HashJoinExec: mode=Partitioned, join_type=Inner, on=[(a@0, a@0)] 07)------------RepartitionExec: partitioning=Hash([a@0], 2), input_partitions=1, maintains_sort_order=true 08)--------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, b, c], output_ordering=[a@0 ASC, b@1 ASC NULLS LAST, c@2 ASC NULLS LAST], file_type=csv, has_header=true diff --git a/datafusion/sqllogictest/test_files/order.slt b/datafusion/sqllogictest/test_files/order.slt index d974eb9dd06fc..343e5db9dd3b5 100644 --- a/datafusion/sqllogictest/test_files/order.slt +++ b/datafusion/sqllogictest/test_files/order.slt @@ -1898,7 +1898,7 @@ EXPLAIN SELECT c1, SUM(c2) as sum_c2 FROM table_with_ordered_not_null GROUP BY c ---- physical_plan 01)ProjectionExec: expr=[c1@0 as c1, sum(table_with_ordered_not_null.c2)@1 as sum_c2] -02)--AggregateExec: mode=Single, gby=[c1@0 as c1], aggr=[sum(table_with_ordered_not_null.c2)], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[c1@0 as c1], aggr=[sum(table_with_ordered_not_null.c2)], group_completion_mode=Full 03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/aggregate_agg_multi_order.csv]]}, projection=[c1, c2], output_ordering=[c1@0 ASC NULLS LAST], file_type=csv, has_header=true statement ok diff --git a/datafusion/sqllogictest/test_files/ordered_aggregate_spill.slt b/datafusion/sqllogictest/test_files/ordered_aggregate_spill.slt index 548dfc69b1b9a..a8e0ff065469c 100644 --- a/datafusion/sqllogictest/test_files/ordered_aggregate_spill.slt +++ b/datafusion/sqllogictest/test_files/ordered_aggregate_spill.slt @@ -55,9 +55,9 @@ FROM generate_series(20000) AS t1(v1) GROUP BY v1 ---- Plan with Metrics -01)AggregateExec: mode=FinalPartitioned,ordering_mode=Sorted, metrics=[spill_count=0,] +01)AggregateExec: mode=FinalPartitioned,group_completion_mode=Full, metrics=[spill_count=0,] 02)--RepartitionExec:preserve_order=true -03)----AggregateExec: mode=Partial,ordering_mode=Sorted, metrics=[spill_count=0,] +03)----AggregateExec: mode=Partial,group_completion_mode=Full, metrics=[spill_count=0,] query II rowsort @@ -100,9 +100,9 @@ FROM generate_series(20000) AS t1(v1) GROUP BY round(v1, -4), v1 % 5000 ---- Plan with Metrics -01)AggregateExec: mode=FinalPartitioned,aggr=[sum(t1.v1 * Int64(2))], ordering_mode=PartiallySorted([0]), metrics=[spill_count=0,] +01)AggregateExec: mode=FinalPartitioned,aggr=[sum(t1.v1 * Int64(2))], group_completion_mode=Partial([0]), metrics=[spill_count=0,] 02)--RepartitionExec:input_partitions=1, maintains_sort_order=true -03)----AggregateExec: mode=Partial,ordering_mode=PartiallySorted([0]), metrics=[spill_count=0,] +03)----AggregateExec: mode=Partial,group_completion_mode=Partial([0]), metrics=[spill_count=0,] # All rounds should have the same result hash @@ -124,9 +124,9 @@ FROM generate_series(20000) AS t1(v1) GROUP BY round(v1, -4), v1 % 5000 ---- Plan with Metrics -01)AggregateExec: mode=FinalPartitioned,aggr=[sum(t1.v1 * Int64(2))], ordering_mode=PartiallySorted([0]), metrics=[spilled_bytes= KB,] +01)AggregateExec: mode=FinalPartitioned,aggr=[sum(t1.v1 * Int64(2))], group_completion_mode=Partial([0]), metrics=[spilled_bytes= KB,] 02)--RepartitionExec:input_partitions=1, maintains_sort_order=true -03)----AggregateExec: mode=Partial,ordering_mode=PartiallySorted([0]), metrics=[spill_count=0,] +03)----AggregateExec: mode=Partial,group_completion_mode=Partial([0]), metrics=[spill_count=0,] # All rounds should have the same result hash @@ -148,9 +148,9 @@ FROM generate_series(20000) AS t1(v1) GROUP BY round(v1, -4), v1 % 5000 ---- Plan with Metrics -01)AggregateExec: mode=FinalPartitioned,aggr=[sum(t1.v1 * Int64(2))], ordering_mode=PartiallySorted([0]), metrics=[spilled_bytes= KB,] +01)AggregateExec: mode=FinalPartitioned,aggr=[sum(t1.v1 * Int64(2))], group_completion_mode=Partial([0]), metrics=[spilled_bytes= KB,] 02)--RepartitionExec:input_partitions=1, maintains_sort_order=true -03)----AggregateExec: mode=Partial,ordering_mode=PartiallySorted([0]), metrics=[spill_count=0,] +03)----AggregateExec: mode=Partial,group_completion_mode=Partial([0]), metrics=[spill_count=0,] # All rounds should have the same result hash @@ -176,9 +176,9 @@ FROM generate_series(20000) AS t1(v1) GROUP BY round(v1, -4), v1 % 5000 ---- Plan with Metrics -01)AggregateExec: mode=FinalPartitioned,aggr=[sum(t1.v1 * Int64(2)), min(t1.v1 % Int64(2))], ordering_mode=PartiallySorted([0]), metrics=[spilled_rows= K,] +01)AggregateExec: mode=FinalPartitioned,aggr=[sum(t1.v1 * Int64(2)), min(t1.v1 % Int64(2))], group_completion_mode=Partial([0]), metrics=[spilled_rows= K,] 02)--RepartitionExec:input_partitions=1, maintains_sort_order=true -03)----AggregateExec: mode=Partial,aggr=[sum(t1.v1 * Int64(2)), min(t1.v1 % Int64(2))], ordering_mode=PartiallySorted([0]), metrics=[spill_count=0,] +03)----AggregateExec: mode=Partial,aggr=[sum(t1.v1 * Int64(2)), min(t1.v1 % Int64(2))], group_completion_mode=Partial([0]), metrics=[spill_count=0,] # ================================================================================== @@ -201,7 +201,7 @@ FROM generate_series(20000) AS t1(v1) GROUP BY round(v1, -4), v1 % 5000 ---- Plan with Metrics -01)AggregateExec: mode=Single,aggr=[min(t1.v1 * Int64(2)), avg(t1.v1)], ordering_mode=PartiallySorted([0]), metrics=[spill_count=0,] +01)AggregateExec: mode=Single,aggr=[min(t1.v1 * Int64(2)), avg(t1.v1)], group_completion_mode=Partial([0]), metrics=[spill_count=0,] query IIIR rowsort @@ -222,7 +222,7 @@ FROM generate_series(20000) AS t1(v1) GROUP BY round(v1, -4), v1 % 5000 ---- Plan with Metrics -01)AggregateExec: mode=Single,aggr=[min(t1.v1 * Int64(2)), avg(t1.v1)], ordering_mode=PartiallySorted([0]), metrics=[spilled_bytes= KB,] +01)AggregateExec: mode=Single,aggr=[min(t1.v1 * Int64(2)), avg(t1.v1)], group_completion_mode=Partial([0]), metrics=[spilled_bytes= KB,] # Same result hash as the no-spill round above diff --git a/datafusion/sqllogictest/test_files/preserve_file_partitioning.slt b/datafusion/sqllogictest/test_files/preserve_file_partitioning.slt index e2dd22cc82bba..7e187669ce56d 100644 --- a/datafusion/sqllogictest/test_files/preserve_file_partitioning.slt +++ b/datafusion/sqllogictest/test_files/preserve_file_partitioning.slt @@ -287,9 +287,9 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, count(Int64(1))@1 as count(*), avg(fact_table_ordered.value)@2 as avg(fact_table_ordered.value)] -03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted +03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], group_completion_mode=Full 04)------RepartitionExec: partitioning=Hash([f_dkey@0], 3), input_partitions=3, preserve_order=true, sort_exprs=f_dkey@0 ASC NULLS LAST -05)--------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted +05)--------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], group_completion_mode=Full 06)----------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=C/data.parquet]]}, projection=[value, f_dkey], output_ordering=[f_dkey@1 ASC NULLS LAST], file_type=parquet # Verify results without optimization @@ -319,7 +319,7 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, count(Int64(1))@1 as count(*), avg(fact_table_ordered.value)@2 as avg(fact_table_ordered.value)] -03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted +03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], group_completion_mode=Full 04)------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=C/data.parquet]]}, projection=[value, f_dkey], output_ordering=[f_dkey@1 ASC NULLS LAST], output_partitioning=Hash([f_dkey@1], 3), file_type=parquet query TIR @@ -359,9 +359,9 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, max(d.env)@1 as max(d.env), max(d.service)@2 as max(d.service), count(Int64(1))@3 as count(*), sum(f.value)@4 as sum(f.value)] -03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[max(d.env), max(d.service), count(Int64(1)), sum(f.value)], ordering_mode=Sorted +03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[max(d.env), max(d.service), count(Int64(1)), sum(f.value)], group_completion_mode=Full 04)------RepartitionExec: partitioning=Hash([f_dkey@0], 3), input_partitions=3, preserve_order=true, sort_exprs=f_dkey@0 ASC NULLS LAST -05)--------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[max(d.env), max(d.service), count(Int64(1)), sum(f.value)], ordering_mode=Sorted +05)--------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[max(d.env), max(d.service), count(Int64(1)), sum(f.value)], group_completion_mode=Full 06)----------HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(d_dkey@0, f_dkey@1)], projection=[value@3, f_dkey@4, env@1, service@2] 07)------------CoalescePartitionsExec 08)--------------FilterExec: service@2 = log @@ -412,7 +412,7 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, max(d.env)@1 as max(d.env), max(d.service)@2 as max(d.service), count(Int64(1))@3 as count(*), sum(f.value)@4 as sum(f.value)] -03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[max(d.env), max(d.service), count(Int64(1)), sum(f.value)], ordering_mode=Sorted +03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[max(d.env), max(d.service), count(Int64(1)), sum(f.value)], group_completion_mode=Full 04)------HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(d_dkey@0, f_dkey@1)], projection=[value@3, f_dkey@4, env@1, service@2] 05)--------CoalescePartitionsExec 06)----------FilterExec: service@2 = log diff --git a/datafusion/sqllogictest/test_files/range_sorted_time_bin_agg.slt b/datafusion/sqllogictest/test_files/range_sorted_time_bin_agg.slt index 18123a492dbd6..058d728f505cb 100644 --- a/datafusion/sqllogictest/test_files/range_sorted_time_bin_agg.slt +++ b/datafusion/sqllogictest/test_files/range_sorted_time_bin_agg.slt @@ -35,9 +35,9 @@ # single streaming SinglePartitioned step with no hash shuffle. # # Today's plan still hash-repartitions: -# Partial AggregateExec (ordering_mode=Sorted) +# Partial AggregateExec (group_completion_mode=Full) # -> RepartitionExec Hash([key, date_bin(...)]) -# -> FinalPartitioned AggregateExec (ordering_mode=Sorted) +# -> FinalPartitioned AggregateExec (group_completion_mode=Full) statement ok set datafusion.explain.physical_plan_only = true; @@ -86,7 +86,7 @@ physical_plan DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion # overlap across the two 60-minute streams. # # Today this is still Partial + hash RepartitionExec + Final, even though -# ordering_mode=Sorted is already recognized. +# group_completion_mode=Full is already recognized. ########## query TT @@ -97,9 +97,9 @@ GROUP BY key, time_bin; ---- physical_plan 01)ProjectionExec: expr=[key@0 as key, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1 as time_bin, sum(range_sorted_time_bin.value)@2 as sum(range_sorted_time_bin.value)] -02)--AggregateExec: mode=FinalPartitioned, gby=[key@0 as key, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)], aggr=[sum(range_sorted_time_bin.value)], ordering_mode=Sorted +02)--AggregateExec: mode=FinalPartitioned, gby=[key@0 as key, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)], aggr=[sum(range_sorted_time_bin.value)], group_completion_mode=Full 03)----RepartitionExec: partitioning=Hash([key@0, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1], 2), input_partitions=2, preserve_order=true, sort_exprs=key@0 ASC, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1 ASC -04)------AggregateExec: mode=Partial, gby=[key@0 as key, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }, timestamp@1) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)], aggr=[sum(range_sorted_time_bin.value)], ordering_mode=Sorted +04)------AggregateExec: mode=Partial, gby=[key@0 as key, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }, timestamp@1) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)], aggr=[sum(range_sorted_time_bin.value)], group_completion_mode=Full 05)--------FilterExec: col4@1 = a, projection=[key@0, timestamp@2, value@3] 06)----------DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch_range_partitioning/range_sorted_time_bin/part-0.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch_range_partitioning/range_sorted_time_bin/part-1.parquet]]}, projection=[key, col4, timestamp, value], output_ordering=[key@0 ASC, timestamp@2 ASC], output_partitioning=Range([timestamp@2 ASC], [(1704070800000000000)], 2), file_type=parquet, predicate=col4@4 = a, pruning_predicate=col4_null_count@2 != row_count@3 AND col4_min@0 <= a AND a <= col4_max@1, required_guarantees=[col4 in (a)] @@ -128,9 +128,9 @@ GROUP BY key, time_bin; ---- physical_plan 01)ProjectionExec: expr=[key@0 as key, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1 as time_bin, sum(range_sorted_time_bin.value)@2 as sum(range_sorted_time_bin.value)] -02)--AggregateExec: mode=FinalPartitioned, gby=[key@0 as key, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)], aggr=[sum(range_sorted_time_bin.value)], ordering_mode=Sorted +02)--AggregateExec: mode=FinalPartitioned, gby=[key@0 as key, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)], aggr=[sum(range_sorted_time_bin.value)], group_completion_mode=Full 03)----RepartitionExec: partitioning=Hash([key@0, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1], 2), input_partitions=2, preserve_order=true, sort_exprs=key@0 ASC, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1 ASC -04)------AggregateExec: mode=Partial, gby=[key@0 as key, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }, timestamp@1) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)], aggr=[sum(range_sorted_time_bin.value)], ordering_mode=Sorted +04)------AggregateExec: mode=Partial, gby=[key@0 as key, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }, timestamp@1) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)], aggr=[sum(range_sorted_time_bin.value)], group_completion_mode=Full 05)--------DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch_range_partitioning/range_sorted_time_bin/part-0.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch_range_partitioning/range_sorted_time_bin/part-1.parquet]]}, projection=[key, timestamp, value], output_ordering=[key@0 ASC, timestamp@1 ASC], output_partitioning=Range([timestamp@1 ASC], [(1704070800000000000)], 2), file_type=parquet query TPI diff --git a/datafusion/sqllogictest/test_files/repartition_subset_satisfaction.slt b/datafusion/sqllogictest/test_files/repartition_subset_satisfaction.slt index 5371ca59beea1..5459fc9512ab6 100644 --- a/datafusion/sqllogictest/test_files/repartition_subset_satisfaction.slt +++ b/datafusion/sqllogictest/test_files/repartition_subset_satisfaction.slt @@ -161,9 +161,9 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST, time_bin@1 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1 as time_bin, count(Int64(1))@2 as count(*), avg(fact_table_ordered.value)@3 as avg(fact_table_ordered.value)] -03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted +03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], group_completion_mode=Full 04)------RepartitionExec: partitioning=Hash([f_dkey@0, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1], 3), input_partitions=3, preserve_order=true, sort_exprs=f_dkey@0 ASC NULLS LAST, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1 ASC NULLS LAST -05)--------AggregateExec: mode=Partial, gby=[f_dkey@2 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@0) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted +05)--------AggregateExec: mode=Partial, gby=[f_dkey@2 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@0) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], group_completion_mode=Full 06)----------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=C/data.parquet]]}, projection=[timestamp, value, f_dkey], output_ordering=[f_dkey@2 ASC NULLS LAST, timestamp@0 ASC NULLS LAST], output_partitioning=Hash([f_dkey@2], 3), file_type=parquet # Verify results without subset satisfaction @@ -203,7 +203,7 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST, time_bin@1 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1 as time_bin, count(Int64(1))@2 as count(*), avg(fact_table_ordered.value)@3 as avg(fact_table_ordered.value)] -03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@2 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@0) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted +03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@2 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@0) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], group_completion_mode=Full 04)------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=C/data.parquet]]}, projection=[timestamp, value, f_dkey], output_ordering=[f_dkey@2 ASC NULLS LAST, timestamp@0 ASC NULLS LAST], output_partitioning=Hash([f_dkey@2], 3), file_type=parquet # Verify results match with subset satisfaction @@ -373,9 +373,9 @@ physical_plan 05)--------RepartitionExec: partitioning=Hash([env@0, time_bin@1], 3), input_partitions=3 06)----------AggregateExec: mode=Partial, gby=[env@1 as env, time_bin@0 as time_bin], aggr=[avg(a.max_bin_value)] 07)------------ProjectionExec: expr=[date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp)@1 as time_bin, env@2 as env, max(j.value)@3 as max_bin_value] -08)--------------AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp), env@2 as env], aggr=[max(j.value)], ordering_mode=PartiallySorted([0, 1]) +08)--------------AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp), env@2 as env], aggr=[max(j.value)], group_completion_mode=Partial([0, 1]) 09)----------------RepartitionExec: partitioning=Hash([f_dkey@0, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp)@1, env@2], 3), input_partitions=3, preserve_order=true, sort_exprs=f_dkey@0 ASC NULLS LAST, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp)@1 ASC NULLS LAST -10)------------------AggregateExec: mode=Partial, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@2) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp), env@1 as env], aggr=[max(j.value)], ordering_mode=PartiallySorted([0, 1]) +10)------------------AggregateExec: mode=Partial, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@2) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp), env@1 as env], aggr=[max(j.value)], group_completion_mode=Partial([0, 1]) 11)--------------------HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(d_dkey@1, f_dkey@2)], projection=[f_dkey@4, env@0, timestamp@2, value@3] 12)----------------------CoalescePartitionsExec 13)------------------------FilterExec: service@1 = log, projection=[env@0, d_dkey@2] @@ -470,7 +470,7 @@ physical_plan 05)--------RepartitionExec: partitioning=Hash([env@0, time_bin@1], 3), input_partitions=3 06)----------AggregateExec: mode=Partial, gby=[env@1 as env, time_bin@0 as time_bin], aggr=[avg(a.max_bin_value)] 07)------------ProjectionExec: expr=[date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp)@1 as time_bin, env@2 as env, max(j.value)@3 as max_bin_value] -08)--------------AggregateExec: mode=SinglePartitioned, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@2) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp), env@1 as env], aggr=[max(j.value)], ordering_mode=PartiallySorted([0, 1]) +08)--------------AggregateExec: mode=SinglePartitioned, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@2) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp), env@1 as env], aggr=[max(j.value)], group_completion_mode=Partial([0, 1]) 09)----------------HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(d_dkey@1, f_dkey@2)], projection=[f_dkey@4, env@0, timestamp@2, value@3] 10)------------------CoalescePartitionsExec 11)--------------------FilterExec: service@1 = log, projection=[env@0, d_dkey@2] diff --git a/datafusion/sqllogictest/test_files/sort_pushdown.slt b/datafusion/sqllogictest/test_files/sort_pushdown.slt index a173e76d6c262..20422dd95a614 100644 --- a/datafusion/sqllogictest/test_files/sort_pushdown.slt +++ b/datafusion/sqllogictest/test_files/sort_pushdown.slt @@ -912,7 +912,7 @@ logical_plan 03)----TableScan: agg_expr_parquet projection=[x, y, v] physical_plan 01)SortExec: expr=[x@0 ASC NULLS LAST, agg_expr_parquet.y % Int64(2)@1 ASC NULLS LAST], preserve_partitioning=[false] -02)--AggregateExec: mode=Single, gby=[x@0 as x, CAST(y@1 AS Int64) % 2 as agg_expr_parquet.y % Int64(2)], aggr=[sum(agg_expr_parquet.v)], ordering_mode=PartiallySorted([0]) +02)--AggregateExec: mode=Single, gby=[x@0 as x, CAST(y@1 AS Int64) % 2 as agg_expr_parquet.y % Int64(2)], aggr=[sum(agg_expr_parquet.v)], group_completion_mode=Partial([0]) 03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/sort_pushdown/agg_expr_sorted.parquet]]}, projection=[x, y, v], output_ordering=[x@0 ASC NULLS LAST, y@1 ASC NULLS LAST], file_type=parquet # Expected output pattern from ORDER BY [x, bucket]: @@ -946,7 +946,7 @@ logical_plan 02)--Aggregate: groupBy=[[agg_expr_parquet.x, CAST(agg_expr_parquet.y AS Int64)]], aggr=[[sum(CAST(agg_expr_parquet.v AS Int64))]] 03)----TableScan: agg_expr_parquet projection=[x, y, v] physical_plan -01)AggregateExec: mode=Single, gby=[x@0 as x, CAST(y@1 AS Int64) as agg_expr_parquet.y], aggr=[sum(agg_expr_parquet.v)], ordering_mode=Sorted +01)AggregateExec: mode=Single, gby=[x@0 as x, CAST(y@1 AS Int64) as agg_expr_parquet.y], aggr=[sum(agg_expr_parquet.v)], group_completion_mode=Full 02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/sort_pushdown/agg_expr_sorted.parquet]]}, projection=[x, y, v], output_ordering=[x@0 ASC NULLS LAST, y@1 ASC NULLS LAST], file_type=parquet query III @@ -978,7 +978,7 @@ logical_plan 03)----TableScan: agg_expr_parquet projection=[x, v] physical_plan 01)SortExec: expr=[sum(agg_expr_parquet.v)@1 ASC NULLS LAST], preserve_partitioning=[false] -02)--AggregateExec: mode=Single, gby=[x@0 as x], aggr=[sum(agg_expr_parquet.v)], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[x@0 as x], aggr=[sum(agg_expr_parquet.v)], group_completion_mode=Full 03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/sort_pushdown/agg_expr_sorted.parquet]]}, projection=[x, v], output_ordering=[x@0 ASC NULLS LAST], file_type=parquet query II @@ -1034,7 +1034,7 @@ logical_plan 03)----TableScan: agg_expr_parquet projection=[x, v] physical_plan 01)SortExec: expr=[CAST(x@0 AS Int64) + 1 DESC], preserve_partitioning=[false] -02)--AggregateExec: mode=Single, gby=[x@0 as x], aggr=[sum(agg_expr_parquet.v)], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[x@0 as x], aggr=[sum(agg_expr_parquet.v)], group_completion_mode=Full 03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/sort_pushdown/agg_expr_sorted.parquet]]}, projection=[x, v], output_ordering=[x@0 ASC NULLS LAST], file_type=parquet query II @@ -1060,7 +1060,7 @@ logical_plan 03)----TableScan: agg_expr_parquet projection=[x, v] physical_plan 01)SortExec: expr=[2 * CAST(x@0 AS Int64) ASC NULLS LAST], preserve_partitioning=[false] -02)--AggregateExec: mode=Single, gby=[x@0 as x], aggr=[sum(agg_expr_parquet.v)], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[x@0 as x], aggr=[sum(agg_expr_parquet.v)], group_completion_mode=Full 03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/sort_pushdown/agg_expr_sorted.parquet]]}, projection=[x, v], output_ordering=[x@0 ASC NULLS LAST], file_type=parquet query II diff --git a/datafusion/sqllogictest/test_files/unnest.slt b/datafusion/sqllogictest/test_files/unnest.slt index 8e6013328f98f..c746adaa18e1c 100644 --- a/datafusion/sqllogictest/test_files/unnest.slt +++ b/datafusion/sqllogictest/test_files/unnest.slt @@ -988,9 +988,9 @@ logical_plan 08)--------------TableScan: range() projection=[value] physical_plan 01)ProjectionExec: expr=[array_agg(unnested.ar)@1 as array_agg(unnested.ar)] -02)--AggregateExec: mode=FinalPartitioned, gby=[generated_id@0 as generated_id], aggr=[array_agg(unnested.ar)], ordering_mode=Sorted +02)--AggregateExec: mode=FinalPartitioned, gby=[generated_id@0 as generated_id], aggr=[array_agg(unnested.ar)], group_completion_mode=Full 03)----RepartitionExec: partitioning=Hash([generated_id@0], 4), input_partitions=4, preserve_order=true, sort_exprs=generated_id@0 ASC NULLS LAST -04)------AggregateExec: mode=Partial, gby=[generated_id@0 as generated_id], aggr=[array_agg(unnested.ar)], ordering_mode=Sorted +04)------AggregateExec: mode=Partial, gby=[generated_id@0 as generated_id], aggr=[array_agg(unnested.ar)], group_completion_mode=Full 05)--------ProjectionExec: expr=[generated_id@0 as generated_id, __unnest_placeholder(make_array(range().value),depth=1)@1 as ar] 06)----------UnnestExec 07)------------ProjectionExec: expr=[row_number() ROWS BETWEEN UNBOUNDED PRECEDING AND UNBOUNDED FOLLOWING@1 as generated_id, make_array(value@0) as __unnest_placeholder(make_array(range().value))] diff --git a/datafusion/sqllogictest/test_files/window.slt b/datafusion/sqllogictest/test_files/window.slt index a6b64b98de0e3..36811abf0ad00 100644 --- a/datafusion/sqllogictest/test_files/window.slt +++ b/datafusion/sqllogictest/test_files/window.slt @@ -357,7 +357,7 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [b@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[b@0 as b, max(d.a)@1 as max_a, max(d.seq)@2 as max(d.seq)] -03)----AggregateExec: mode=SinglePartitioned, gby=[b@2 as b], aggr=[max(d.a), max(d.seq)], ordering_mode=Sorted +03)----AggregateExec: mode=SinglePartitioned, gby=[b@2 as b], aggr=[max(d.a), max(d.seq)], group_completion_mode=Full 04)------ProjectionExec: expr=[row_number() PARTITION BY [s.b] ORDER BY [s.a ASC NULLS LAST] RANGE BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW@2 as seq, a@0 as a, b@1 as b] 05)--------BoundedWindowAggExec: wdw=[row_number() PARTITION BY [s.b] ORDER BY [s.a ASC NULLS LAST] RANGE BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW: Field { "row_number() PARTITION BY [s.b] ORDER BY [s.a ASC NULLS LAST] RANGE BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW": UInt64 }, frame: RANGE BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW], mode=[Sorted] 06)----------SortExec: expr=[b@1 ASC NULLS LAST, a@0 ASC NULLS LAST], preserve_partitioning=[true] diff --git a/docs/source/library-user-guide/upgrading/56.0.0.md b/docs/source/library-user-guide/upgrading/56.0.0.md index 1ea0f4671009d..d3958da7e89f6 100644 --- a/docs/source/library-user-guide/upgrading/56.0.0.md +++ b/docs/source/library-user-guide/upgrading/56.0.0.md @@ -75,6 +75,28 @@ The Minimum Supported Rust Version (MSRV) has been updated to [`1.95.0`]. [`1.95.0`]: https://releases.rs/docs/1.95.0/ +### Aggregate group-completion APIs + +`AggregateExec::input_order_mode()` has been replaced by +`AggregateExec::group_completion_mode()`. It returns a `GroupCompletionMode`: +`None`, `Partial(indices)`, or `Full`, corresponding to the previous aggregate +modes `Linear`, `PartiallySorted(indices)`, and `Sorted`. +`AggregateExec::compute_properties` now accepts `&GroupCompletionMode` in place of +`&InputOrderMode`. + +Completion describes when groups can be emitted, not their sort order. Use +`ExecutionPlanProperties::output_ordering()` to inspect actual output ordering. +`InputOrderMode` remains available for window execution. + +In `datafusion_physical_plan::aggregates::order`, `GroupOrdering`, +`GroupOrderingPartial`, and `GroupOrderingFull` have been renamed to +`GroupCompletion`, `GroupCompletionPartial`, and `GroupCompletionFull`. +`GroupCompletion::try_new` accepts `&GroupCompletionMode`. + +Aggregate plans now display `group_completion_mode=Full` or +`group_completion_mode=Partial(indices)` instead of `ordering_mode=Sorted` or +`ordering_mode=PartiallySorted(indices)`. + ### Upgrade arrow/parquet to 60.0.0 and object_store to 0.14.2 DataFusion 56.0.0 uses `arrow` and `parquet` 60.0.0, and `object_store` 0.14.2. From 7b2a483881a46ef59b4c53dfba86cfc06a302d0f Mon Sep 17 00:00:00 2001 From: "xavier.lee" Date: Sun, 4 Oct 2026 23:54:24 -0400 Subject: [PATCH 2/5] docs: clarify aggregate completion migration guidance --- docs/source/library-user-guide/upgrading/56.0.0.md | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/docs/source/library-user-guide/upgrading/56.0.0.md b/docs/source/library-user-guide/upgrading/56.0.0.md index d3958da7e89f6..988735a838fc7 100644 --- a/docs/source/library-user-guide/upgrading/56.0.0.md +++ b/docs/source/library-user-guide/upgrading/56.0.0.md @@ -84,9 +84,8 @@ modes `Linear`, `PartiallySorted(indices)`, and `Sorted`. `AggregateExec::compute_properties` now accepts `&GroupCompletionMode` in place of `&InputOrderMode`. -Completion describes when groups can be emitted, not their sort order. Use -`ExecutionPlanProperties::output_ordering()` to inspect actual output ordering. -`InputOrderMode` remains available for window execution. +`GroupCompletionMode` describes when groups can be emitted. To inspect the +aggregate's output ordering, use `ExecutionPlanProperties::output_ordering()`. In `datafusion_physical_plan::aggregates::order`, `GroupOrdering`, `GroupOrderingPartial`, and `GroupOrderingFull` have been renamed to From ce2345fc0b90731c85958f9694c1456e3c5f461c Mon Sep 17 00:00:00 2001 From: "xavier.lee" Date: Mon, 5 Oct 2026 00:07:34 -0400 Subject: [PATCH 3/5] docs: describe group completion clustering guarantees --- docs/source/library-user-guide/upgrading/56.0.0.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/source/library-user-guide/upgrading/56.0.0.md b/docs/source/library-user-guide/upgrading/56.0.0.md index 988735a838fc7..8163ab3fc182d 100644 --- a/docs/source/library-user-guide/upgrading/56.0.0.md +++ b/docs/source/library-user-guide/upgrading/56.0.0.md @@ -84,8 +84,8 @@ modes `Linear`, `PartiallySorted(indices)`, and `Sorted`. `AggregateExec::compute_properties` now accepts `&GroupCompletionMode` in place of `&InputOrderMode`. -`GroupCompletionMode` describes when groups can be emitted. To inspect the -aggregate's output ordering, use `ExecutionPlanProperties::output_ordering()`. +`GroupCompletionMode` describes the input's group-clustering guarantees, which +determine when groups can be emitted during aggregation. In `datafusion_physical_plan::aggregates::order`, `GroupOrdering`, `GroupOrderingPartial`, and `GroupOrderingFull` have been renamed to From cca3e3ff767cd26345a9265babdac2bb274ef848 Mon Sep 17 00:00:00 2001 From: "xavier.lee" Date: Wed, 7 Oct 2026 18:23:24 -0400 Subject: [PATCH 4/5] refactor: unify aggregate group clustering terminology --- datafusion/core/tests/dataframe/mod.rs | 8 +- .../core/tests/fuzz_cases/aggregate_fuzz.rs | 12 +- .../aggregation_fuzzer/query_builder.rs | 4 +- .../enforce_distribution.rs | 8 +- .../limited_distinct_aggregation.rs | 2 +- .../tests/physical_optimizer/pushdown_sort.rs | 4 +- .../benches/dictionary_group_values.rs | 14 +- .../benches/ordered_group_values.rs | 18 +-- .../physical-plan/benches/partial_ordering.rs | 10 +- .../clustered_final_table.rs | 10 +- .../clustered_partial_table.rs | 6 +- .../clustered_single_table.rs | 4 +- .../aggregates/aggregate_hash_table/common.rs | 4 +- .../aggregate_hash_table/common_clustered.rs | 36 ++--- .../aggregate_hash_table/partial_table.rs | 4 +- .../src/aggregates/clustered_final_stream.rs | 42 +++--- .../aggregates/clustered_partial_stream.rs | 28 ++-- .../src/aggregates/clustered_single_stream.rs | 24 ++-- .../src/aggregates/group_values/mod.rs | 16 +-- .../group_values/multi_group_by/clustered.rs | 10 +- .../group_values/multi_group_by/list.rs | 12 +- .../src/aggregates/grouped_hash_stream.rs | 48 +++---- .../src/aggregates/hash_stream.rs | 14 +- .../physical-plan/src/aggregates/mod.rs | 126 +++++++++--------- .../src/aggregates/order/full.rs | 8 +- .../physical-plan/src/aggregates/order/mod.rs | 117 ++++++++-------- .../src/aggregates/order/partial.rs | 28 ++-- .../src/aggregates/partial_reduce_stream.rs | 4 +- .../src/aggregates/single_stream.rs | 6 +- .../physical-plan/src/aggregates/spill.rs | 22 +-- .../physical-plan/src/recursive_query.rs | 4 +- .../test_files/agg_func_substitute.slt | 12 +- .../sqllogictest/test_files/aggregate.slt | 12 +- .../sqllogictest/test_files/group_by.slt | 58 ++++---- datafusion/sqllogictest/test_files/joins.slt | 8 +- datafusion/sqllogictest/test_files/order.slt | 2 +- .../test_files/ordered_aggregate_spill.slt | 24 ++-- .../test_files/preserve_file_partitioning.slt | 12 +- .../test_files/range_sorted_time_bin_agg.slt | 14 +- .../repartition_subset_satisfaction.slt | 12 +- .../sqllogictest/test_files/sort_pushdown.slt | 10 +- datafusion/sqllogictest/test_files/unnest.slt | 4 +- datafusion/sqllogictest/test_files/window.slt | 2 +- .../library-user-guide/upgrading/56.0.0.md | 16 +-- 44 files changed, 420 insertions(+), 419 deletions(-) diff --git a/datafusion/core/tests/dataframe/mod.rs b/datafusion/core/tests/dataframe/mod.rs index 9ffe34893b9e1..d011a25273fdf 100644 --- a/datafusion/core/tests/dataframe/mod.rs +++ b/datafusion/core/tests/dataframe/mod.rs @@ -3422,9 +3422,9 @@ async fn union_with_mix_of_presorted_and_explicitly_resorted_inputs_with_reparti assert_snapshot!( union_with_mix_of_presorted_and_explicitly_resorted_inputs_impl(false).await?, @r" - AggregateExec: mode=Final, gby=[id@0 as id], aggr=[], group_completion_mode=Full + AggregateExec: mode=Final, gby=[id@0 as id], aggr=[], group_clustering_mode=Full SortPreservingMergeExec: [id@0 ASC NULLS LAST] - AggregateExec: mode=Partial, gby=[id@0 as id], aggr=[], group_completion_mode=Full + AggregateExec: mode=Partial, gby=[id@0 as id], aggr=[], group_clustering_mode=Full UnionExec DataSourceExec: file_groups={1 group: [[{testdata}/alltypes_tiny_pages.parquet]]}, projection=[id], output_ordering=[id@0 ASC NULLS LAST], file_type=parquet SortExec: expr=[id@0 ASC NULLS LAST], preserve_partitioning=[false] @@ -3440,9 +3440,9 @@ async fn union_with_mix_of_presorted_and_explicitly_resorted_inputs_with_reparti assert_snapshot!( union_with_mix_of_presorted_and_explicitly_resorted_inputs_impl(true).await?, @r" - AggregateExec: mode=Final, gby=[id@0 as id], aggr=[], group_completion_mode=Full + AggregateExec: mode=Final, gby=[id@0 as id], aggr=[], group_clustering_mode=Full SortPreservingMergeExec: [id@0 ASC NULLS LAST] - AggregateExec: mode=Partial, gby=[id@0 as id], aggr=[], group_completion_mode=Full + AggregateExec: mode=Partial, gby=[id@0 as id], aggr=[], group_clustering_mode=Full UnionExec DataSourceExec: file_groups={1 group: [[{testdata}/alltypes_tiny_pages.parquet]]}, projection=[id], output_ordering=[id@0 ASC NULLS LAST], file_type=parquet SortExec: expr=[id@0 ASC NULLS LAST], preserve_partitioning=[false] diff --git a/datafusion/core/tests/fuzz_cases/aggregate_fuzz.rs b/datafusion/core/tests/fuzz_cases/aggregate_fuzz.rs index 5db48946ea3cb..74cc881c22baf 100644 --- a/datafusion/core/tests/fuzz_cases/aggregate_fuzz.rs +++ b/datafusion/core/tests/fuzz_cases/aggregate_fuzz.rs @@ -48,7 +48,7 @@ use datafusion_execution::memory_pool::FairSpillPool; use datafusion_execution::runtime_env::RuntimeEnvBuilder; use datafusion_physical_expr::aggregate::AggregateExprBuilder; use datafusion_physical_plan::aggregates::{ - AggregateExec, AggregateMode, GroupCompletionMode, PhysicalGroupBy, + AggregateExec, AggregateMode, GroupClusteringMode, PhysicalGroupBy, }; use datafusion_physical_plan::metrics::MetricValue; use datafusion_physical_plan::{ExecutionPlan, collect, displayable}; @@ -353,8 +353,8 @@ async fn run_aggregate_test(input1: Vec, group_by_columns: Vec<&str .unwrap(), ); assert_ne!( - aggregate_exec_running.group_completion_mode(), - &GroupCompletionMode::None, + aggregate_exec_running.group_clustering_mode(), + &GroupClusteringMode::None, "running aggregate should observe ordered input for group_by: {group_by:?}" ); @@ -560,11 +560,11 @@ async fn verify_ordered_aggregate(frame: &DataFrame, expected_sort: bool) { ); if self.expected_sort { assert!(matches!( - exec.group_completion_mode(), - GroupCompletionMode::Partial(_) | GroupCompletionMode::Full + exec.group_clustering_mode(), + GroupClusteringMode::Partial(_) | GroupClusteringMode::Full )); } else { - assert_eq!(*exec.group_completion_mode(), GroupCompletionMode::None); + assert_eq!(*exec.group_clustering_mode(), GroupClusteringMode::None); } } Ok(TreeNodeRecursion::Continue) diff --git a/datafusion/core/tests/fuzz_cases/aggregation_fuzzer/query_builder.rs b/datafusion/core/tests/fuzz_cases/aggregation_fuzzer/query_builder.rs index debeceb433293..b4df25a00660c 100644 --- a/datafusion/core/tests/fuzz_cases/aggregation_fuzzer/query_builder.rs +++ b/datafusion/core/tests/fuzz_cases/aggregation_fuzzer/query_builder.rs @@ -93,9 +93,9 @@ pub struct QueryBuilder { /// ... /// ``` /// - /// More details can see [`GroupCompletion`]. + /// More details can see [`GroupClustering`]. /// - /// [`GroupCompletion`]: datafusion_physical_plan::aggregates::order::GroupCompletion + /// [`GroupClustering`]: datafusion_physical_plan::aggregates::order::GroupClustering dataset_sort_keys: Vec>, /// If we will also test the no grouping case like: diff --git a/datafusion/core/tests/physical_optimizer/enforce_distribution.rs b/datafusion/core/tests/physical_optimizer/enforce_distribution.rs index 66e14ec8b1181..7fee51f528035 100644 --- a/datafusion/core/tests/physical_optimizer/enforce_distribution.rs +++ b/datafusion/core/tests/physical_optimizer/enforce_distribution.rs @@ -4867,9 +4867,9 @@ fn preserve_ordering_for_streaming_sorted_aggregate() -> Result<()> { let plan_distrib = test_config.to_plan(physical_plan.clone(), &DISTRIB_DISTRIB_SORT); assert_plan!(plan_distrib, @r" - AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[COUNT(b)], group_completion_mode=Full + AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[COUNT(b)], group_clustering_mode=Full RepartitionExec: partitioning=Hash([a@0], 2), input_partitions=2, preserve_order=true, sort_exprs=a@0 ASC - AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[COUNT(b)], group_completion_mode=Full + AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[COUNT(b)], group_clustering_mode=Full DataSourceExec: file_groups={2 groups: [[x], [y]]}, projection=[a, b, c, d, e], output_ordering=[a@0 ASC], file_type=parquet "); @@ -4901,9 +4901,9 @@ fn preserve_ordering_for_streaming_partially_sorted_aggregate() -> Result<()> { let plan_distrib = test_config.to_plan(physical_plan.clone(), &DISTRIB_DISTRIB_SORT); assert_plan!(plan_distrib, @r" - AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b], aggr=[COUNT(c)], group_completion_mode=Partial([0]) + AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b], aggr=[COUNT(c)], group_clustering_mode=Partial([0]) RepartitionExec: partitioning=Hash([a@0, b@1], 2), input_partitions=2, preserve_order=true, sort_exprs=a@0 ASC - AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b], aggr=[COUNT(c)], group_completion_mode=Partial([0]) + AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b], aggr=[COUNT(c)], group_clustering_mode=Partial([0]) DataSourceExec: file_groups={2 groups: [[x], [y]]}, projection=[a, b, c, d, e], output_ordering=[a@0 ASC], file_type=parquet "); diff --git a/datafusion/core/tests/physical_optimizer/limited_distinct_aggregation.rs b/datafusion/core/tests/physical_optimizer/limited_distinct_aggregation.rs index 6164445b98e2d..833c9485f4cc3 100644 --- a/datafusion/core/tests/physical_optimizer/limited_distinct_aggregation.rs +++ b/datafusion/core/tests/physical_optimizer/limited_distinct_aggregation.rs @@ -520,7 +520,7 @@ fn test_has_order_by() -> Result<()> { actual, @r" LocalLimitExec: fetch=10 - AggregateExec: mode=Single, gby=[a@0 as a], aggr=[], group_completion_mode=Full + AggregateExec: mode=Single, gby=[a@0 as a], aggr=[], group_clustering_mode=Full DataSourceExec: file_groups={1 group: [[x]]}, projection=[a, b, c, d, e], output_ordering=[a@0 ASC], file_type=parquet " ); diff --git a/datafusion/core/tests/physical_optimizer/pushdown_sort.rs b/datafusion/core/tests/physical_optimizer/pushdown_sort.rs index 97cfa05d12785..6cf05ecfd8fc4 100644 --- a/datafusion/core/tests/physical_optimizer/pushdown_sort.rs +++ b/datafusion/core/tests/physical_optimizer/pushdown_sort.rs @@ -731,13 +731,13 @@ fn test_pushdown_through_blocking_node() { OptimizationTest: input: - SortExec: expr=[a@0 ASC], preserve_partitioning=[false] - - AggregateExec: mode=Final, gby=[a@0 as a], aggr=[COUNT(b)], group_completion_mode=Full + - AggregateExec: mode=Final, gby=[a@0 as a], aggr=[COUNT(b)], group_clustering_mode=Full - SortExec: expr=[a@0 DESC NULLS LAST], preserve_partitioning=[false] - DataSourceExec: file_groups={1 group: [[x]]}, projection=[a, b, c, d, e], output_ordering=[a@0 ASC], file_type=parquet output: Ok: - SortExec: expr=[a@0 ASC], preserve_partitioning=[false] - - AggregateExec: mode=Final, gby=[a@0 as a], aggr=[COUNT(b)], group_completion_mode=Full + - AggregateExec: mode=Final, gby=[a@0 as a], aggr=[COUNT(b)], group_clustering_mode=Full - SortExec: expr=[a@0 DESC NULLS LAST], preserve_partitioning=[false] - DataSourceExec: file_groups={1 group: [[x]]}, projection=[a, b, c, d, e], file_type=parquet, sort_order_for_reorder=[a@0 DESC NULLS LAST], reverse_row_groups=true " diff --git a/datafusion/physical-plan/benches/dictionary_group_values.rs b/datafusion/physical-plan/benches/dictionary_group_values.rs index c9ff2cf34b7dc..51363c0bdffe3 100644 --- a/datafusion/physical-plan/benches/dictionary_group_values.rs +++ b/datafusion/physical-plan/benches/dictionary_group_values.rs @@ -29,7 +29,7 @@ use criterion::{ }; use datafusion_expr::EmitTo; use datafusion_physical_plan::aggregates::group_values::new_group_values; -use datafusion_physical_plan::aggregates::order::{GroupCompletion, GroupCompletionFull}; +use datafusion_physical_plan::aggregates::order::{GroupClustering, GroupClusteringFull}; use rand::rngs::StdRng; use rand::seq::SliceRandom; use rand::{Rng, SeedableRng}; @@ -106,7 +106,7 @@ fn bench_intern_emit(c: &mut Criterion) { b.iter_batched_ref( || { ( - new_group_values(schema.clone(), &GroupCompletion::None) + new_group_values(schema.clone(), &GroupClustering::None) .unwrap(), Vec::::with_capacity(size), ) @@ -151,7 +151,7 @@ fn bench_repeated_intern_emit(c: &mut Criterion) { b.iter_batched_ref( || { ( - new_group_values(schema.clone(), &GroupCompletion::None) + new_group_values(schema.clone(), &GroupClustering::None) .unwrap(), Vec::::with_capacity(size), ) @@ -172,7 +172,7 @@ fn bench_repeated_intern_emit(c: &mut Criterion) { group.finish(); } -// GroupCompletion::Full -> GroupValuesColumn::: scalar append_val/equal_to path. +// GroupClustering::Full -> GroupValuesColumn::: scalar append_val/equal_to path. fn bench_scalar_append_equal(c: &mut Criterion) { let mut group = c.benchmark_group("dict_scalar_append_equal"); let schema = dict_schema(); @@ -192,7 +192,7 @@ fn bench_scalar_append_equal(c: &mut Criterion) { ( new_group_values( schema.clone(), - &GroupCompletion::Full(GroupCompletionFull::new()), + &GroupClustering::Full(GroupClusteringFull::new()), ) .unwrap(), Vec::::with_capacity(size), @@ -227,7 +227,7 @@ fn bench_take_n(c: &mut Criterion) { b.iter_batched_ref( || { ( - new_group_values(schema.clone(), &GroupCompletion::None).unwrap(), + new_group_values(schema.clone(), &GroupClustering::None).unwrap(), Vec::::with_capacity(size), ) }, @@ -298,7 +298,7 @@ fn bench_shared_values_arc(c: &mut Criterion) { b.iter_batched_ref( || { ( - new_group_values(schema.clone(), &GroupCompletion::None) + new_group_values(schema.clone(), &GroupClustering::None) .unwrap(), Vec::::with_capacity(size), ) diff --git a/datafusion/physical-plan/benches/ordered_group_values.rs b/datafusion/physical-plan/benches/ordered_group_values.rs index c7eb02b93826e..fc60423c8be92 100644 --- a/datafusion/physical-plan/benches/ordered_group_values.rs +++ b/datafusion/physical-plan/benches/ordered_group_values.rs @@ -38,9 +38,9 @@ use datafusion_physical_expr::expressions::col; use datafusion_physical_expr::{LexOrdering, PhysicalSortExpr}; use datafusion_physical_plan::aggregates::group_values::multi_group_by::GroupValuesColumn; use datafusion_physical_plan::aggregates::group_values::{GroupValues, new_group_values}; -use datafusion_physical_plan::aggregates::order::GroupCompletion; +use datafusion_physical_plan::aggregates::order::GroupClustering; use datafusion_physical_plan::aggregates::{ - AggregateExec, AggregateMode, GroupCompletionMode, PhysicalGroupBy, + AggregateExec, AggregateMode, GroupClusteringMode, PhysicalGroupBy, }; use datafusion_physical_plan::test::TestMemoryExec; use datafusion_physical_plan::{ExecutionPlan, collect}; @@ -112,8 +112,8 @@ fn grouping(c: &mut Criterion) { let values: Box = if selected { new_group_values( Arc::clone(&schema), - &GroupCompletion::try_new( - &GroupCompletionMode::Full, + &GroupClustering::try_new( + &GroupClusteringMode::Full, ) .unwrap(), ) @@ -160,7 +160,7 @@ fn check_case(schema: &SchemaRef, batches: &[Vec]) -> (usize, usize) { let mut hashed = GroupValuesColumn::::try_new(Arc::clone(schema)).unwrap(); let mut selected = new_group_values( Arc::clone(schema), - &GroupCompletion::try_new(&GroupCompletionMode::Full).unwrap(), + &GroupClustering::try_new(&GroupClusteringMode::Full).unwrap(), ) .unwrap(); let mut expected = Vec::new(); @@ -190,7 +190,7 @@ fn aggregate_plan( schema: &SchemaRef, keys: Vec>, sort_columns: &[&str], - group_completion_mode: &GroupCompletionMode, + group_clustering_mode: &GroupClusteringMode, ) -> Arc { let mut fields = schema.fields().to_vec(); fields.push(Arc::new(Field::new("v", DataType::Int64, false))); @@ -234,7 +234,7 @@ fn aggregate_plan( schema, ) .unwrap(); - assert_eq!(plan.group_completion_mode(), group_completion_mode); + assert_eq!(plan.group_clustering_mode(), group_clustering_mode); Arc::new(plan) } @@ -248,7 +248,7 @@ fn aggregation(c: &mut Criterion) { for run_length in [1, 8, 128, 8192] { let (schema, keys) = inputs(run_length, 8192, strings); let plan = - aggregate_plan(&schema, keys, &["a", "b"], &GroupCompletionMode::Full); + aggregate_plan(&schema, keys, &["a", "b"], &GroupClusteringMode::Full); let name = format!("{}_run{run_length}", if strings { "string" } else { "int" }); group.bench_function(name, |b| { @@ -293,7 +293,7 @@ fn partially_ordered_aggregation(c: &mut Criterion) { &schema, keys, &["a"], - &GroupCompletionMode::Partial(vec![0]), + &GroupClusteringMode::Partial(vec![0]), ); let runtime = Runtime::new().unwrap(); let mut group = c.benchmark_group("partially_ordered_aggregate_exec"); diff --git a/datafusion/physical-plan/benches/partial_ordering.rs b/datafusion/physical-plan/benches/partial_ordering.rs index ef90904ca1a76..967e84d78a88b 100644 --- a/datafusion/physical-plan/benches/partial_ordering.rs +++ b/datafusion/physical-plan/benches/partial_ordering.rs @@ -18,7 +18,7 @@ use std::sync::Arc; use arrow::array::{ArrayRef, Int32Array}; -use datafusion_physical_plan::aggregates::order::GroupCompletionPartial; +use datafusion_physical_plan::aggregates::order::GroupClusteringPartial; use criterion::{Criterion, criterion_group, criterion_main}; @@ -34,7 +34,7 @@ fn create_test_arrays(num_columns: usize) -> Vec { .collect() } fn bench_new_groups(c: &mut Criterion) { - let mut group = c.benchmark_group("group_completion_partial"); + let mut group = c.benchmark_group("group_clustering_partial"); // Test with 1, 2, 4, and 8 grouping indices for num_columns in [1, 2, 4, 8] { @@ -45,9 +45,9 @@ fn bench_new_groups(c: &mut Criterion) { let group_indices: Vec = (0..BATCH_SIZE).collect(); b.iter(|| { - let mut completion = - GroupCompletionPartial::try_new(grouping_indices.clone()).unwrap(); - completion + let mut clustering = + GroupClusteringPartial::try_new(grouping_indices.clone()).unwrap(); + clustering .new_groups(&batch_group_values, &group_indices, BATCH_SIZE) .unwrap(); }); diff --git a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_final_table.rs b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_final_table.rs index 0e41489a3ec7b..e2bad901f0760 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_final_table.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_final_table.rs @@ -26,7 +26,7 @@ use arrow::record_batch::RecordBatch; use datafusion_common::Result; use crate::aggregates::aggregate_hash_table::FinalMarker; -use crate::aggregates::order::GroupCompletionMode; +use crate::aggregates::order::GroupClusteringMode; use crate::aggregates::{AggregateExec, AggregateMode, group_values::AccumulatorPhase}; use super::common::HashAggregateAccumulator; @@ -42,11 +42,11 @@ use super::common_clustered::{ClusteredAggregateTable, ClusteredAggregateTableMe /// /// See comments at [`ClusteredAggregateTable`] for details. impl ClusteredAggregateTable { - pub(in crate::aggregates) fn new_with_group_completion( + pub(in crate::aggregates) fn new_with_group_clustering( agg: &AggregateExec, input_schema: &SchemaRef, output_schema: SchemaRef, - group_completion_mode: &GroupCompletionMode, + group_clustering_mode: &GroupClusteringMode, metrics: ClusteredAggregateTableMetrics, ) -> Result { Self::new_for_mode( @@ -54,7 +54,7 @@ impl ClusteredAggregateTable { input_schema, output_schema, Arc::clone(input_schema), - group_completion_mode, + group_clustering_mode, &AggregateMode::Final, vec![None; agg.aggr_expr().len()], metrics, @@ -88,7 +88,7 @@ impl ClusteredAggregateTable { if self.is_empty() { return Ok(None); } - let Some(emit_to) = self.group_completion().emit_to() else { + let Some(emit_to) = self.group_clustering().emit_to() else { return Ok(None); }; self.materialize_groups( diff --git a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_partial_table.rs b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_partial_table.rs index db3b5f91274fa..fef9659fc2165 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_partial_table.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_partial_table.rs @@ -19,7 +19,7 @@ //! //! See the [`super::common_clustered`] comments for the high-level ideas. //! -//! Ordering can establish either group-completion mode: +//! Ordering can establish either group-clustering mode: //! - Full: `GROUP BY a, b`, input is `ORDER BY a, b` //! - Partial: `GROUP BY a, b`, input is `ORDER BY a` //! @@ -66,7 +66,7 @@ impl ClusteredAggregateTable { &input_schema, output_schema, state_schema, - &agg.group_completion_mode, + &agg.group_clustering_mode, &AggregateMode::Partial, agg.filter_expr().to_vec(), metrics, @@ -95,7 +95,7 @@ impl ClusteredAggregateTable { if self.is_empty() { return Ok(None); } - let Some(emit_to) = self.group_completion().emit_to() else { + let Some(emit_to) = self.group_clustering().emit_to() else { return Ok(None); }; self.materialize_groups( diff --git a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_single_table.rs b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_single_table.rs index 108a21a5d063b..c8ecbf4f7f154 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_single_table.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_single_table.rs @@ -57,7 +57,7 @@ impl ClusteredAggregateTable { &input_schema, output_schema, state_schema, - &agg.group_completion_mode, + &agg.group_clustering_mode, &agg.mode, agg.filter_expr().to_vec(), metrics, @@ -88,7 +88,7 @@ impl ClusteredAggregateTable { if self.is_empty() { return Ok(None); } - let Some(emit_to) = self.group_completion().emit_to() else { + let Some(emit_to) = self.group_clustering().emit_to() else { return Ok(None); }; self.materialize_groups( diff --git a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common.rs b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common.rs index 5f33500140c04..a40c5e5fae693 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common.rs @@ -38,7 +38,7 @@ use crate::aggregates::group_values::{ AccumulatorPhase, AggregateAccumulatorMetrics, AggregateArgumentMetrics, GroupByMetrics, GroupValues, new_group_values, }; -use crate::aggregates::order::GroupCompletion; +use crate::aggregates::order::GroupClustering; use crate::aggregates::{ AggregateExec, PhysicalGroupBy, aggregate_expressions, evaluate_group_by, group_id_array, max_duplicate_ordinal, @@ -180,7 +180,7 @@ impl AggregateHashTable { .collect::>()?; let group_schema = agg.group_by().group_schema(&input_schema)?; - let group_values = new_group_values(group_schema, &GroupCompletion::None)?; + let group_values = new_group_values(group_schema, &GroupClustering::None)?; Ok(Self { group_by_metrics: metrics.group_by, diff --git a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common_clustered.rs b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common_clustered.rs index c719c868c9abc..4cb5796a54f4c 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common_clustered.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common_clustered.rs @@ -31,7 +31,7 @@ use crate::aggregates::group_values::{ AccumulatorPhase, AggregateAccumulatorMetrics, AggregateArgumentMetrics, GroupByMetrics, GroupValues, new_group_values, }; -use crate::aggregates::order::{GroupCompletion, GroupCompletionMode}; +use crate::aggregates::order::{GroupClustering, GroupClusteringMode}; use crate::aggregates::{ AggregateExec, AggregateMode, PhysicalGroupBy, aggregate_expressions, evaluate_group_by, @@ -76,9 +76,9 @@ impl ClusteredAggregateTableMetrics { /// Aggregate table shared by the clustered single, partial and final paths. /// -/// # Group completion optimization +/// # Group clustering optimization /// -/// The table consumes input batches while [`GroupCompletion`] tracks which groups +/// The table consumes input batches while [`GroupClustering`] tracks which groups /// are proven complete. Completed groups can be emitted before the input stream /// ends, so completed groups no longer occupy the table. /// @@ -112,7 +112,7 @@ impl ClusteredAggregateTableMetrics { /// `AggrMode` selects the aggregate semantics. For example, /// `ClusteredAggregateTable::::new(...)` consumes raw rows /// and emits partial states, while -/// `ClusteredAggregateTable::::new_with_group_completion(...)` +/// `ClusteredAggregateTable::::new_with_group_clustering(...)` /// consumes partial states and emits final values. /// /// Shared methods live on `impl`; single/partial/final behavior lives on @@ -148,7 +148,7 @@ pub(in crate::aggregates) struct ClusteredAggregateTable { /// It accumulates input during aggregation and emits output rows as soon as /// groups are known to be complete. /// -/// [`GroupCompletion`] tracks when and how to do early emit. +/// [`GroupClustering`] tracks when and how to do early emit. /// [`GroupValues`] stores the physical group-key layout, while /// [`datafusion_expr::GroupsAccumulator`] stores per-group aggregate state. pub(super) struct ClusteredAggregateTableBuffer { @@ -156,7 +156,7 @@ pub(super) struct ClusteredAggregateTableBuffer { pub(super) group_by: Arc, /// Tracks which groups are complete and can be emitted safely. - pub(super) group_completion: GroupCompletion, + pub(super) group_clustering: GroupClustering, /// Interned group keys, in the same group-id order used by accumulators. pub(super) group_values: Box, @@ -182,14 +182,14 @@ impl ClusteredAggregateTable { input_schema: &SchemaRef, output_schema: SchemaRef, state_schema: SchemaRef, - group_completion_mode: &GroupCompletionMode, + group_clustering_mode: &GroupClusteringMode, aggregate_mode: &AggregateMode, filters: Vec>>, metrics: ClusteredAggregateTableMetrics, ) -> Result { - let group_completion = GroupCompletion::try_new(group_completion_mode)?; + let group_clustering = GroupClustering::try_new(group_clustering_mode)?; let group_schema = agg.group_by().group_schema(input_schema)?; - let group_values = new_group_values(group_schema, &group_completion)?; + let group_values = new_group_values(group_schema, &group_clustering)?; let aggregate_arguments = aggregate_expressions( agg.aggr_expr(), aggregate_mode, @@ -223,7 +223,7 @@ impl ClusteredAggregateTable { aggregate_submetrics: metrics.submetrics, buffer: ClusteredAggregateTableBuffer { group_by: Arc::clone(agg.group_by()), - group_completion, + group_clustering, group_values, group_indices: vec![], accumulators, @@ -266,15 +266,15 @@ impl ClusteredAggregateTable { /// Called after the input stream is exhausted and the last batch has been /// aggregated. /// - /// Updates the internal [`GroupCompletion`] so it can continue emitting until + /// Updates the internal [`GroupClustering`] so it can continue emitting until /// the buffer is empty. pub(in crate::aggregates) fn input_done(&mut self) { - self.buffer.group_completion.input_done(); + self.buffer.group_clustering.input_done(); } /// Returns the completion state used to decide how memory pressure is handled. - pub(in crate::aggregates) fn group_completion(&self) -> &GroupCompletion { - &self.buffer.group_completion + pub(in crate::aggregates) fn group_clustering(&self) -> &GroupClustering { + &self.buffer.group_clustering } /// Number of groups currently buffered. @@ -295,7 +295,7 @@ impl ClusteredAggregateTable { .map(|acc| acc.size()) .sum::() + self.buffer.group_values.size() - + self.buffer.group_completion.size() + + self.buffer.group_clustering.size() + self.buffer.group_indices.allocated_size() } @@ -344,7 +344,7 @@ impl ClusteredAggregateTable { self.buffer.group_values.clear_shrink(0); self.buffer.group_indices.clear(); self.buffer.group_indices.shrink_to_fit(); - self.buffer.group_completion.reset(); + self.buffer.group_clustering.reset(); Ok(Some(batch)) } @@ -372,7 +372,7 @@ impl ClusteredAggregateTable { .intern(group_values, &mut self.buffer.group_indices)?; let total_num_groups = self.buffer.group_values.len(); if total_num_groups > starting_num_groups { - self.buffer.group_completion.new_groups( + self.buffer.group_clustering.new_groups( group_values, &self.buffer.group_indices, total_num_groups, @@ -419,7 +419,7 @@ impl ClusteredAggregateTable { // `EmitTo::All` is only used after `input_done`, when the completion // state no longer tracks group indexes. if let EmitTo::First(n) = emit_to { - self.buffer.group_completion.remove_groups(n); + self.buffer.group_clustering.remove_groups(n); } for (idx, acc) in self.buffer.accumulators.iter_mut().enumerate() { diff --git a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/partial_table.rs b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/partial_table.rs index b61b954e4d527..9c95bec83e3c8 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/partial_table.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/partial_table.rs @@ -23,7 +23,7 @@ use arrow::record_batch::RecordBatch; use datafusion_common::{Result, assert_eq_or_internal_err}; use crate::aggregates::group_values::{AccumulatorPhase, new_group_values}; -use crate::aggregates::order::GroupCompletion; +use crate::aggregates::order::GroupClustering; use crate::aggregates::{AggregateExec, evaluate_group_by}; use super::common::{ @@ -78,7 +78,7 @@ impl AggregateHashTable { ) -> Result> { let state = self.state.building(); let group_schema = state.group_by.group_schema(&self.input_schema)?; - let group_values = new_group_values(group_schema, &GroupCompletion::None)?; + let group_values = new_group_values(group_schema, &GroupClustering::None)?; let accumulators = state .accumulators .iter() diff --git a/datafusion/physical-plan/src/aggregates/clustered_final_stream.rs b/datafusion/physical-plan/src/aggregates/clustered_final_stream.rs index 33c7a87791a1c..9c71c5ee13328 100644 --- a/datafusion/physical-plan/src/aggregates/clustered_final_stream.rs +++ b/datafusion/physical-plan/src/aggregates/clustered_final_stream.rs @@ -15,7 +15,7 @@ // specific language governing permissions and limitations // under the License. -//! Final aggregate stream for partial-state input with group-completion +//! Final aggregate stream for partial-state input with group-clustering //! guarantees. use std::sync::Arc; @@ -31,15 +31,15 @@ use super::AggregateExec; use super::aggregate_hash_table::{ ClusteredAggregateTable, ClusteredAggregateTableMetrics, FinalMarker, }; -use super::order::GroupCompletionMode; +use super::order::GroupClusteringMode; use super::spill::AggregateSpill; use crate::SendableRecordBatchStream; use crate::aggregates::AggregateMode; use crate::metrics::{BaselineMetrics, SpillMetrics}; use crate::stream::{ObservedStream, RecordBatchStreamAdapter}; -/// Final aggregate stream for [`GroupCompletionMode::Partial`] and -/// [`GroupCompletionMode::Full`]. +/// Final aggregate stream for [`GroupClusteringMode::Partial`] and +/// [`GroupClusteringMode::Full`]. /// /// See comments at [`super::clustered_partial_stream::ClusteredPartialAggregateStream`] for details. /// @@ -47,7 +47,7 @@ use crate::stream::{ObservedStream, RecordBatchStreamAdapter}; /// /// This section is only for implementation notes, for background, see [`super::clustered_partial_stream::ClusteredPartialAggregateStream`] /// -/// For partial group completion, spilling works as follows: +/// For partial group clustering, spilling works as follows: /// /// - Reserve the table footprint plus one `u32` sort index per buffered group. The /// extra index array is used in later sorting before spilling. @@ -73,7 +73,7 @@ enum ExecutionStage { struct Aggregating { input: SendableRecordBatchStream, table: ClusteredAggregateTable, - /// None when temporary files are disabled or group completion is full. + /// None when temporary files are disabled or group clustering is full. spill_context: Option>, } @@ -101,10 +101,10 @@ impl ClusteredFinalAggregateStream { agg.mode, AggregateMode::Final | AggregateMode::FinalPartitioned )); - debug_assert_ne!(agg.group_completion_mode, GroupCompletionMode::None); + debug_assert_ne!(agg.group_clustering_mode, GroupClusteringMode::None); let input = agg.input.execute(partition, Arc::clone(context))?; - Self::new_with_input(agg, context, partition, input, &agg.group_completion_mode) + Self::new_with_input(agg, context, partition, input, &agg.group_clustering_mode) } pub(in crate::aggregates) fn new_with_input( @@ -112,14 +112,14 @@ impl ClusteredFinalAggregateStream { context: &Arc, partition: usize, input: SendableRecordBatchStream, - group_completion_mode: &GroupCompletionMode, + group_clustering_mode: &GroupClusteringMode, ) -> Result { let baseline_metrics = BaselineMetrics::new(&agg.metrics, partition); let metrics = ClusteredAggregateTableMetrics::new(agg, partition); let spill_metrics = SpillMetrics::new(&agg.metrics, partition); let reservation = MemoryConsumer::new(format!("ClusteredFinalAggregateStream[{partition}]")) - // HACK: Full group completion uses a non-spilling execution + // HACK: Full group clustering uses a non-spilling execution // path. There is a known race // condition bug, and we set it to spillable to let it have larger // memory budget to suppress the bug. @@ -131,7 +131,7 @@ impl ClusteredFinalAggregateStream { context, partition, input, - group_completion_mode, + group_clustering_mode, baseline_metrics, metrics, Some(spill_metrics), @@ -151,7 +151,7 @@ impl ClusteredFinalAggregateStream { context: &Arc, partition: usize, input: SendableRecordBatchStream, - group_completion_mode: &GroupCompletionMode, + group_clustering_mode: &GroupClusteringMode, baseline_metrics: BaselineMetrics, metrics: ClusteredAggregateTableMetrics, spill_metrics: Option, @@ -161,13 +161,13 @@ impl ClusteredFinalAggregateStream { agg.mode, AggregateMode::Final | AggregateMode::FinalPartitioned )); - debug_assert_ne!(*group_completion_mode, GroupCompletionMode::None); + debug_assert_ne!(*group_clustering_mode, GroupClusteringMode::None); let schema = Arc::clone(&agg.schema); let input_schema = input.schema(); let batch_size = context.session_config().batch_size(); - let can_spill = matches!(group_completion_mode, GroupCompletionMode::Partial(_)) + let can_spill = matches!(group_clustering_mode, GroupClusteringMode::Partial(_)) && context.runtime_env().disk_manager.tmp_files_enabled(); let spill_context = if can_spill { let Some(spill_metrics) = spill_metrics else { @@ -181,7 +181,7 @@ impl ClusteredFinalAggregateStream { context, partition, batch_size, - group_completion_mode, + group_clustering_mode, &input_schema, spill_metrics, )?)) @@ -189,11 +189,11 @@ impl ClusteredFinalAggregateStream { None }; - let table = ClusteredAggregateTable::::new_with_group_completion( + let table = ClusteredAggregateTable::::new_with_group_clustering( agg, &input_schema, Arc::clone(&schema), - group_completion_mode, + group_clustering_mode, metrics, )?; @@ -458,7 +458,7 @@ impl Outputting { mod tests { use super::*; use crate::ExecutionPlan; - use crate::aggregates::GroupCompletionMode; + use crate::aggregates::GroupClusteringMode; use crate::aggregates::PhysicalGroupBy; use crate::common::collect; use crate::test::TestMemoryExec; @@ -561,8 +561,8 @@ mod tests { schema, )?; assert_eq!( - aggregate.group_completion_mode(), - &GroupCompletionMode::Partial(vec![0]) + aggregate.group_clustering_mode(), + &GroupClusteringMode::Partial(vec![0]) ); let pool: Arc = Arc::new(GreedyMemoryPool::new(limit)); @@ -592,7 +592,7 @@ mod tests { &context, partition, input, - &aggregate.group_completion_mode, + &aggregate.group_clustering_mode, )?; Ok((sender, stream.into_stream())) }; diff --git a/datafusion/physical-plan/src/aggregates/clustered_partial_stream.rs b/datafusion/physical-plan/src/aggregates/clustered_partial_stream.rs index 577ae86d18dba..f7f7e2d9031dd 100644 --- a/datafusion/physical-plan/src/aggregates/clustered_partial_stream.rs +++ b/datafusion/physical-plan/src/aggregates/clustered_partial_stream.rs @@ -15,7 +15,7 @@ // specific language governing permissions and limitations // under the License. -//! Partial aggregate stream for input with group-completion guarantees. +//! Partial aggregate stream for input with group-clustering guarantees. use std::sync::Arc; @@ -29,13 +29,13 @@ use futures::stream::StreamExt; use super::AggregateExec; use super::aggregate_hash_table::{ClusteredAggregateTable, PartialMarker}; use crate::aggregates::AggregateMode; -use crate::aggregates::order::{GroupCompletion, GroupCompletionMode}; +use crate::aggregates::order::{GroupClustering, GroupClusteringMode}; use crate::metrics::{BaselineMetrics, MetricBuilder, SpillMetrics}; use crate::stream::{ObservedStream, RecordBatchStreamAdapter}; use crate::{SendableRecordBatchStream, metrics}; -/// Partial aggregate stream for [`GroupCompletionMode::Partial`] and -/// [`GroupCompletionMode::Full`]. +/// Partial aggregate stream for [`GroupClusteringMode::Partial`] and +/// [`GroupClusteringMode::Full`]. /// /// # Example /// @@ -59,11 +59,11 @@ use crate::{SendableRecordBatchStream, metrics}; /// Output: results for all groups (for example, `AVG(x)` calculated from the /// state) /// -/// # Group Completion Optimization +/// # Group Clustering Optimization /// /// For the aggregation work, the hash aggregation implementation is reused. /// -/// After each input batch, the group-completion mode determines whether any +/// After each input batch, the group-clustering mode determines whether any /// groups can be emitted eagerly to improve memory efficiency. For example, if /// the input is ordered by `k` and the last group key seen is `k = 100`, all /// groups with keys less than 100 are complete. Materialize that entire completed @@ -73,7 +73,7 @@ use crate::{SendableRecordBatchStream, metrics}; /// /// # Memory Pressure and Spilling /// -/// ## Full group completion +/// ## Full group clustering /// /// Every complete grouping tuple is contiguous. Ordering by every group key is /// one way to establish this mode, for example: @@ -88,7 +88,7 @@ use crate::{SendableRecordBatchStream, metrics}; /// If a memory reservation nevertheless fails, the stream returns the error /// directly, indicating an unexpected behavior. /// -/// ## Partial group completion +/// ## Partial group clustering /// /// Rows are contiguous for a subset of the group keys. Ordering by that subset /// is one way to establish this mode, for example: @@ -147,7 +147,7 @@ impl ClusteredPartialAggregateStream { partition: usize, ) -> Result { debug_assert_eq!(agg.mode, AggregateMode::Partial); - debug_assert_ne!(agg.group_completion_mode, GroupCompletionMode::None); + debug_assert_ne!(agg.group_clustering_mode, GroupClusteringMode::None); let schema = Arc::clone(&agg.schema); let input = agg.input.execute(partition, Arc::clone(context))?; @@ -168,8 +168,8 @@ impl ClusteredPartialAggregateStream { let reservation = MemoryConsumer::new(format!("ClusteredPartialAggregateStream[{partition}]")) .with_can_spill(matches!( - table.group_completion(), - GroupCompletion::Partial(_) + table.group_clustering(), + GroupClustering::Partial(_) )) .register(context.memory_pool()); @@ -261,7 +261,7 @@ impl ClusteredPartialAggregateStream { /// 3. Prepare output: /// - A prefix of groups is complete: materialize the entire prefix once, /// retaining the input and active groups to resume aggregation. - /// - On memory pressure with partial group completion, materialize all + /// - On memory pressure with partial group clustering, materialize all /// current states, including incomplete groups, and reset the table. /// - At EOF, materialize all remaining states and prepare to output. /// 4. Input was exhausted with no remaining groups, directly end. @@ -318,9 +318,9 @@ impl Aggregating { let output = match reservation.try_resize(self.table.memory_size()) { Ok(()) => self.table.take_completed_state_batch()?, Err(oom @ DataFusionError::ResourcesExhausted(_)) => { - // Partial group completion may have an unbounded active key range. + // Partial group clustering may have an unbounded active key range. // The final stage can merge incomplete states emitted here. - if matches!(self.table.group_completion(), GroupCompletion::Full(_)) { + if matches!(self.table.group_clustering(), GroupClustering::Full(_)) { return Err(oom); } let Some(batch) = self.table.take_state_batch()? else { diff --git a/datafusion/physical-plan/src/aggregates/clustered_single_stream.rs b/datafusion/physical-plan/src/aggregates/clustered_single_stream.rs index e41c89277645d..7095921ac78fc 100644 --- a/datafusion/physical-plan/src/aggregates/clustered_single_stream.rs +++ b/datafusion/physical-plan/src/aggregates/clustered_single_stream.rs @@ -15,7 +15,7 @@ // specific language governing permissions and limitations // under the License. -//! Single-stage aggregate stream for raw input with group-completion guarantees. +//! Single-stage aggregate stream for raw input with group-clustering guarantees. use std::ops::ControlFlow; use std::sync::Arc; @@ -29,7 +29,7 @@ use datafusion_execution::memory_pool::{MemoryConsumer, MemoryReservation}; use futures::stream::{Stream, StreamExt}; use super::aggregate_hash_table::{ClusteredAggregateTable, SingleMarker}; -use super::order::GroupCompletionMode; +use super::order::GroupClusteringMode; use super::spill::AggregateSpill; use super::{AggregateExec, create_schema}; use crate::aggregates::AggregateMode; @@ -37,8 +37,8 @@ use crate::metrics::{BaselineMetrics, RecordOutput, SpillMetrics}; use crate::stream::EmptyRecordBatchStream; use crate::{RecordBatchStream, SendableRecordBatchStream}; -/// Single aggregate stream for [`GroupCompletionMode::Partial`] and -/// [`GroupCompletionMode::Full`]. +/// Single aggregate stream for [`GroupClusteringMode::Partial`] and +/// [`GroupClusteringMode::Full`]. /// /// # Example /// @@ -55,11 +55,11 @@ use crate::{RecordBatchStream, SendableRecordBatchStream}; /// Input: raw rows /// Output: final results for all groups (for example, `AVG(x)`) /// -/// # Group Completion Optimization +/// # Group Clustering Optimization /// /// For the aggregation work, the hash aggregation implementation is reused. /// -/// After each input batch, the group-completion mode determines whether any +/// After each input batch, the group-clustering mode determines whether any /// groups can be emitted eagerly to improve memory efficiency. Materialize /// that entire completed prefix once, then emit slices of it before reading /// more input. See @@ -70,7 +70,7 @@ use crate::{RecordBatchStream, SendableRecordBatchStream}; /// /// # Memory Pressure and Spilling /// -/// ## Full group completion +/// ## Full group clustering /// /// Every complete grouping tuple is contiguous. Ordering by every group key is /// one way to establish this mode, for example: @@ -85,7 +85,7 @@ use crate::{RecordBatchStream, SendableRecordBatchStream}; /// If a memory reservation nevertheless fails, the stream returns the error /// directly, indicating an unexpected behavior. /// -/// ## Partial group completion +/// ## Partial group clustering /// /// Rows are contiguous for a subset of the group keys. Ordering by that subset /// is one way to establish this mode, for example: @@ -115,7 +115,7 @@ enum ClusteredSingleAggregateState { table: ClusteredAggregateTable, /// None if either /// - Disk Manager doesn't enable temporary file creation - /// - Full group completion is used, so completed groups can be released + /// - Full group clustering is used, so completed groups can be released spill_context: Option>, }, Spilling { @@ -160,7 +160,7 @@ impl ClusteredSingleAggregateStream { agg.mode, AggregateMode::Single | AggregateMode::SinglePartitioned )); - debug_assert_ne!(agg.group_completion_mode, GroupCompletionMode::None); + debug_assert_ne!(agg.group_clustering_mode, GroupClusteringMode::None); let schema = Arc::clone(&agg.schema); let input = agg.input.execute(partition, Arc::clone(context))?; @@ -183,7 +183,7 @@ impl ClusteredSingleAggregateStream { )?; let can_spill = - matches!(agg.group_completion_mode, GroupCompletionMode::Partial(_)) + matches!(agg.group_clustering_mode, GroupClusteringMode::Partial(_)) && context.runtime_env().disk_manager.tmp_files_enabled(); let spill_context = if can_spill { Some(Box::new(AggregateSpill::try_new( @@ -192,7 +192,7 @@ impl ClusteredSingleAggregateStream { context, partition, batch_size, - &agg.group_completion_mode, + &agg.group_clustering_mode, &state_schema, spill_metrics, )?)) diff --git a/datafusion/physical-plan/src/aggregates/group_values/mod.rs b/datafusion/physical-plan/src/aggregates/group_values/mod.rs index 8843fe54afea8..93e2efbfbd648 100644 --- a/datafusion/physical-plan/src/aggregates/group_values/mod.rs +++ b/datafusion/physical-plan/src/aggregates/group_values/mod.rs @@ -38,7 +38,7 @@ use crate::aggregates::{ boolean::GroupValuesBoolean, bytes::GroupValuesBytes, bytes_view::GroupValuesBytesView, primitive::GroupValuesPrimitive, }, - order::GroupCompletion, + order::GroupClustering, }; mod metrics; @@ -156,9 +156,9 @@ pub trait GroupValues: Send { /// `GroupValuesRows`: crate::aggregates::group_values::GroupValuesRows pub fn new_group_values( schema: SchemaRef, - group_completion: &GroupCompletion, + group_clustering: &GroupClustering, ) -> Result> { - if matches!(group_completion, GroupCompletion::Full(_)) + if matches!(group_clustering, GroupClustering::Full(_)) && GroupValuesClustered::supports_schema(&schema) { return Ok(Box::new(GroupValuesClustered::try_new(schema)?)); @@ -200,7 +200,7 @@ pub fn new_group_values( } if multi_group_by::supported_schema(schema.as_ref()) { - if matches!(group_completion, GroupCompletion::None) { + if matches!(group_clustering, GroupClustering::None) { Ok(Box::new(GroupValuesColumn::::try_new(schema)?)) } else { Ok(Box::new(GroupValuesColumn::::try_new(schema)?)) @@ -221,7 +221,7 @@ mod tests { use datafusion_expr::{EmitTo, GroupSelection}; use super::new_group_values; - use crate::aggregates::order::GroupCompletion; + use crate::aggregates::order::GroupClustering; #[test] fn preserving_values_keep_group_indices_valid() { @@ -230,7 +230,7 @@ mod tests { DataType::Int32, true, )])); - let mut group_values = new_group_values(schema, &GroupCompletion::None).unwrap(); + let mut group_values = new_group_values(schema, &GroupClustering::None).unwrap(); assert!(group_values.supports_values_preserving()); let input = Arc::new(Int32Array::from(vec![ @@ -287,7 +287,7 @@ mod tests { Field::new("primitive", DataType::Int32, false), Field::new("boolean", DataType::Boolean, false), ])); - let mut group_values = new_group_values(schema, &GroupCompletion::None).unwrap(); + let mut group_values = new_group_values(schema, &GroupClustering::None).unwrap(); let input = vec![ Arc::new(Int32Array::from(vec![10, 20, 10])) as ArrayRef, Arc::new(BooleanArray::from(vec![true, false, true])) as ArrayRef, @@ -318,7 +318,7 @@ mod tests { true, )])); let mut group_values = - new_group_values(schema, &GroupCompletion::None).unwrap(); + new_group_values(schema, &GroupClustering::None).unwrap(); let input: ArrayRef = match data_type { DataType::Utf8 => Arc::new(StringArray::from(vec![ Some("a"), diff --git a/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/clustered.rs b/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/clustered.rs index 233bdef086594..1fcd1fa0bff03 100644 --- a/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/clustered.rs +++ b/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/clustered.rs @@ -352,7 +352,7 @@ mod tests { #[tokio::test] async fn ordered_single_and_partial_final_match_unordered_execution() -> Result<()> { use crate::aggregates::{ - AggregateExec, AggregateMode, GroupCompletionMode, PhysicalGroupBy, + AggregateExec, AggregateMode, GroupClusteringMode, PhysicalGroupBy, }; use crate::test::TestMemoryExec; use crate::{ExecutionPlan, collect}; @@ -453,8 +453,8 @@ mod tests { )?; if sorted { assert_eq!( - plan.group_completion_mode(), - &GroupCompletionMode::Full + plan.group_clustering_mode(), + &GroupClusteringMode::Full ); } let plan: Arc = if two_stage { @@ -468,8 +468,8 @@ mod tests { )?; if sorted { assert_eq!( - plan.group_completion_mode(), - &GroupCompletionMode::Full + plan.group_clustering_mode(), + &GroupClusteringMode::Full ); } Arc::new(plan) diff --git a/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/list.rs b/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/list.rs index 96cd4dde16ba4..a59be76298a32 100644 --- a/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/list.rs +++ b/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/list.rs @@ -445,7 +445,7 @@ mod tests { // List/Struct together correctly and that intern/emit round-trips // through `GroupValuesColumn` rather than `GroupValuesRows`. use crate::aggregates::group_values::new_group_values; - use crate::aggregates::order::GroupCompletion; + use crate::aggregates::order::GroupClustering; use arrow::array::{ Int32Array, LargeListArray, StringArray, StructArray, builder::Int32Builder, builder::LargeListBuilder, builder::StringBuilder, builder::StructBuilder, @@ -507,7 +507,7 @@ mod tests { Arc::new(list_builder.finish()) }; - let mut gv = new_group_values(schema, &GroupCompletion::None).unwrap(); + let mut gv = new_group_values(schema, &GroupClustering::None).unwrap(); // Batch 1: a mix of duplicate / distinct / null lists. let batch1 = notes_v(&[ @@ -584,7 +584,7 @@ mod tests { #[test] fn list_dispatcher_round_trip_through_new_group_values() { use crate::aggregates::group_values::new_group_values; - use crate::aggregates::order::GroupCompletion; + use crate::aggregates::order::GroupClustering; use arrow::datatypes::Schema; use datafusion_expr::EmitTo; @@ -593,7 +593,7 @@ mod tests { DataType::List(child_field()), true, )])); - let mut gv = new_group_values(schema, &GroupCompletion::None).unwrap(); + let mut gv = new_group_values(schema, &GroupClustering::None).unwrap(); // Batch 1. let batch1: ArrayRef = list_array(&[ @@ -869,7 +869,7 @@ mod tests { // ints. Exercises a recursive child GroupColumn built via the // dispatcher. use crate::aggregates::group_values::new_group_values; - use crate::aggregates::order::GroupCompletion; + use crate::aggregates::order::GroupClustering; use arrow::array::{ListArray, builder::Int32Builder, builder::ListBuilder}; use arrow::datatypes::Schema; use datafusion_expr::EmitTo; @@ -917,7 +917,7 @@ mod tests { Arc::new(outer.finish()) }; - let mut gv = new_group_values(schema, &GroupCompletion::None).unwrap(); + let mut gv = new_group_values(schema, &GroupClustering::None).unwrap(); // Three groups: [[1,2],[3]], its duplicate, a distinct value, and a null. let batch = mk(&[ diff --git a/datafusion/physical-plan/src/aggregates/grouped_hash_stream.rs b/datafusion/physical-plan/src/aggregates/grouped_hash_stream.rs index a874c523ec9a2..15166e4278d60 100644 --- a/datafusion/physical-plan/src/aggregates/grouped_hash_stream.rs +++ b/datafusion/physical-plan/src/aggregates/grouped_hash_stream.rs @@ -35,14 +35,14 @@ use std::task::{Context, Poll}; use std::vec; use super::aggregate_hash_table::{accumulator_phases, create_group_accumulator}; -use super::order::GroupCompletion; +use super::order::GroupClustering; use super::skip_partial::SkipAggregationProbe; use super::{AggregateExec, format_human_display}; use crate::aggregates::group_values::{ AccumulatorPhase, AggregateAccumulatorMetrics, AggregateArgumentMetrics, GroupByMetrics, GroupValues, aggregate_sub_metrics, new_group_values, }; -use crate::aggregates::order::GroupCompletionFull; +use crate::aggregates::order::GroupClusteringFull; use crate::aggregates::{ AggregateInputMode, AggregateMode, AggregateOutputMode, PhysicalGroupBy, aggregate_metric_label, create_schema, evaluate_group_by, evaluate_optional, @@ -354,7 +354,7 @@ pub(crate) struct GroupedHashAggregateStream { // Inner states groups together properties, states for a specific task. // ======================================================================== /// Tracks groups that can be emitted from the hash table before the input ends. - group_completion: GroupCompletion, + group_clustering: GroupClustering, /// The spill state object spill_state: SpillState, @@ -526,19 +526,19 @@ impl GroupedHashAggregateStream { .collect::>() .join(", "); let name = format!("GroupedHashAggregateStream[{partition}] ({agg_fn_names})"); - let group_completion = GroupCompletion::try_new(&agg.group_completion_mode)?; - let oom_mode = match (agg.mode, &group_completion) { + let group_clustering = GroupClustering::try_new(&agg.group_clustering_mode)?; + let oom_mode = match (agg.mode, &group_clustering) { // In partial aggregation mode, always prefer to emit incomplete results early. (AggregateMode::Partial, _) => OutOfMemoryMode::EmitEarly, // For non-partial aggregation modes, emitting incomplete results is not an option. // Instead, use disk spilling to store sorted, incomplete results, and merge them // afterwards. - (_, GroupCompletion::None | GroupCompletion::Partial(_)) + (_, GroupClustering::None | GroupClustering::Partial(_)) if context.runtime_env().disk_manager.tmp_files_enabled() => { OutOfMemoryMode::Spill } - // For `GroupCompletion::Full`, each group is contiguous in the input. This keeps + // For `GroupClustering::Full`, each group is contiguous in the input. This keeps // the number of incomplete groups small at all times. If we still hit // an out-of-memory condition, spilling to disk would not be beneficial since the same // situation is likely to reoccur when reading back the spilled data. @@ -548,7 +548,7 @@ impl GroupedHashAggregateStream { _ => OutOfMemoryMode::ReportError, }; - let group_values = new_group_values(group_schema, &group_completion)?; + let group_values = new_group_values(group_schema, &group_clustering)?; let reservation = MemoryConsumer::new(name) // We interpret 'can spill' as 'can handle memory back pressure'. // This value needs to be set to true for the default memory pool implementations @@ -584,7 +584,7 @@ impl GroupedHashAggregateStream { // since Final mode expects unique group values as its input // - there is only one GROUP BY expressions set let skip_aggregation_probe = if agg.mode == AggregateMode::Partial - && matches!(group_completion, GroupCompletion::None) + && matches!(group_clustering, GroupClustering::None) && agg_group_by.is_single() { let options = &context.session_config().options().execution; @@ -639,7 +639,7 @@ impl GroupedHashAggregateStream { aggregate_argument_metrics, aggregate_accumulator_metrics, batch_size, - group_completion, + group_clustering, input_done: false, spill_state, group_values_soft_limit: agg.limit_options().map(|config| config.limit()), @@ -696,7 +696,7 @@ impl Stream for GroupedHashAggregateStream { // this might lead to incorrect output ordering if (self.spill_state.spills.is_empty() || self.spill_state.is_stream_merging) - && let Some(to_emit) = self.group_completion.emit_to() + && let Some(to_emit) = self.group_clustering.emit_to() { timer.done(); if let Some(batch) = self.emit(to_emit, false)? { @@ -906,7 +906,7 @@ impl GroupedHashAggregateStream { let group_indices = &self.current_group_indices; let total_num_groups = self.group_values.len(); if total_num_groups > starting_num_groups { - self.group_completion.new_groups( + self.group_clustering.new_groups( group_values, group_indices, total_num_groups, @@ -995,7 +995,7 @@ impl GroupedHashAggregateStream { self.group_values.len() }; - if let Some(emit_to) = self.group_completion.oom_emit_to(n) + if let Some(emit_to) = self.group_clustering.oom_emit_to(n) && let Some(batch) = self.emit(emit_to, false)? { return Ok(Some(ExecutionState::ProducingOutput(batch))); @@ -1012,7 +1012,7 @@ impl GroupedHashAggregateStream { let acc = self.accumulators.iter().map(|x| x.size()).sum::(); let groups_and_acc_size = acc + self.group_values.size() - + self.group_completion.size() + + self.group_clustering.size() + self.current_group_indices.allocated_size(); // Reserve extra headroom for sorting during potential spill. @@ -1058,7 +1058,7 @@ impl GroupedHashAggregateStream { let output = group_by_metrics.time_emitting(|| { let mut output = self.group_values.emit(emit_to)?; if let EmitTo::First(n) = emit_to { - self.group_completion.remove_groups(n); + self.group_clustering.remove_groups(n); } // Next output each aggregate value. @@ -1138,7 +1138,7 @@ impl GroupedHashAggregateStream { .intern(&cols, &mut self.current_group_indices)?; let total_groups = self.group_values.len(); if total_groups > starting_groups { - self.group_completion.new_groups( + self.group_clustering.new_groups( &cols, &self.current_group_indices, total_groups, @@ -1312,7 +1312,7 @@ impl GroupedHashAggregateStream { /// in case of disk spilling, the SPM stream have been drained. fn set_input_done_and_produce_output(&mut self) -> Result<()> { self.input_done = true; - self.group_completion.input_done(); + self.group_clustering.input_done(); // Release the original input pipeline's resources now that we're done // reading from it. In the spill branch below, `self.input` is replaced // again with a stream that merges spill files. @@ -1357,12 +1357,12 @@ impl GroupedHashAggregateStream { // Reset the group values collectors. self.clear_all(); - // We can now use `GroupCompletion::Full` since the spill files are sorted + // We can now use `GroupClustering::Full` since the spill files are sorted // on the grouping columns. - self.group_completion = GroupCompletion::Full(GroupCompletionFull::new()); + self.group_clustering = GroupClustering::Full(GroupClusteringFull::new()); // Recreate `group_values` for streaming merge so group ids are assigned - // in first-seen order, as required by `GroupCompletionFull`. + // in first-seen order, as required by `GroupClusteringFull`. // The pre-spill collector may use `vectorized_intern`, which can assign // new group ids out of input order under hash collisions. That is the // multi-column collector, which also serves a single group column @@ -1372,7 +1372,7 @@ impl GroupedHashAggregateStream { .spill_state .merging_group_by .group_schema(&self.spill_state.spill_schema)?; - self.group_values = new_group_values(group_schema, &self.group_completion)?; + self.group_values = new_group_values(group_schema, &self.group_clustering)?; // Use `OutOfMemoryMode::ReportError` from this point on // to ensure we don't spill the spilled data to disk again. @@ -1481,7 +1481,7 @@ impl GroupedHashAggregateStream { mod tests { use super::*; use crate::ExecutionPlan; - use crate::aggregates::GroupCompletionMode; + use crate::aggregates::GroupClusteringMode; use crate::test::TestMemoryExec; use arrow::array::{Int32Array, Int64Array, UInt32Array}; use arrow::datatypes::{DataType, Field, Schema}; @@ -1836,8 +1836,8 @@ mod tests { Arc::clone(&schema), )?; assert!(matches!( - aggregate_exec.group_completion_mode(), - GroupCompletionMode::Partial(_) + aggregate_exec.group_clustering_mode(), + GroupClusteringMode::Partial(_) )); // Must not panic with "assertion failed: *current_run_start >= n" diff --git a/datafusion/physical-plan/src/aggregates/hash_stream.rs b/datafusion/physical-plan/src/aggregates/hash_stream.rs index 93992041f1d8d..0b7612d14556f 100644 --- a/datafusion/physical-plan/src/aggregates/hash_stream.rs +++ b/datafusion/physical-plan/src/aggregates/hash_stream.rs @@ -37,7 +37,7 @@ use super::aggregate_hash_table::{ AggregateHashTable, ClusteredAggregateTableMetrics, FinalMarker, PartialMarker, PartialSkipMarker, }; -use super::order::GroupCompletionMode; +use super::order::GroupClusteringMode; use super::skip_partial::SkipAggregationProbe; use super::spill::AggregateSpill; use crate::metrics::{ @@ -218,7 +218,7 @@ impl PartialHashAggregateStream { partition: usize, ) -> Result { debug_assert_eq!(agg.mode, super::AggregateMode::Partial); - debug_assert_eq!(agg.group_completion_mode, GroupCompletionMode::None); + debug_assert_eq!(agg.group_clustering_mode, GroupClusteringMode::None); let schema = Arc::clone(&agg.schema); let input = agg.input.execute(partition, Arc::clone(context))?; @@ -574,7 +574,7 @@ impl FinalHashAggregateStream { agg.mode, super::AggregateMode::Final | super::AggregateMode::FinalPartitioned )); - debug_assert_eq!(agg.group_completion_mode, GroupCompletionMode::None); + debug_assert_eq!(agg.group_clustering_mode, GroupClusteringMode::None); let input = agg.input.execute(partition, Arc::clone(context))?; Self::new_with_input(agg, context, partition, input) @@ -608,7 +608,7 @@ impl FinalHashAggregateStream { context, partition, batch_size, - &GroupCompletionMode::None, + &GroupClusteringMode::None, &input_schema, spill_metrics, )?)) @@ -880,7 +880,7 @@ mod tests { use std::time::Duration; use super::*; - use crate::aggregates::GroupCompletionMode; + use crate::aggregates::GroupClusteringMode; use crate::aggregates::{AggregateMode, PhysicalGroupBy}; use crate::common::collect; use crate::execution_plan::ExecutionPlan; @@ -1570,8 +1570,8 @@ mod tests { schema, )?; assert_eq!( - aggregate.group_completion_mode(), - &GroupCompletionMode::None + aggregate.group_clustering_mode(), + &GroupClusteringMode::None ); let pool: Arc = Arc::new(GreedyMemoryPool::new(limit)); diff --git a/datafusion/physical-plan/src/aggregates/mod.rs b/datafusion/physical-plan/src/aggregates/mod.rs index 101be7f8f4929..2498c91c30663 100644 --- a/datafusion/physical-plan/src/aggregates/mod.rs +++ b/datafusion/physical-plan/src/aggregates/mod.rs @@ -48,7 +48,7 @@ //! //! See [`PartialHashAggregateStream`] and [`FinalHashAggregateStream`] for details. //! -//! ### Group completion optimization +//! ### Group clustering optimization //! //! When the input is ordered by group keys, rows are clustered by those keys. //! The clustered paths use this guarantee to emit completed groups early. @@ -207,7 +207,7 @@ use datafusion_physical_expr_common::sort_expr::{ use datafusion_expr::utils::AggregateOrderSensitivity; use datafusion_physical_expr_common::utils::evaluate_expressions_to_arrays; use itertools::Itertools; -pub use order::GroupCompletionMode; +pub use order::GroupClusteringMode; use topk::hash_table::is_supported_hash_key_type; use topk::heap::is_supported_heap_type; @@ -911,12 +911,12 @@ pub struct AggregateExec { /// Execution metrics metrics: ExecutionPlanMetricsSet, required_input_ordering: Option, - /// Describes when the executor can determine that groups are complete. + /// Describes how input rows are clustered by grouping expressions. /// /// Input ordering describes a subset of the cases in which groups can be - /// safely emitted before the input ends. Full group completion requires only + /// safely emitted before the input ends. Full group clustering requires only /// that rows for each complete grouping tuple are contiguous. - group_completion_mode: GroupCompletionMode, + group_clustering_mode: GroupClusteringMode, cache: Arc, /// During initialization, if the plan supports dynamic filtering (see [`AggrDynFilter`]), /// it is set to `Some(..)` regardless of whether it can be pushed down to a child node. @@ -1107,7 +1107,7 @@ impl AggregateExec { // Commit the kind and properties together: heap output is unordered and final. self.kind = kind; - self.group_completion_mode = GroupCompletionMode::None; + self.group_clustering_mode = GroupClusteringMode::None; self.required_input_ordering = None; // Keep unchanged properties so parent aggregates do not need rebuilding. if !self.cache.eq_properties.oeq_class().is_empty() @@ -1332,21 +1332,21 @@ impl AggregateExec { .iter() .filter(|expr| input_eq_properties.is_expr_constant(expr).is_none()) .count(); - let mut group_completion_mode = if indices.len() == num_non_constant_groupby_exprs + let mut group_clustering_mode = if indices.len() == num_non_constant_groupby_exprs && !indices.is_empty() && group_by.groups.len() == 1 { - GroupCompletionMode::Full + GroupClusteringMode::Full } else if !indices.is_empty() { - GroupCompletionMode::Partial(indices) + GroupClusteringMode::Partial(indices) } else { - GroupCompletionMode::None + GroupClusteringMode::None }; // Grouping sets can change group keys. PartialReduce combines intermediate // states without using group boundaries to recognize completed groups. if group_by.has_grouping_set() || mode == AggregateMode::PartialReduce { - group_completion_mode = GroupCompletionMode::None; + group_clustering_mode = GroupClusteringMode::None; } // construct a map from the input expression to the output expression of the Aggregation group by @@ -1362,7 +1362,7 @@ impl AggregateExec { &group_expr_mapping, group_by.is_true_no_grouping(), &mode, - &group_completion_mode, + &group_clustering_mode, aggr_expr.as_ref(), )? }; @@ -1379,7 +1379,7 @@ impl AggregateExec { input_schema, metrics: ExecutionPlanMetricsSet::new(), required_input_ordering, - group_completion_mode, + group_clustering_mode, cache: Arc::new(cache), dynamic_filter: None, }; @@ -1687,45 +1687,45 @@ impl AggregateExec { // Choose the execution path based on aggregation mode and when groups // are known to be complete. use AggregateMode::*; - let stream = match (self.mode, &self.group_completion_mode) { - (Partial, GroupCompletionMode::None) => StreamType::PartialHash( + let stream = match (self.mode, &self.group_clustering_mode) { + (Partial, GroupClusteringMode::None) => StreamType::PartialHash( PartialHashAggregateStream::new(self, context, partition)?, ), - (Partial, GroupCompletionMode::Partial(_) | GroupCompletionMode::Full) => { + (Partial, GroupClusteringMode::Partial(_) | GroupClusteringMode::Full) => { StreamType::ClusteredPartialAggregate( ClusteredPartialAggregateStream::new(self, context, partition)?, ) } - (PartialReduce, GroupCompletionMode::None) => StreamType::PartialReduceHash( + (PartialReduce, GroupClusteringMode::None) => StreamType::PartialReduceHash( PartialReduceHashAggregateStream::new(self, context, partition)?, ), ( PartialReduce, - GroupCompletionMode::Partial(_) | GroupCompletionMode::Full, + GroupClusteringMode::Partial(_) | GroupClusteringMode::Full, ) => { return internal_err!( - "PartialReduce aggregation must use GroupCompletionMode::None" + "PartialReduce aggregation must use GroupClusteringMode::None" ); } - (Final | FinalPartitioned, GroupCompletionMode::None) => { + (Final | FinalPartitioned, GroupClusteringMode::None) => { StreamType::FinalHash(FinalHashAggregateStream::new( self, context, partition, )?) } ( Final | FinalPartitioned, - GroupCompletionMode::Partial(_) | GroupCompletionMode::Full, + GroupClusteringMode::Partial(_) | GroupClusteringMode::Full, ) => StreamType::ClusteredFinalAggregate(ClusteredFinalAggregateStream::new( self, context, partition, )?), - (Single | SinglePartitioned, GroupCompletionMode::None) => { + (Single | SinglePartitioned, GroupClusteringMode::None) => { StreamType::SingleHash(SingleHashAggregateStream::new( self, context, partition, )?) } ( Single | SinglePartitioned, - GroupCompletionMode::Partial(_) | GroupCompletionMode::Full, + GroupClusteringMode::Partial(_) | GroupClusteringMode::Full, ) => StreamType::ClusteredSingleAggregate( ClusteredSingleAggregateStream::new(self, context, partition)?, ), @@ -1786,7 +1786,7 @@ impl AggregateExec { group_expr_mapping: &ProjectionMapping, is_true_no_grouping: bool, mode: &AggregateMode, - group_completion_mode: &GroupCompletionMode, + group_clustering_mode: &GroupClusteringMode, aggr_exprs: &[Arc], ) -> Result { // Construct equivalence properties: @@ -1795,10 +1795,10 @@ impl AggregateExec { .project(group_expr_mapping, schema); // Only the clustered paths preserve existing ordering on group keys. - // Project the input's actual sort expressions; completion alone does + // Project the input's actual sort expressions; clustering alone does // not establish an output ordering. Keep this consistent with // `maintains_input_order`. - if *group_completion_mode == GroupCompletionMode::None { + if *group_clustering_mode == GroupClusteringMode::None { eq_properties.clear_orderings(); } @@ -1846,7 +1846,7 @@ impl AggregateExec { }; // TODO: Emission type and boundedness information can be enhanced here - let emission_type = if *group_completion_mode == GroupCompletionMode::None { + let emission_type = if *group_clustering_mode == GroupClusteringMode::None { EmissionType::Final } else { input.pipeline_behavior() @@ -1876,12 +1876,12 @@ impl AggregateExec { ) } - /// Describes when groups can be completed before the input ends. + /// Describes how input rows are clustered by grouping expressions. /// /// This does not imply a sort order. See [`ExecutionPlanProperties::output_ordering`] /// for the ordering of the aggregate's output. - pub fn group_completion_mode(&self) -> &GroupCompletionMode { - &self.group_completion_mode + pub fn group_clustering_mode(&self) -> &GroupClusteringMode { + &self.group_clustering_mode } /// Estimates output statistics for this aggregate node. @@ -2365,11 +2365,11 @@ impl DisplayAs for AggregateExec { write!(f, ", lim=[{}]", config.limit)?; } - if self.group_completion_mode != GroupCompletionMode::None { + if self.group_clustering_mode != GroupClusteringMode::None { write!( f, - ", group_completion_mode={:?}", - self.group_completion_mode + ", group_clustering_mode={:?}", + self.group_clustering_mode )?; } } @@ -2499,7 +2499,7 @@ impl ExecutionPlan for AggregateExec { /// Clustered aggregation preserves existing ordering on the group-by /// columns. Aggregate result columns do not inherit input ordering. fn maintains_input_order(&self) -> Vec { - vec![self.group_completion_mode != GroupCompletionMode::None] + vec![self.group_clustering_mode != GroupClusteringMode::None] } fn children(&self) -> Vec<&Arc> { @@ -2761,7 +2761,7 @@ impl ExecutionPlan for AggregateExec { // Derived at construction from the input ordering and `group_by`. required_input_ordering: _, // Derived at construction from the input ordering and `group_by`. - group_completion_mode: _, + group_clustering_mode: _, // Derived at construction by `Self::compute_properties`. cache: _, dynamic_filter, @@ -4174,8 +4174,8 @@ mod tests { (AggregateMode::Final, true, StreamType::ClusteredFinalAggregate(_)) | (AggregateMode::Single, true, StreamType::ClusteredSingleAggregate(_)) => { assert_eq!( - aggregate.group_completion_mode(), - &GroupCompletionMode::Partial(vec![0]) + aggregate.group_clustering_mode(), + &GroupClusteringMode::Partial(vec![0]) ); } _ => panic!("unexpected stream for {mode:?}, ordered={ordered}"), @@ -5613,7 +5613,7 @@ mod tests { #[rstest::rstest] #[case::full(true)] #[case::partial(false)] - fn group_completion_preserves_sort_options( + fn group_clustering_preserves_sort_options( #[case] full: bool, #[values(AggregateMode::Partial, AggregateMode::Single)] mode: AggregateMode, #[values(false, true)] descending: bool, @@ -5658,11 +5658,11 @@ mod tests { )?; let expected_mode = if full { - GroupCompletionMode::Full + GroupClusteringMode::Full } else { - GroupCompletionMode::Partial(vec![1]) + GroupClusteringMode::Partial(vec![1]) }; - assert_eq!(aggregate.group_completion_mode(), &expected_mode); + assert_eq!(aggregate.group_clustering_mode(), &expected_mode); assert_eq!(aggregate.maintains_input_order(), vec![true]); assert_eq!( aggregate.properties().emission_type, @@ -5680,7 +5680,7 @@ mod tests { } #[test] - fn group_completion_is_recomputed_with_new_children() -> Result<()> { + fn group_clustering_is_recomputed_with_new_children() -> Result<()> { let aggregate = Arc::new(single_test_aggregate()?); let original_input = Arc::clone(aggregate.input()); let schema = original_input.schema(); @@ -5698,8 +5698,8 @@ mod tests { )?; let sorted_aggregate = sorted.downcast_ref::().unwrap(); assert_eq!( - sorted_aggregate.group_completion_mode(), - &GroupCompletionMode::Full + sorted_aggregate.group_clustering_mode(), + &GroupClusteringMode::Full ); assert_eq!(sorted.output_ordering(), Some(&ordering)); assert_eq!(sorted.pipeline_behavior(), EmissionType::Incremental); @@ -5710,8 +5710,8 @@ mod tests { )?; let unordered_aggregate = unordered.downcast_ref::().unwrap(); assert_eq!( - unordered_aggregate.group_completion_mode(), - &GroupCompletionMode::None + unordered_aggregate.group_clustering_mode(), + &GroupClusteringMode::None ); assert_eq!(unordered.maintains_input_order(), vec![false]); assert!(unordered.output_ordering().is_none()); @@ -5769,8 +5769,8 @@ mod tests { Arc::clone(&schema), )?; assert_eq!( - aggregate.group_completion_mode(), - &GroupCompletionMode::Partial(vec![0]) + aggregate.group_clustering_mode(), + &GroupClusteringMode::Partial(vec![0]) ); let task_ctx = new_migrated_hash_ctx(2); @@ -5829,8 +5829,8 @@ mod tests { )?; assert_eq!( - partial_reduce.group_completion_mode(), - &GroupCompletionMode::None + partial_reduce.group_clustering_mode(), + &GroupClusteringMode::None ); assert_eq!(partial_reduce.maintains_input_order(), vec![false]); assert!( @@ -5887,8 +5887,8 @@ mod tests { )?; let aggregate = build_aggregate(unordered_input)?; assert_eq!( - aggregate.group_completion_mode(), - &GroupCompletionMode::None + aggregate.group_clustering_mode(), + &GroupClusteringMode::None ); assert_eq!( aggregate.schema().as_ref(), @@ -5922,8 +5922,8 @@ mod tests { Arc::new(TestMemoryExec::update_cache(&Arc::new(ordered_input))); let aggregate = build_aggregate(ordered_input)?; assert_eq!( - aggregate.group_completion_mode(), - &GroupCompletionMode::Full + aggregate.group_clustering_mode(), + &GroupClusteringMode::Full ); Ok(()) @@ -6243,8 +6243,8 @@ mod tests { )?; assert_eq!( - aggregate.group_completion_mode(), - &GroupCompletionMode::None + aggregate.group_clustering_mode(), + &GroupClusteringMode::None ); // This captures the behavior before #24438. When the source can declare // `(key, time_bin)` group-contiguous, the corresponding case can use @@ -6324,8 +6324,8 @@ mod tests { Arc::clone(&schema), )?; assert!(matches!( - aggregate.group_completion_mode(), - GroupCompletionMode::Partial(_) + aggregate.group_clustering_mode(), + GroupClusteringMode::Partial(_) )); let task_ctx = new_migrated_hash_ctx(2); @@ -6406,8 +6406,8 @@ mod tests { Arc::clone(&schema), )?; assert_eq!( - final_aggregate.group_completion_mode(), - &GroupCompletionMode::Full + final_aggregate.group_clustering_mode(), + &GroupClusteringMode::Full ); let task_ctx = new_migrated_hash_ctx(2); @@ -6482,8 +6482,8 @@ mod tests { Arc::clone(&schema), )?; assert!(matches!( - aggregate.group_completion_mode(), - GroupCompletionMode::Partial(_) + aggregate.group_clustering_mode(), + GroupClusteringMode::Partial(_) )); let runtime = RuntimeEnvBuilder::default() @@ -6781,7 +6781,7 @@ mod tests { // // "AggregateExec: mode=Final, gby=[a@0 as a], aggr=[FIRST_VALUE(b)]", // " CoalescePartitionsExec", - // " AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[FIRST_VALUE(b)], group_completion_mode=None", + // " AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[FIRST_VALUE(b)], group_clustering_mode=None", // " DataSourceExec: partitions=4, partition_sizes=[1, 1, 1, 1]", // // and checks whether the function `merge_batch` works correctly for diff --git a/datafusion/physical-plan/src/aggregates/order/full.rs b/datafusion/physical-plan/src/aggregates/order/full.rs index 30909f482f390..11bed6e533f3f 100644 --- a/datafusion/physical-plan/src/aggregates/order/full.rs +++ b/datafusion/physical-plan/src/aggregates/order/full.rs @@ -55,7 +55,7 @@ use std::mem::size_of; /// `0..12` can be emitted. Note that `13` can not yet be emitted as /// there may be more values in the next batch with the same group_id. #[derive(Debug)] -pub struct GroupCompletionFull { +pub struct GroupClusteringFull { state: State, } @@ -72,7 +72,7 @@ enum State { Complete, } -impl GroupCompletionFull { +impl GroupClusteringFull { pub fn new() -> Self { Self { state: State::Start, @@ -121,7 +121,7 @@ impl GroupCompletionFull { } /// Called when new groups are added in a batch. See documentation - /// on [`super::GroupCompletion::new_groups`] + /// on [`super::GroupClustering::new_groups`] pub fn new_groups(&mut self, total_num_groups: usize) { assert_ne!(total_num_groups, 0); @@ -149,7 +149,7 @@ impl GroupCompletionFull { } } -impl Default for GroupCompletionFull { +impl Default for GroupClusteringFull { fn default() -> Self { Self::new() } diff --git a/datafusion/physical-plan/src/aggregates/order/mod.rs b/datafusion/physical-plan/src/aggregates/order/mod.rs index b3a756d7739ed..8707a4f4820e7 100644 --- a/datafusion/physical-plan/src/aggregates/order/mod.rs +++ b/datafusion/physical-plan/src/aggregates/order/mod.rs @@ -24,14 +24,15 @@ use datafusion_expr::EmitTo; mod full; mod partial; -pub use full::GroupCompletionFull; -pub use partial::GroupCompletionPartial; +pub use full::GroupClusteringFull; +pub use partial::GroupClusteringPartial; -/// Describes how an aggregate can determine that groups are complete. +/// Describes how rows are clustered by grouping expressions within each input +/// partition. /// -/// Input ordering is one way to establish a group-completion mode, but the -/// execution machinery only needs to know when it can safely emit completed -/// groups. This mode does not describe the sort order of the input or output. +/// Input ordering is one way to establish clustering. Aggregate execution uses +/// these guarantees to determine when groups are complete and can be emitted. +/// This mode does not describe the sort order of the input or output. /// /// For example, when grouping by `key`, both inputs have fully contiguous /// groups within the input partition: @@ -44,8 +45,8 @@ pub use partial::GroupCompletionPartial; /// In both cases, once the key changes, the previous key will not appear again, /// so its group is complete and can be emitted. #[derive(Clone, Debug, PartialEq, Eq)] -pub enum GroupCompletionMode { - /// No group can be known complete before the input ends. +pub enum GroupClusteringMode { + /// No clustering guarantee is available to complete groups before the input ends. None, /// Rows with the same values at these grouping-expression indices form one /// contiguous range. When those values change, every group in the previous @@ -62,29 +63,29 @@ pub enum GroupCompletionMode { /// Tracks when groups in the hash table are complete and can be emitted. #[derive(Debug)] -pub enum GroupCompletion { +pub enum GroupClustering { /// No group can be known complete before the input ends. None, /// Rows are contiguous for a subset of the grouping keys. /// When those key values change, all groups in the previous run /// are complete and can be emitted. - Partial(GroupCompletionPartial), + Partial(GroupClusteringPartial), /// Rows are contiguous for the complete grouping tuple. /// When the tuple changes, the previous group can be emitted. - Full(GroupCompletionFull), + Full(GroupClusteringFull), } -impl GroupCompletion { - /// Create a `GroupCompletion` for the specified group-completion mode. - pub fn try_new(mode: &GroupCompletionMode) -> Result { +impl GroupClustering { + /// Create a `GroupClustering` for the specified group-clustering mode. + pub fn try_new(mode: &GroupClusteringMode) -> Result { match mode { - GroupCompletionMode::None => Ok(GroupCompletion::None), - GroupCompletionMode::Partial(grouping_indices) => { - GroupCompletionPartial::try_new(grouping_indices.clone()) - .map(GroupCompletion::Partial) + GroupClusteringMode::None => Ok(GroupClustering::None), + GroupClusteringMode::Partial(grouping_indices) => { + GroupClusteringPartial::try_new(grouping_indices.clone()) + .map(GroupClustering::Partial) } - GroupCompletionMode::Full => { - Ok(GroupCompletion::Full(GroupCompletionFull::new())) + GroupClusteringMode::Full => { + Ok(GroupClustering::Full(GroupClusteringFull::new())) } } } @@ -93,16 +94,16 @@ impl GroupCompletion { /// can be emitted. pub fn emit_to(&self) -> Option { match self { - GroupCompletion::None => None, - GroupCompletion::Partial(partial) => partial.emit_to(), - GroupCompletion::Full(full) => full.emit_to(), + GroupClustering::None => None, + GroupClustering::Partial(partial) => partial.emit_to(), + GroupClustering::Full(full) => full.emit_to(), } } /// Returns the emit strategy to use under memory pressure (OOM). /// /// Returns the strategy that must be used when emitting up to `n` groups - /// while respecting the configured group-completion mode. + /// while respecting the configured group-clustering mode. /// /// Returns `None` if no data can be emitted. pub fn oom_emit_to(&self, n: usize) -> Option { @@ -111,8 +112,8 @@ impl GroupCompletion { } match self { - GroupCompletion::None => Some(EmitTo::First(n)), - GroupCompletion::Partial(_) | GroupCompletion::Full(_) => { + GroupClustering::None => Some(EmitTo::First(n)), + GroupClustering::Partial(_) | GroupClustering::Full(_) => { self.emit_to().map(|emit_to| match emit_to { EmitTo::First(max) => EmitTo::First(n.min(max)), EmitTo::All => EmitTo::First(n), @@ -124,9 +125,9 @@ impl GroupCompletion { /// Updates the state to indicate that the input is complete. pub fn input_done(&mut self) { match self { - GroupCompletion::None => {} - GroupCompletion::Partial(partial) => partial.input_done(), - GroupCompletion::Full(full) => full.input_done(), + GroupClustering::None => {} + GroupClustering::Partial(partial) => partial.input_done(), + GroupClustering::Full(full) => full.input_done(), } } @@ -138,9 +139,9 @@ impl GroupCompletion { /// input batch from a fresh completion state. pub fn reset(&mut self) { match self { - GroupCompletion::None => {} - GroupCompletion::Partial(partial) => partial.reset(), - GroupCompletion::Full(full) => full.reset(), + GroupClustering::None => {} + GroupClustering::Partial(partial) => partial.reset(), + GroupClustering::Full(full) => full.reset(), } } @@ -148,9 +149,9 @@ impl GroupCompletion { /// existing indexes down by `n`. pub fn remove_groups(&mut self, n: usize) { match self { - GroupCompletion::None => {} - GroupCompletion::Partial(partial) => partial.remove_groups(n), - GroupCompletion::Full(full) => full.remove_groups(n), + GroupClustering::None => {} + GroupClustering::Partial(partial) => partial.remove_groups(n), + GroupClustering::Full(full) => full.remove_groups(n), } } @@ -169,15 +170,15 @@ impl GroupCompletion { total_num_groups: usize, ) -> Result<()> { match self { - GroupCompletion::None => {} - GroupCompletion::Partial(partial) => { + GroupClustering::None => {} + GroupClustering::Partial(partial) => { partial.new_groups( batch_group_values, group_indices, total_num_groups, )?; } - GroupCompletion::Full(full) => { + GroupClustering::Full(full) => { full.new_groups(total_num_groups); } } @@ -188,9 +189,9 @@ impl GroupCompletion { pub fn size(&self) -> usize { size_of::() + match self { - GroupCompletion::None => 0, - GroupCompletion::Partial(partial) => partial.size(), - GroupCompletion::Full(full) => full.size(), + GroupClustering::None => 0, + GroupClustering::Partial(partial) => partial.size(), + GroupClustering::Full(full) => full.size(), } } } @@ -204,21 +205,21 @@ mod tests { use arrow::array::Int32Array; #[test] - fn test_oom_emit_to_none_completion() { - let group_completion = GroupCompletion::None; + fn test_oom_emit_to_none_clustering() { + let group_clustering = GroupClustering::None; - assert_eq!(group_completion.oom_emit_to(0), None); - assert_eq!(group_completion.oom_emit_to(5), Some(EmitTo::First(5))); + assert_eq!(group_clustering.oom_emit_to(0), None); + assert_eq!(group_clustering.oom_emit_to(5), Some(EmitTo::First(5))); } - /// Creates a partial group-completion tracker with three groups. + /// Creates a partial group-clustering tracker with three groups. /// /// `group_key_values` controls whether a run boundary exists in the batch: /// distinct values such as `[1, 2, 3]` create boundaries, while repeated /// values such as `[1, 1, 1]` do not. - fn partial_completion(group_key_values: Vec) -> Result { - let mut group_completion = - GroupCompletion::Partial(GroupCompletionPartial::try_new(vec![0])?); + fn partial_clustering(group_key_values: Vec) -> Result { + let mut group_clustering = + GroupClustering::Partial(GroupClusteringPartial::try_new(vec![0])?); let batch_group_values: Vec = vec![ Arc::new(Int32Array::from(group_key_values)), @@ -226,30 +227,30 @@ mod tests { ]; let group_indices = vec![0, 1, 2]; - group_completion.new_groups(&batch_group_values, &group_indices, 3)?; + group_clustering.new_groups(&batch_group_values, &group_indices, 3)?; - Ok(group_completion) + Ok(group_clustering) } #[test] fn test_oom_emit_to_partial_clamps_to_boundary() -> Result<()> { - let group_completion = partial_completion(vec![1, 2, 3])?; + let group_clustering = partial_clustering(vec![1, 2, 3])?; // Can emit both `1` and `2` groups because we have seen `3` - assert_eq!(group_completion.emit_to(), Some(EmitTo::First(2))); - assert_eq!(group_completion.oom_emit_to(1), Some(EmitTo::First(1))); - assert_eq!(group_completion.oom_emit_to(3), Some(EmitTo::First(2))); + assert_eq!(group_clustering.emit_to(), Some(EmitTo::First(2))); + assert_eq!(group_clustering.oom_emit_to(1), Some(EmitTo::First(1))); + assert_eq!(group_clustering.oom_emit_to(3), Some(EmitTo::First(2))); Ok(()) } #[test] fn test_oom_emit_to_partial_without_boundary() -> Result<()> { - let group_completion = partial_completion(vec![1, 1, 1])?; + let group_clustering = partial_clustering(vec![1, 1, 1])?; // Can't emit the last `1` group as it may have more values - assert_eq!(group_completion.emit_to(), None); - assert_eq!(group_completion.oom_emit_to(3), None); + assert_eq!(group_clustering.emit_to(), None); + assert_eq!(group_clustering.oom_emit_to(3), None); Ok(()) } diff --git a/datafusion/physical-plan/src/aggregates/order/partial.rs b/datafusion/physical-plan/src/aggregates/order/partial.rs index df1637bd77785..c02d19ef4e472 100644 --- a/datafusion/physical-plan/src/aggregates/order/partial.rs +++ b/datafusion/physical-plan/src/aggregates/order/partial.rs @@ -62,7 +62,7 @@ use datafusion_expr::EmitTo; /// order) recent group index /// ``` #[derive(Debug)] -pub struct GroupCompletionPartial { +pub struct GroupClusteringPartial { /// State machine state: State, @@ -114,7 +114,7 @@ impl State { } } -impl GroupCompletionPartial { +impl GroupClusteringPartial { /// Creates a tracker for runs defined by the specified grouping columns. pub fn try_new(grouping_indices: Vec) -> Result { debug_assert!(!grouping_indices.is_empty()); @@ -208,7 +208,7 @@ impl GroupCompletionPartial { } /// Called when new groups are added in a batch. See documentation - /// on [`super::GroupCompletion::new_groups`] + /// on [`super::GroupClustering::new_groups`] pub fn new_groups( &mut self, batch_group_values: &[ArrayRef], @@ -280,11 +280,11 @@ mod tests { #[rstest::rstest] #[case::sorted([1, 2, 3, 4])] #[case::clustered([3, 1, 4, 2])] - fn test_group_completion_partial(#[case] keys: [i32; 4]) -> Result<()> { + fn test_group_clustering_partial(#[case] keys: [i32; 4]) -> Result<()> { let [first, second, third, fourth] = keys; // Contiguous on column a. let grouping_indices = vec![0]; - let mut group_completion = GroupCompletionPartial::try_new(grouping_indices)?; + let mut group_clustering = GroupClusteringPartial::try_new(grouping_indices)?; let batch_group_values: Vec = vec![ Arc::new(Int32Array::from(vec![first, second, third])), @@ -294,21 +294,21 @@ mod tests { let group_indices = vec![0, 1, 2]; let total_num_groups = 3; - group_completion.new_groups( + group_clustering.new_groups( &batch_group_values, &group_indices, total_num_groups, )?; assert_eq!( - group_completion.state, + group_clustering.state, State::InProgress { current_run_start: 2, group_key: vec![ScalarValue::Int32(Some(third))], current: 2 } ); - assert_eq!(group_completion.emit_to(), Some(EmitTo::First(2))); + assert_eq!(group_clustering.emit_to(), Some(EmitTo::First(2))); // push without a boundary let batch_group_values: Vec = vec![ @@ -318,21 +318,21 @@ mod tests { let group_indices = vec![3, 4, 5]; let total_num_groups = 6; - group_completion.new_groups( + group_clustering.new_groups( &batch_group_values, &group_indices, total_num_groups, )?; assert_eq!( - group_completion.state, + group_clustering.state, State::InProgress { current_run_start: 2, group_key: vec![ScalarValue::Int32(Some(third))], current: 5 } ); - assert_eq!(group_completion.emit_to(), Some(EmitTo::First(2))); + assert_eq!(group_clustering.emit_to(), Some(EmitTo::First(2))); // push with only a boundary to previous batch let batch_group_values: Vec = vec![ @@ -342,20 +342,20 @@ mod tests { let group_indices = vec![6, 7, 8]; let total_num_groups = 9; - group_completion.new_groups( + group_clustering.new_groups( &batch_group_values, &group_indices, total_num_groups, )?; assert_eq!( - group_completion.state, + group_clustering.state, State::InProgress { current_run_start: 6, group_key: vec![ScalarValue::Int32(Some(fourth))], current: 8 } ); - assert_eq!(group_completion.emit_to(), Some(EmitTo::First(6))); + assert_eq!(group_clustering.emit_to(), Some(EmitTo::First(6))); Ok(()) } diff --git a/datafusion/physical-plan/src/aggregates/partial_reduce_stream.rs b/datafusion/physical-plan/src/aggregates/partial_reduce_stream.rs index 5e930f1b72239..941edca2dd9fd 100644 --- a/datafusion/physical-plan/src/aggregates/partial_reduce_stream.rs +++ b/datafusion/physical-plan/src/aggregates/partial_reduce_stream.rs @@ -29,7 +29,7 @@ use datafusion_execution::memory_pool::{MemoryConsumer, MemoryReservation}; use futures::stream::{Stream, StreamExt}; use super::AggregateExec; -use super::GroupCompletionMode; +use super::GroupClusteringMode; use super::aggregate_hash_table::{AggregateHashTable, PartialReduceMarker}; use crate::metrics::{BaselineMetrics, Count, MetricBuilder, RecordOutput, SpillMetrics}; use crate::stream::EmptyRecordBatchStream; @@ -183,7 +183,7 @@ impl PartialReduceHashAggregateStream { partition: usize, ) -> Result { debug_assert_eq!(agg.mode, super::AggregateMode::PartialReduce); - debug_assert_eq!(agg.group_completion_mode, GroupCompletionMode::None); + debug_assert_eq!(agg.group_clustering_mode, GroupClusteringMode::None); let schema = Arc::clone(&agg.schema); let input = agg.input.execute(partition, Arc::clone(context))?; diff --git a/datafusion/physical-plan/src/aggregates/single_stream.rs b/datafusion/physical-plan/src/aggregates/single_stream.rs index c04d93f9d6b15..8d2edb7407244 100644 --- a/datafusion/physical-plan/src/aggregates/single_stream.rs +++ b/datafusion/physical-plan/src/aggregates/single_stream.rs @@ -32,7 +32,7 @@ use futures::stream::StreamExt; use super::aggregate_hash_table::{ AggregateHashTable, ClusteredAggregateTableMetrics, SingleMarker, }; -use super::order::GroupCompletionMode; +use super::order::GroupClusteringMode; use super::spill::AggregateSpill; use super::{AggregateExec, create_schema}; use crate::SendableRecordBatchStream; @@ -163,7 +163,7 @@ impl SingleHashAggregateStream { agg.mode, AggregateMode::Single | AggregateMode::SinglePartitioned )); - debug_assert_eq!(agg.group_completion_mode, GroupCompletionMode::None); + debug_assert_eq!(agg.group_clustering_mode, GroupClusteringMode::None); let schema = Arc::clone(&agg.schema); let input = agg.input.execute(partition, Arc::clone(context))?; @@ -194,7 +194,7 @@ impl SingleHashAggregateStream { context, partition, batch_size, - &GroupCompletionMode::None, + &GroupClusteringMode::None, &state_schema, spill_metrics, )?)) diff --git a/datafusion/physical-plan/src/aggregates/spill.rs b/datafusion/physical-plan/src/aggregates/spill.rs index 1bce4fd8d23bd..d2f763cfbd868 100644 --- a/datafusion/physical-plan/src/aggregates/spill.rs +++ b/datafusion/physical-plan/src/aggregates/spill.rs @@ -30,7 +30,7 @@ use datafusion_physical_expr_common::sort_expr::LexOrdering; use super::aggregate_hash_table::ClusteredAggregateTableMetrics; use super::clustered_final_stream::ClusteredFinalAggregateStream; -use super::order::GroupCompletionMode; +use super::order::GroupClusteringMode; use super::{AggregateExec, AggregateMode}; use crate::SendableRecordBatchStream; use crate::metrics::{BaselineMetrics, SpillMetrics}; @@ -127,11 +127,11 @@ impl AggregateSpill { /// Creates the spill context of a stream, whose spill requests are described /// as `label`. /// - /// `group_completion_mode` determines which group columns are already + /// `group_clustering_mode` determines which group columns are already /// contiguous. Spill files are sorted by those columns first, followed by /// the remaining ones. Existing output sort options are retained so replay - /// preserves any advertised ordering as well as group completion. - /// Full group completion aggregates in bounded memory and never spills. + /// preserves any advertised ordering as well as group clustering. + /// Full group clustering aggregates in bounded memory and never spills. /// /// `spill_schema` is the schema of the intermediate state batches. #[expect(clippy::too_many_arguments)] @@ -141,12 +141,12 @@ impl AggregateSpill { context: &Arc, partition: usize, batch_size: usize, - group_completion_mode: &GroupCompletionMode, + group_clustering_mode: &GroupClusteringMode, spill_schema: &SchemaRef, spill_metrics: SpillMetrics, ) -> Result { let mut replay_agg = agg.clone(); - replay_agg.group_completion_mode = GroupCompletionMode::Full; + replay_agg.group_clustering_mode = GroupClusteringMode::Full; let group_schema = match agg.mode { AggregateMode::Final | AggregateMode::FinalPartitioned => { agg.group_by().group_schema(spill_schema)? @@ -166,10 +166,10 @@ impl AggregateSpill { }; let num_group_columns = group_schema.fields().len(); - let contiguous_indices: &[usize] = match group_completion_mode { - GroupCompletionMode::None => &[], - GroupCompletionMode::Partial(contiguous_indices) => contiguous_indices, - GroupCompletionMode::Full => { + let contiguous_indices: &[usize] = match group_clustering_mode { + GroupClusteringMode::None => &[], + GroupClusteringMode::Partial(contiguous_indices) => contiguous_indices, + GroupClusteringMode::Full => { return internal_err!("{label}: fully contiguous groups do not spill"); } }; @@ -290,7 +290,7 @@ impl AggregateSpill { &context, partition, merged, - &GroupCompletionMode::Full, + &GroupClusteringMode::Full, baseline_metrics.clone(), metrics, None, diff --git a/datafusion/physical-plan/src/recursive_query.rs b/datafusion/physical-plan/src/recursive_query.rs index 270d03ccafca2..494887f9068ef 100644 --- a/datafusion/physical-plan/src/recursive_query.rs +++ b/datafusion/physical-plan/src/recursive_query.rs @@ -23,7 +23,7 @@ use std::task::{Context, Poll}; use super::work_table::{ReservedBatches, WorkTable}; use crate::aggregates::group_values::{GroupValues, new_group_values}; -use crate::aggregates::order::GroupCompletion; +use crate::aggregates::order::GroupClustering; use crate::common::project_plan_to_schema; use crate::execution_plan::{Boundedness, EmissionType, reset_plan_states}; use crate::metrics::{ @@ -464,7 +464,7 @@ struct DistinctDeduplicator { impl DistinctDeduplicator { fn new(schema: SchemaRef, task_context: &TaskContext) -> Result { - let group_values = new_group_values(schema, &GroupCompletion::None)?; + let group_values = new_group_values(schema, &GroupClustering::None)?; let reservation = MemoryConsumer::new("RecursiveQueryHashTable") .register(task_context.memory_pool()); Ok(Self { diff --git a/datafusion/sqllogictest/test_files/agg_func_substitute.slt b/datafusion/sqllogictest/test_files/agg_func_substitute.slt index 8402644f8ed90..f926b8d72df0e 100644 --- a/datafusion/sqllogictest/test_files/agg_func_substitute.slt +++ b/datafusion/sqllogictest/test_files/agg_func_substitute.slt @@ -44,9 +44,9 @@ logical_plan 03)----TableScan: multiple_ordered_table projection=[a, c] physical_plan 01)ProjectionExec: expr=[a@0 as a, nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]@1 as result] -02)--AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], group_completion_mode=Full +02)--AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], group_clustering_mode=Full 03)----RepartitionExec: partitioning=Hash([a@0], 4), input_partitions=4, preserve_order=true, sort_exprs=a@0 ASC NULLS LAST -04)------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], group_completion_mode=Full +04)------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], group_clustering_mode=Full 05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, maintains_sort_order=true 06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, c], output_orderings=[[a@0 ASC NULLS LAST], [c@1 ASC NULLS LAST]], file_type=csv, has_header=true @@ -62,9 +62,9 @@ logical_plan 03)----TableScan: multiple_ordered_table projection=[a, c] physical_plan 01)ProjectionExec: expr=[a@0 as a, nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]@1 as result] -02)--AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], group_completion_mode=Full +02)--AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], group_clustering_mode=Full 03)----RepartitionExec: partitioning=Hash([a@0], 4), input_partitions=4, preserve_order=true, sort_exprs=a@0 ASC NULLS LAST -04)------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], group_completion_mode=Full +04)------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], group_clustering_mode=Full 05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, maintains_sort_order=true 06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, c], output_orderings=[[a@0 ASC NULLS LAST], [c@1 ASC NULLS LAST]], file_type=csv, has_header=true @@ -79,9 +79,9 @@ logical_plan 03)----TableScan: multiple_ordered_table projection=[a, c] physical_plan 01)ProjectionExec: expr=[a@0 as a, nth_value(multiple_ordered_table.c,Int64(1) + Int64(100)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]@1 as result] -02)--AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c, 101) ORDER BY [multiple_ordered_table.c ASC NULLS LAST] as nth_value(multiple_ordered_table.c,Int64(1) + Int64(100)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], group_completion_mode=Full +02)--AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c, 101) ORDER BY [multiple_ordered_table.c ASC NULLS LAST] as nth_value(multiple_ordered_table.c,Int64(1) + Int64(100)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], group_clustering_mode=Full 03)----RepartitionExec: partitioning=Hash([a@0], 4), input_partitions=4, preserve_order=true, sort_exprs=a@0 ASC NULLS LAST -04)------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c, 101) ORDER BY [multiple_ordered_table.c ASC NULLS LAST] as nth_value(multiple_ordered_table.c,Int64(1) + Int64(100)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], group_completion_mode=Full +04)------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c, 101) ORDER BY [multiple_ordered_table.c ASC NULLS LAST] as nth_value(multiple_ordered_table.c,Int64(1) + Int64(100)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], group_clustering_mode=Full 05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, maintains_sort_order=true 06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, c], output_orderings=[[a@0 ASC NULLS LAST], [c@1 ASC NULLS LAST]], file_type=csv, has_header=true diff --git a/datafusion/sqllogictest/test_files/aggregate.slt b/datafusion/sqllogictest/test_files/aggregate.slt index f0a2385ab4f57..812d4c0d25202 100644 --- a/datafusion/sqllogictest/test_files/aggregate.slt +++ b/datafusion/sqllogictest/test_files/aggregate.slt @@ -9540,7 +9540,7 @@ CREATE TABLE stream_test ( (3, 1.0, 1.0, 7, false, 'e'), (3, 2.0, 2.0, 8, false, 'f'); # Test comprehensive aggregates with streaming -# This verifies that CORR and other aggregates work together in a streaming plan (group_completion_mode=Full) +# This verifies that CORR and other aggregates work together in a streaming plan (group_clustering_mode=Full) # Basic Aggregates query TT @@ -9579,7 +9579,7 @@ logical_plan 05)--------TableScan: stream_test projection=[g, x, y, i, b] physical_plan 01)ProjectionExec: expr=[g@0 as g, count(Int64(1))@1 as count(*), sum(stream_test.x)@2 as sum(stream_test.x), avg(stream_test.x)@3 as avg(stream_test.x), avg(stream_test.x)@3 as mean(stream_test.x), min(stream_test.x)@4 as min(stream_test.x), max(stream_test.y)@5 as max(stream_test.y), bit_and(stream_test.i)@6 as bit_and(stream_test.i), bit_or(stream_test.i)@7 as bit_or(stream_test.i), bit_xor(stream_test.i)@8 as bit_xor(stream_test.i), bool_and(stream_test.b)@9 as bool_and(stream_test.b), bool_or(stream_test.b)@10 as bool_or(stream_test.b), median(stream_test.x)@11 as median(stream_test.x), 0 as grouping(stream_test.g), var(stream_test.x)@12 as var(stream_test.x), var(stream_test.x)@12 as var_samp(stream_test.x), var_pop(stream_test.x)@13 as var_pop(stream_test.x), var(stream_test.x)@12 as var_sample(stream_test.x), var_pop(stream_test.x)@13 as var_population(stream_test.x), stddev(stream_test.x)@14 as stddev(stream_test.x), stddev(stream_test.x)@14 as stddev_samp(stream_test.x), stddev_pop(stream_test.x)@15 as stddev_pop(stream_test.x)] -02)--AggregateExec: mode=Single, gby=[g@0 as g], aggr=[count(Int64(1)), sum(stream_test.x), avg(stream_test.x), min(stream_test.x), max(stream_test.y), bit_and(stream_test.i), bit_or(stream_test.i), bit_xor(stream_test.i), bool_and(stream_test.b), bool_or(stream_test.b), median(stream_test.x), var(stream_test.x), var_pop(stream_test.x), stddev(stream_test.x), stddev_pop(stream_test.x)], group_completion_mode=Full +02)--AggregateExec: mode=Single, gby=[g@0 as g], aggr=[count(Int64(1)), sum(stream_test.x), avg(stream_test.x), min(stream_test.x), max(stream_test.y), bit_and(stream_test.i), bit_or(stream_test.i), bit_xor(stream_test.i), bool_and(stream_test.b), bool_or(stream_test.b), median(stream_test.x), var(stream_test.x), var_pop(stream_test.x), stddev(stream_test.x), stddev_pop(stream_test.x)], group_clustering_mode=Full 03)----SortExec: TopK(fetch=10000), expr=[g@0 ASC NULLS LAST], preserve_partitioning=[false] 04)------DataSourceExec: partitions=1, partition_sizes=[1] @@ -9634,7 +9634,7 @@ logical_plan 03)----Sort: stream_test.g ASC NULLS LAST, fetch=10000 04)------TableScan: stream_test projection=[g, x] physical_plan -01)AggregateExec: mode=Single, gby=[g@0 as g], aggr=[array_agg(stream_test.x) ORDER BY [stream_test.x ASC NULLS LAST], array_agg(DISTINCT stream_test.x) ORDER BY [stream_test.x ASC NULLS LAST], first_value(stream_test.x) ORDER BY [stream_test.x ASC NULLS LAST], last_value(stream_test.x) ORDER BY [stream_test.x ASC NULLS LAST], nth_value(stream_test.x,Int64(1)) ORDER BY [stream_test.x ASC NULLS LAST]], group_completion_mode=Full +01)AggregateExec: mode=Single, gby=[g@0 as g], aggr=[array_agg(stream_test.x) ORDER BY [stream_test.x ASC NULLS LAST], array_agg(DISTINCT stream_test.x) ORDER BY [stream_test.x ASC NULLS LAST], first_value(stream_test.x) ORDER BY [stream_test.x ASC NULLS LAST], last_value(stream_test.x) ORDER BY [stream_test.x ASC NULLS LAST], nth_value(stream_test.x,Int64(1)) ORDER BY [stream_test.x ASC NULLS LAST]], group_clustering_mode=Full 02)--SortExec: TopK(fetch=10000), expr=[g@0 ASC NULLS LAST, x@1 ASC NULLS LAST], preserve_partitioning=[false] 03)----DataSourceExec: partitions=1, partition_sizes=[1] @@ -9671,7 +9671,7 @@ logical_plan 03)----Sort: stream_test.g ASC NULLS LAST, fetch=10000 04)------TableScan: stream_test projection=[g, s] physical_plan -01)AggregateExec: mode=Single, gby=[g@0 as g], aggr=[array_agg(stream_test.s) ORDER BY [stream_test.s ASC NULLS LAST], string_agg(stream_test.s,Utf8("|")) ORDER BY [stream_test.s ASC NULLS LAST], string_agg(DISTINCT stream_test.s,Utf8("|")) ORDER BY [stream_test.s ASC NULLS LAST]], group_completion_mode=Full +01)AggregateExec: mode=Single, gby=[g@0 as g], aggr=[array_agg(stream_test.s) ORDER BY [stream_test.s ASC NULLS LAST], string_agg(stream_test.s,Utf8("|")) ORDER BY [stream_test.s ASC NULLS LAST], string_agg(DISTINCT stream_test.s,Utf8("|")) ORDER BY [stream_test.s ASC NULLS LAST]], group_clustering_mode=Full 02)--SortExec: TopK(fetch=10000), expr=[g@0 ASC NULLS LAST, s@1 ASC NULLS LAST], preserve_partitioning=[false] 03)----DataSourceExec: partitions=1, partition_sizes=[1] @@ -9718,7 +9718,7 @@ logical_plan 05)--------TableScan: stream_test projection=[g, x, y] physical_plan 01)ProjectionExec: expr=[g@0 as g, corr(stream_test.x,stream_test.y)@1 as corr(stream_test.x,stream_test.y), covar_samp(stream_test.x,stream_test.y)@2 as covar(stream_test.x,stream_test.y), covar_samp(stream_test.x,stream_test.y)@2 as covar_samp(stream_test.x,stream_test.y), covar_pop(stream_test.x,stream_test.y)@3 as covar_pop(stream_test.x,stream_test.y), regr_sxx(stream_test.x,stream_test.y)@4 as regr_sxx(stream_test.x,stream_test.y), regr_sxy(stream_test.x,stream_test.y)@5 as regr_sxy(stream_test.x,stream_test.y), regr_syy(stream_test.x,stream_test.y)@6 as regr_syy(stream_test.x,stream_test.y), regr_avgx(stream_test.x,stream_test.y)@7 as regr_avgx(stream_test.x,stream_test.y), regr_avgy(stream_test.x,stream_test.y)@8 as regr_avgy(stream_test.x,stream_test.y), regr_count(stream_test.x,stream_test.y)@9 as regr_count(stream_test.x,stream_test.y), regr_slope(stream_test.x,stream_test.y)@10 as regr_slope(stream_test.x,stream_test.y), regr_intercept(stream_test.x,stream_test.y)@11 as regr_intercept(stream_test.x,stream_test.y), regr_r2(stream_test.x,stream_test.y)@12 as regr_r2(stream_test.x,stream_test.y)] -02)--AggregateExec: mode=Single, gby=[g@0 as g], aggr=[corr(stream_test.x,stream_test.y), covar_samp(stream_test.x,stream_test.y), covar_pop(stream_test.x,stream_test.y), regr_sxx(stream_test.x,stream_test.y), regr_sxy(stream_test.x,stream_test.y), regr_syy(stream_test.x,stream_test.y), regr_avgx(stream_test.x,stream_test.y), regr_avgy(stream_test.x,stream_test.y), regr_count(stream_test.x,stream_test.y), regr_slope(stream_test.x,stream_test.y), regr_intercept(stream_test.x,stream_test.y), regr_r2(stream_test.x,stream_test.y)], group_completion_mode=Full +02)--AggregateExec: mode=Single, gby=[g@0 as g], aggr=[corr(stream_test.x,stream_test.y), covar_samp(stream_test.x,stream_test.y), covar_pop(stream_test.x,stream_test.y), regr_sxx(stream_test.x,stream_test.y), regr_sxy(stream_test.x,stream_test.y), regr_syy(stream_test.x,stream_test.y), regr_avgx(stream_test.x,stream_test.y), regr_avgy(stream_test.x,stream_test.y), regr_count(stream_test.x,stream_test.y), regr_slope(stream_test.x,stream_test.y), regr_intercept(stream_test.x,stream_test.y), regr_r2(stream_test.x,stream_test.y)], group_clustering_mode=Full 03)----SortExec: TopK(fetch=10000), expr=[g@0 ASC NULLS LAST], preserve_partitioning=[false] 04)------DataSourceExec: partitions=1, partition_sizes=[1] @@ -9771,7 +9771,7 @@ logical_plan 05)--------TableScan: stream_test projection=[g, x, i] physical_plan 01)ProjectionExec: expr=[g@0 as g, approx_distinct(stream_test.i)@1 as approx_distinct(stream_test.i), approx_median(stream_test.x)@2 as approx_median(stream_test.x), percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST]@3 as percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST]@3 as quantile_cont(stream_test.x,Float64(0.5)), approx_percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST]@4 as approx_percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], approx_percentile_cont_with_weight(Float64(1),Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST]@5 as approx_percentile_cont_with_weight(Float64(1),Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], percentile_cont(stream_test.x,Float64(0.5))@6 as percentile_cont(stream_test.x,Float64(0.5)), approx_percentile_cont(stream_test.x,Float64(0.5))@7 as approx_percentile_cont(stream_test.x,Float64(0.5)), approx_percentile_cont_with_weight(stream_test.x,Float64(1),Float64(0.5))@8 as approx_percentile_cont_with_weight(stream_test.x,Float64(1),Float64(0.5))] -02)--AggregateExec: mode=Single, gby=[g@0 as g], aggr=[approx_distinct(stream_test.i), approx_median(stream_test.x), percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], approx_percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], approx_percentile_cont_with_weight(Float64(1),Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], percentile_cont(stream_test.x,Float64(0.5)), approx_percentile_cont(stream_test.x,Float64(0.5)), approx_percentile_cont_with_weight(stream_test.x,Float64(1),Float64(0.5))], group_completion_mode=Full +02)--AggregateExec: mode=Single, gby=[g@0 as g], aggr=[approx_distinct(stream_test.i), approx_median(stream_test.x), percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], approx_percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], approx_percentile_cont_with_weight(Float64(1),Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], percentile_cont(stream_test.x,Float64(0.5)), approx_percentile_cont(stream_test.x,Float64(0.5)), approx_percentile_cont_with_weight(stream_test.x,Float64(1),Float64(0.5))], group_clustering_mode=Full 03)----SortExec: TopK(fetch=10000), expr=[g@0 ASC NULLS LAST], preserve_partitioning=[false] 04)------DataSourceExec: partitions=1, partition_sizes=[1] diff --git a/datafusion/sqllogictest/test_files/group_by.slt b/datafusion/sqllogictest/test_files/group_by.slt index ec1970a1101b7..281038f62ee9c 100644 --- a/datafusion/sqllogictest/test_files/group_by.slt +++ b/datafusion/sqllogictest/test_files/group_by.slt @@ -2112,7 +2112,7 @@ logical_plan 03)----TableScan: annotated_data_infinite2 projection=[a, b, c] physical_plan 01)ProjectionExec: expr=[a@1 as a, b@0 as b, sum(annotated_data_infinite2.c)@2 as summation1] -02)--AggregateExec: mode=Single, gby=[b@1 as b, a@0 as a], aggr=[sum(annotated_data_infinite2.c)], group_completion_mode=Full +02)--AggregateExec: mode=Single, gby=[b@1 as b, a@0 as a], aggr=[sum(annotated_data_infinite2.c)], group_clustering_mode=Full 03)----StreamingTableExec: partition_sizes=1, projection=[a, b, c], infinite_source=true, output_ordering=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, c@2 ASC NULLS LAST] @@ -2143,7 +2143,7 @@ logical_plan 03)----TableScan: annotated_data_infinite2 projection=[a, c, d] physical_plan 01)ProjectionExec: expr=[a@1 as a, d@0 as d, sum(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]@2 as summation1] -02)--AggregateExec: mode=Single, gby=[d@2 as d, a@0 as a], aggr=[sum(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]], group_completion_mode=Partial([1]) +02)--AggregateExec: mode=Single, gby=[d@2 as d, a@0 as a], aggr=[sum(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]], group_clustering_mode=Partial([1]) 03)----StreamingTableExec: partition_sizes=1, projection=[a, c, d], infinite_source=true, output_ordering=[a@0 ASC NULLS LAST] query III @@ -2176,7 +2176,7 @@ logical_plan 03)----TableScan: annotated_data_infinite2 projection=[a, b, c] physical_plan 01)ProjectionExec: expr=[a@0 as a, b@1 as b, first_value(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]@2 as first_c] -02)--AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[first_value(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]], group_completion_mode=Full +02)--AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[first_value(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]], group_clustering_mode=Full 03)----StreamingTableExec: partition_sizes=1, projection=[a, b, c], infinite_source=true, output_ordering=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, c@2 ASC NULLS LAST] query III @@ -2202,7 +2202,7 @@ logical_plan 03)----TableScan: annotated_data_infinite2 projection=[a, b, c] physical_plan 01)ProjectionExec: expr=[a@0 as a, b@1 as b, last_value(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]@2 as last_c] -02)--AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[last_value(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]], group_completion_mode=Full +02)--AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[last_value(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]], group_clustering_mode=Full 03)----StreamingTableExec: partition_sizes=1, projection=[a, b, c], infinite_source=true, output_ordering=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, c@2 ASC NULLS LAST] query III @@ -2229,7 +2229,7 @@ logical_plan 03)----TableScan: annotated_data_infinite2 projection=[a, b, c] physical_plan 01)ProjectionExec: expr=[a@0 as a, b@1 as b, last_value(annotated_data_infinite2.c)@2 as last_c] -02)--AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[last_value(annotated_data_infinite2.c)], group_completion_mode=Full +02)--AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[last_value(annotated_data_infinite2.c)], group_clustering_mode=Full 03)----StreamingTableExec: partition_sizes=1, projection=[a, b, c], infinite_source=true, output_ordering=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, c@2 ASC NULLS LAST] query III @@ -2288,7 +2288,7 @@ logical_plan 01)Aggregate: groupBy=[[annotated_data_infinite2.a, annotated_data_infinite2.b]], aggr=[[array_agg(annotated_data_infinite2.d) ORDER BY [annotated_data_infinite2.d ASC NULLS LAST]]] 02)--TableScan: annotated_data_infinite2 projection=[a, b, d] physical_plan -01)AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[array_agg(annotated_data_infinite2.d) ORDER BY [annotated_data_infinite2.d ASC NULLS LAST]], group_completion_mode=Full +01)AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[array_agg(annotated_data_infinite2.d) ORDER BY [annotated_data_infinite2.d ASC NULLS LAST]], group_clustering_mode=Full 02)--PartialSortExec: expr=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, d@2 ASC NULLS LAST], common_prefix_length=[2] 03)----StreamingTableExec: partition_sizes=1, projection=[a, b, d], infinite_source=true, output_ordering=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST] @@ -2612,7 +2612,7 @@ logical_plan 05)--------TableScan: sales_global projection=[country, amount] physical_plan 01)ProjectionExec: expr=[country@0 as country, array_agg(s.amount) ORDER BY [s.amount DESC NULLS FIRST]@1 as amounts, sum(s.amount)@2 as sum1] -02)--AggregateExec: mode=Single, gby=[country@0 as country], aggr=[array_agg(s.amount) ORDER BY [s.amount DESC NULLS FIRST], sum(s.amount)], group_completion_mode=Full +02)--AggregateExec: mode=Single, gby=[country@0 as country], aggr=[array_agg(s.amount) ORDER BY [s.amount DESC NULLS FIRST], sum(s.amount)], group_clustering_mode=Full 03)----SortExec: TopK(fetch=10), expr=[country@0 ASC NULLS LAST, amount@1 DESC], preserve_partitioning=[false] 04)------DataSourceExec: partitions=1, partition_sizes=[1] @@ -2650,7 +2650,7 @@ logical_plan 05)--------TableScan: sales_global projection=[zip_code, country, amount] physical_plan 01)ProjectionExec: expr=[country@0 as country, zip_code@1 as zip_code, array_agg(s.amount) ORDER BY [s.amount DESC NULLS FIRST]@2 as amounts, sum(s.amount)@3 as sum1] -02)--AggregateExec: mode=Single, gby=[country@1 as country, zip_code@0 as zip_code], aggr=[array_agg(s.amount) ORDER BY [s.amount DESC NULLS FIRST], sum(s.amount)], group_completion_mode=Partial([0]) +02)--AggregateExec: mode=Single, gby=[country@1 as country, zip_code@0 as zip_code], aggr=[array_agg(s.amount) ORDER BY [s.amount DESC NULLS FIRST], sum(s.amount)], group_clustering_mode=Partial([0]) 03)----SortExec: TopK(fetch=10), expr=[country@1 ASC NULLS LAST, amount@2 DESC], preserve_partitioning=[false] 04)------DataSourceExec: partitions=1, partition_sizes=[1] @@ -2687,7 +2687,7 @@ logical_plan 05)--------TableScan: sales_global projection=[country, amount] physical_plan 01)ProjectionExec: expr=[country@0 as country, array_agg(s.amount) ORDER BY [s.country DESC NULLS FIRST]@1 as amounts, sum(s.amount)@2 as sum1] -02)--AggregateExec: mode=Single, gby=[country@0 as country], aggr=[array_agg(s.amount) ORDER BY [s.country DESC NULLS FIRST], sum(s.amount)], group_completion_mode=Full +02)--AggregateExec: mode=Single, gby=[country@0 as country], aggr=[array_agg(s.amount) ORDER BY [s.country DESC NULLS FIRST], sum(s.amount)], group_clustering_mode=Full 03)----SortExec: TopK(fetch=10), expr=[country@0 ASC NULLS LAST], preserve_partitioning=[false] 04)------DataSourceExec: partitions=1, partition_sizes=[1] @@ -2723,7 +2723,7 @@ logical_plan 05)--------TableScan: sales_global projection=[country, amount] physical_plan 01)ProjectionExec: expr=[country@0 as country, array_agg(s.amount) ORDER BY [s.country DESC NULLS FIRST, s.amount DESC NULLS FIRST]@1 as amounts, sum(s.amount)@2 as sum1] -02)--AggregateExec: mode=Single, gby=[country@0 as country], aggr=[array_agg(s.amount) ORDER BY [s.country DESC NULLS FIRST, s.amount DESC NULLS FIRST], sum(s.amount)], group_completion_mode=Full +02)--AggregateExec: mode=Single, gby=[country@0 as country], aggr=[array_agg(s.amount) ORDER BY [s.country DESC NULLS FIRST, s.amount DESC NULLS FIRST], sum(s.amount)], group_clustering_mode=Full 03)----SortExec: TopK(fetch=10), expr=[country@0 ASC NULLS LAST, amount@1 DESC], preserve_partitioning=[false] 04)------DataSourceExec: partitions=1, partition_sizes=[1] @@ -4027,7 +4027,7 @@ logical_plan 12)------------------TableScan: multiple_ordered_table projection=[a, d] physical_plan 01)ProjectionExec: expr=[last_value(l.d) ORDER BY [l.a ASC NULLS LAST]@1 as amount_usd] -02)--AggregateExec: mode=Single, gby=[row_n@2 as row_n], aggr=[last_value(l.d) ORDER BY [l.a ASC NULLS LAST]], group_completion_mode=Full +02)--AggregateExec: mode=Single, gby=[row_n@2 as row_n], aggr=[last_value(l.d) ORDER BY [l.a ASC NULLS LAST]], group_clustering_mode=Full 03)----HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(d@1, d@1)], filter=CAST(a@0 AS Int64) >= CAST(a@1 AS Int64) - 10, projection=[a@0, d@1, row_n@4] 04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, d], output_ordering=[a@0 ASC NULLS LAST], file_type=csv, has_header=true 05)------ProjectionExec: expr=[a@0 as a, d@1 as d, row_number() ORDER BY [r.a ASC NULLS LAST] RANGE BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW@2 as row_n] @@ -4067,9 +4067,9 @@ logical_plan 01)Aggregate: groupBy=[[multiple_ordered_table_with_pk.c, multiple_ordered_table_with_pk.b]], aggr=[[sum(CAST(multiple_ordered_table_with_pk.d AS Int64))]] 02)--TableScan: multiple_ordered_table_with_pk projection=[b, c, d] physical_plan -01)AggregateExec: mode=FinalPartitioned, gby=[c@0 as c, b@1 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Partial([0]) +01)AggregateExec: mode=FinalPartitioned, gby=[c@0 as c, b@1 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Partial([0]) 02)--RepartitionExec: partitioning=Hash([c@0, b@1], 8), input_partitions=8, preserve_order=true, sort_exprs=c@0 ASC NULLS LAST -03)----AggregateExec: mode=Partial, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Partial([0]) +03)----AggregateExec: mode=Partial, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Partial([0]) 04)------RepartitionExec: partitioning=RoundRobinBatch(8), input_partitions=1, maintains_sort_order=true 05)--------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[b, c, d], output_ordering=[c@1 ASC NULLS LAST], constraints=[PrimaryKey([3])], file_type=csv, has_header=true @@ -4106,9 +4106,9 @@ logical_plan 01)Aggregate: groupBy=[[multiple_ordered_table_with_pk.c, multiple_ordered_table_with_pk.b]], aggr=[[sum(CAST(multiple_ordered_table_with_pk.d AS Int64))]] 02)--TableScan: multiple_ordered_table_with_pk projection=[b, c, d] physical_plan -01)AggregateExec: mode=FinalPartitioned, gby=[c@0 as c, b@1 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Partial([0]) +01)AggregateExec: mode=FinalPartitioned, gby=[c@0 as c, b@1 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Partial([0]) 02)--RepartitionExec: partitioning=Hash([c@0, b@1], 8), input_partitions=8, preserve_order=true, sort_exprs=c@0 ASC NULLS LAST -03)----AggregateExec: mode=Partial, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Partial([0]) +03)----AggregateExec: mode=Partial, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Partial([0]) 04)------RepartitionExec: partitioning=RoundRobinBatch(8), input_partitions=1, maintains_sort_order=true 05)--------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[b, c, d], output_ordering=[c@1 ASC NULLS LAST], constraints=[PrimaryKey([3])], file_type=csv, has_header=true @@ -4129,9 +4129,9 @@ logical_plan 03)----Aggregate: groupBy=[[multiple_ordered_table_with_pk.c]], aggr=[[sum(CAST(multiple_ordered_table_with_pk.d AS Int64))]] 04)------TableScan: multiple_ordered_table_with_pk projection=[c, d] physical_plan -01)AggregateExec: mode=Single, gby=[c@0 as c, sum1@1 as sum1], aggr=[], group_completion_mode=Partial([0]) +01)AggregateExec: mode=Single, gby=[c@0 as c, sum1@1 as sum1], aggr=[], group_clustering_mode=Partial([0]) 02)--ProjectionExec: expr=[c@0 as c, sum(multiple_ordered_table_with_pk.d)@1 as sum1] -03)----AggregateExec: mode=Single, gby=[c@0 as c], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Full +03)----AggregateExec: mode=Single, gby=[c@0 as c], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Full 04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[c, d], output_ordering=[c@0 ASC NULLS LAST], constraints=[PrimaryKey([3])], file_type=csv, has_header=true query TT @@ -4151,7 +4151,7 @@ physical_plan 01)ProjectionExec: expr=[c@0 as c, sum1@2 as sum1, sum(multiple_ordered_table_with_pk.b) ROWS BETWEEN UNBOUNDED PRECEDING AND UNBOUNDED FOLLOWING@3 as sumb] 02)--WindowAggExec: wdw=[sum(multiple_ordered_table_with_pk.b) ROWS BETWEEN UNBOUNDED PRECEDING AND UNBOUNDED FOLLOWING: Ok(Field { name: "sum(multiple_ordered_table_with_pk.b) ROWS BETWEEN UNBOUNDED PRECEDING AND UNBOUNDED FOLLOWING", data_type: Int64, nullable: true }), frame: WindowFrame { units: Rows, start_bound: Preceding(UInt64(NULL)), end_bound: Following(UInt64(NULL)), is_causal: false }] 03)----ProjectionExec: expr=[c@0 as c, b@1 as b, sum(multiple_ordered_table_with_pk.d)@2 as sum1] -04)------AggregateExec: mode=Single, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Partial([0]) +04)------AggregateExec: mode=Single, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Partial([0]) 05)--------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[b, c, d], output_ordering=[c@1 ASC NULLS LAST], constraints=[PrimaryKey([3])], file_type=csv, has_header=true query TT @@ -4180,10 +4180,10 @@ logical_plan physical_plan 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(b@1, b@1)], projection=[c@0, c@3, sum1@2, sum1@5] 02)--ProjectionExec: expr=[c@0 as c, b@1 as b, sum(multiple_ordered_table_with_pk.d)@2 as sum1] -03)----AggregateExec: mode=Single, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Partial([0]) +03)----AggregateExec: mode=Single, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Partial([0]) 04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[b, c, d], output_ordering=[c@1 ASC NULLS LAST], constraints=[PrimaryKey([3])], file_type=csv, has_header=true 05)--ProjectionExec: expr=[c@0 as c, b@1 as b, sum(multiple_ordered_table_with_pk.d)@2 as sum1] -06)----AggregateExec: mode=Single, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Partial([0]) +06)----AggregateExec: mode=Single, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Partial([0]) 07)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[b, c, d], output_ordering=[c@1 ASC NULLS LAST], constraints=[PrimaryKey([3])], file_type=csv, has_header=true query TT @@ -4212,10 +4212,10 @@ physical_plan 01)ProjectionExec: expr=[c@0 as c, c@2 as c, sum1@1 as sum1, sum1@3 as sum1] 02)--CrossJoinExec 03)----ProjectionExec: expr=[c@0 as c, sum(multiple_ordered_table_with_pk.d)@1 as sum1] -04)------AggregateExec: mode=Single, gby=[c@0 as c], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Full +04)------AggregateExec: mode=Single, gby=[c@0 as c], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Full 05)--------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[c, d], output_ordering=[c@0 ASC NULLS LAST], constraints=[PrimaryKey([3])], file_type=csv, has_header=true 06)----ProjectionExec: expr=[c@0 as c, sum(multiple_ordered_table_with_pk.d)@1 as sum1] -07)------AggregateExec: mode=Single, gby=[c@0 as c], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Full +07)------AggregateExec: mode=Single, gby=[c@0 as c], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Full 08)--------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[c, d], output_ordering=[c@0 ASC NULLS LAST], constraints=[PrimaryKey([3])], file_type=csv, has_header=true # we do not generate physical plan for Repartition yet (e.g Distribute By queries). @@ -4254,10 +4254,10 @@ logical_plan physical_plan 01)UnionExec 02)--ProjectionExec: expr=[c@0 as c, a@1 as a, sum(multiple_ordered_table_with_pk.d)@2 as sum1] -03)----AggregateExec: mode=Single, gby=[c@1 as c, a@0 as a], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Full +03)----AggregateExec: mode=Single, gby=[c@1 as c, a@0 as a], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Full 04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, c, d], output_orderings=[[a@0 ASC NULLS LAST], [c@1 ASC NULLS LAST]], constraints=[PrimaryKey([3])], file_type=csv, has_header=true 05)--ProjectionExec: expr=[c@0 as c, a@1 as a, sum(multiple_ordered_table_with_pk.d)@2 as sum1] -06)----AggregateExec: mode=Single, gby=[c@1 as c, a@0 as a], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Full +06)----AggregateExec: mode=Single, gby=[c@1 as c, a@0 as a], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Full 07)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, c, d], output_orderings=[[a@0 ASC NULLS LAST], [c@1 ASC NULLS LAST]], constraints=[PrimaryKey([3])], file_type=csv, has_header=true # table scan should be simplified. @@ -4272,7 +4272,7 @@ logical_plan 03)----TableScan: multiple_ordered_table_with_pk projection=[a, c, d] physical_plan 01)ProjectionExec: expr=[c@0 as c, a@1 as a, sum(multiple_ordered_table_with_pk.d)@2 as sum1] -02)--AggregateExec: mode=Single, gby=[c@1 as c, a@0 as a], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Full +02)--AggregateExec: mode=Single, gby=[c@1 as c, a@0 as a], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Full 03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, c, d], output_orderings=[[a@0 ASC NULLS LAST], [c@1 ASC NULLS LAST]], constraints=[PrimaryKey([3])], file_type=csv, has_header=true # limit should be simplified @@ -4291,7 +4291,7 @@ logical_plan physical_plan 01)ProjectionExec: expr=[c@0 as c, a@1 as a, sum(multiple_ordered_table_with_pk.d)@2 as sum1] 02)--GlobalLimitExec: skip=0, fetch=5 -03)----AggregateExec: mode=Single, gby=[c@1 as c, a@0 as a], aggr=[sum(multiple_ordered_table_with_pk.d)], group_completion_mode=Full +03)----AggregateExec: mode=Single, gby=[c@1 as c, a@0 as a], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Full 04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, c, d], output_orderings=[[a@0 ASC NULLS LAST], [c@1 ASC NULLS LAST]], constraints=[PrimaryKey([3])], file_type=csv, has_header=true statement ok @@ -4374,9 +4374,9 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [time_chunks@0 DESC], fetch=5 02)--ProjectionExec: expr=[date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)@0 as time_chunks] -03)----AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)@0 as date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)], aggr=[], group_completion_mode=Full +03)----AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)@0 as date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)], aggr=[], group_clustering_mode=Full 04)------RepartitionExec: partitioning=Hash([date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)@0], 8), input_partitions=8, preserve_order=true, sort_exprs=date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)@0 DESC -05)--------AggregateExec: mode=Partial, gby=[date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 900000000000 }, ts@0) as date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)], aggr=[], group_completion_mode=Full +05)--------AggregateExec: mode=Partial, gby=[date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 900000000000 }, ts@0) as date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)], aggr=[], group_clustering_mode=Full 06)----------RepartitionExec: partitioning=RoundRobinBatch(8), input_partitions=1, maintains_sort_order=true 07)------------StreamingTableExec: partition_sizes=1, projection=[ts], infinite_source=true, output_ordering=[ts@0 DESC] @@ -5099,7 +5099,7 @@ logical_plan 02)--Aggregate: groupBy=[[multiple_ordered_table.a, multiple_ordered_table.b]], aggr=[[array_agg(multiple_ordered_table.c) ORDER BY [multiple_ordered_table.c DESC NULLS FIRST]]] 03)----TableScan: multiple_ordered_table projection=[a, b, c] physical_plan -01)AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[array_agg(multiple_ordered_table.c) ORDER BY [multiple_ordered_table.c DESC NULLS FIRST]], group_completion_mode=Full +01)AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[array_agg(multiple_ordered_table.c) ORDER BY [multiple_ordered_table.c DESC NULLS FIRST]], group_clustering_mode=Full 02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, b, c], output_orderings=[[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST], [c@2 ASC NULLS LAST]], file_type=csv, has_header=true query II? diff --git a/datafusion/sqllogictest/test_files/joins.slt b/datafusion/sqllogictest/test_files/joins.slt index c929822263bd8..5d2ba6025d42c 100644 --- a/datafusion/sqllogictest/test_files/joins.slt +++ b/datafusion/sqllogictest/test_files/joins.slt @@ -3510,7 +3510,7 @@ logical_plan 08)----------TableScan: annotated_data projection=[a, b] physical_plan 01)ProjectionExec: expr=[a@0 as a, last_value(r.b) ORDER BY [r.a ASC NULLS FIRST]@3 as last_col1] -02)--AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b, c@2 as c], aggr=[last_value(r.b) ORDER BY [r.a ASC NULLS FIRST]], group_completion_mode=Partial([0]) +02)--AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b, c@2 as c], aggr=[last_value(r.b) ORDER BY [r.a ASC NULLS FIRST]], group_clustering_mode=Partial([0]) 03)----HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(a@0, a@0)] 04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, b, c], output_ordering=[a@0 ASC, b@1 ASC NULLS LAST, c@2 ASC NULLS LAST], file_type=csv, has_header=true 05)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, b], output_ordering=[a@0 ASC, b@1 ASC NULLS LAST], file_type=csv, has_header=true @@ -3557,7 +3557,7 @@ logical_plan 12)------------------TableScan: multiple_ordered_table projection=[a, d] physical_plan 01)ProjectionExec: expr=[last_value(l.d) ORDER BY [l.a ASC NULLS LAST]@1 as amount_usd] -02)--AggregateExec: mode=Single, gby=[row_n@2 as row_n], aggr=[last_value(l.d) ORDER BY [l.a ASC NULLS LAST]], group_completion_mode=Full +02)--AggregateExec: mode=Single, gby=[row_n@2 as row_n], aggr=[last_value(l.d) ORDER BY [l.a ASC NULLS LAST]], group_clustering_mode=Full 03)----HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(d@1, d@1)], filter=CAST(a@0 AS Int64) >= CAST(a@1 AS Int64) - 10, projection=[a@0, d@1, row_n@4] 04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, d], output_ordering=[a@0 ASC NULLS LAST], file_type=csv, has_header=true 05)------ProjectionExec: expr=[a@0 as a, d@1 as d, row_number() ORDER BY [r.a ASC NULLS LAST] RANGE BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW@2 as row_n] @@ -3592,9 +3592,9 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [a@0 ASC] 02)--ProjectionExec: expr=[a@0 as a, last_value(r.b) ORDER BY [r.a ASC NULLS FIRST]@3 as last_col1] -03)----AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b, c@2 as c], aggr=[last_value(r.b) ORDER BY [r.a ASC NULLS FIRST]], group_completion_mode=Partial([0]) +03)----AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b, c@2 as c], aggr=[last_value(r.b) ORDER BY [r.a ASC NULLS FIRST]], group_clustering_mode=Partial([0]) 04)------RepartitionExec: partitioning=Hash([a@0, b@1, c@2], 2), input_partitions=2, preserve_order=true, sort_exprs=a@0 ASC -05)--------AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b, c@2 as c], aggr=[last_value(r.b) ORDER BY [r.a ASC NULLS FIRST]], group_completion_mode=Partial([0]) +05)--------AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b, c@2 as c], aggr=[last_value(r.b) ORDER BY [r.a ASC NULLS FIRST]], group_clustering_mode=Partial([0]) 06)----------HashJoinExec: mode=Partitioned, join_type=Inner, on=[(a@0, a@0)] 07)------------RepartitionExec: partitioning=Hash([a@0], 2), input_partitions=1, maintains_sort_order=true 08)--------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, b, c], output_ordering=[a@0 ASC, b@1 ASC NULLS LAST, c@2 ASC NULLS LAST], file_type=csv, has_header=true diff --git a/datafusion/sqllogictest/test_files/order.slt b/datafusion/sqllogictest/test_files/order.slt index 343e5db9dd3b5..bcd069fb03134 100644 --- a/datafusion/sqllogictest/test_files/order.slt +++ b/datafusion/sqllogictest/test_files/order.slt @@ -1898,7 +1898,7 @@ EXPLAIN SELECT c1, SUM(c2) as sum_c2 FROM table_with_ordered_not_null GROUP BY c ---- physical_plan 01)ProjectionExec: expr=[c1@0 as c1, sum(table_with_ordered_not_null.c2)@1 as sum_c2] -02)--AggregateExec: mode=Single, gby=[c1@0 as c1], aggr=[sum(table_with_ordered_not_null.c2)], group_completion_mode=Full +02)--AggregateExec: mode=Single, gby=[c1@0 as c1], aggr=[sum(table_with_ordered_not_null.c2)], group_clustering_mode=Full 03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/aggregate_agg_multi_order.csv]]}, projection=[c1, c2], output_ordering=[c1@0 ASC NULLS LAST], file_type=csv, has_header=true statement ok diff --git a/datafusion/sqllogictest/test_files/ordered_aggregate_spill.slt b/datafusion/sqllogictest/test_files/ordered_aggregate_spill.slt index a8e0ff065469c..6e0c72e58ea12 100644 --- a/datafusion/sqllogictest/test_files/ordered_aggregate_spill.slt +++ b/datafusion/sqllogictest/test_files/ordered_aggregate_spill.slt @@ -55,9 +55,9 @@ FROM generate_series(20000) AS t1(v1) GROUP BY v1 ---- Plan with Metrics -01)AggregateExec: mode=FinalPartitioned,group_completion_mode=Full, metrics=[spill_count=0,] +01)AggregateExec: mode=FinalPartitioned,group_clustering_mode=Full, metrics=[spill_count=0,] 02)--RepartitionExec:preserve_order=true -03)----AggregateExec: mode=Partial,group_completion_mode=Full, metrics=[spill_count=0,] +03)----AggregateExec: mode=Partial,group_clustering_mode=Full, metrics=[spill_count=0,] query II rowsort @@ -100,9 +100,9 @@ FROM generate_series(20000) AS t1(v1) GROUP BY round(v1, -4), v1 % 5000 ---- Plan with Metrics -01)AggregateExec: mode=FinalPartitioned,aggr=[sum(t1.v1 * Int64(2))], group_completion_mode=Partial([0]), metrics=[spill_count=0,] +01)AggregateExec: mode=FinalPartitioned,aggr=[sum(t1.v1 * Int64(2))], group_clustering_mode=Partial([0]), metrics=[spill_count=0,] 02)--RepartitionExec:input_partitions=1, maintains_sort_order=true -03)----AggregateExec: mode=Partial,group_completion_mode=Partial([0]), metrics=[spill_count=0,] +03)----AggregateExec: mode=Partial,group_clustering_mode=Partial([0]), metrics=[spill_count=0,] # All rounds should have the same result hash @@ -124,9 +124,9 @@ FROM generate_series(20000) AS t1(v1) GROUP BY round(v1, -4), v1 % 5000 ---- Plan with Metrics -01)AggregateExec: mode=FinalPartitioned,aggr=[sum(t1.v1 * Int64(2))], group_completion_mode=Partial([0]), metrics=[spilled_bytes= KB,] +01)AggregateExec: mode=FinalPartitioned,aggr=[sum(t1.v1 * Int64(2))], group_clustering_mode=Partial([0]), metrics=[spilled_bytes= KB,] 02)--RepartitionExec:input_partitions=1, maintains_sort_order=true -03)----AggregateExec: mode=Partial,group_completion_mode=Partial([0]), metrics=[spill_count=0,] +03)----AggregateExec: mode=Partial,group_clustering_mode=Partial([0]), metrics=[spill_count=0,] # All rounds should have the same result hash @@ -148,9 +148,9 @@ FROM generate_series(20000) AS t1(v1) GROUP BY round(v1, -4), v1 % 5000 ---- Plan with Metrics -01)AggregateExec: mode=FinalPartitioned,aggr=[sum(t1.v1 * Int64(2))], group_completion_mode=Partial([0]), metrics=[spilled_bytes= KB,] +01)AggregateExec: mode=FinalPartitioned,aggr=[sum(t1.v1 * Int64(2))], group_clustering_mode=Partial([0]), metrics=[spilled_bytes= KB,] 02)--RepartitionExec:input_partitions=1, maintains_sort_order=true -03)----AggregateExec: mode=Partial,group_completion_mode=Partial([0]), metrics=[spill_count=0,] +03)----AggregateExec: mode=Partial,group_clustering_mode=Partial([0]), metrics=[spill_count=0,] # All rounds should have the same result hash @@ -176,9 +176,9 @@ FROM generate_series(20000) AS t1(v1) GROUP BY round(v1, -4), v1 % 5000 ---- Plan with Metrics -01)AggregateExec: mode=FinalPartitioned,aggr=[sum(t1.v1 * Int64(2)), min(t1.v1 % Int64(2))], group_completion_mode=Partial([0]), metrics=[spilled_rows= K,] +01)AggregateExec: mode=FinalPartitioned,aggr=[sum(t1.v1 * Int64(2)), min(t1.v1 % Int64(2))], group_clustering_mode=Partial([0]), metrics=[spilled_rows= K,] 02)--RepartitionExec:input_partitions=1, maintains_sort_order=true -03)----AggregateExec: mode=Partial,aggr=[sum(t1.v1 * Int64(2)), min(t1.v1 % Int64(2))], group_completion_mode=Partial([0]), metrics=[spill_count=0,] +03)----AggregateExec: mode=Partial,aggr=[sum(t1.v1 * Int64(2)), min(t1.v1 % Int64(2))], group_clustering_mode=Partial([0]), metrics=[spill_count=0,] # ================================================================================== @@ -201,7 +201,7 @@ FROM generate_series(20000) AS t1(v1) GROUP BY round(v1, -4), v1 % 5000 ---- Plan with Metrics -01)AggregateExec: mode=Single,aggr=[min(t1.v1 * Int64(2)), avg(t1.v1)], group_completion_mode=Partial([0]), metrics=[spill_count=0,] +01)AggregateExec: mode=Single,aggr=[min(t1.v1 * Int64(2)), avg(t1.v1)], group_clustering_mode=Partial([0]), metrics=[spill_count=0,] query IIIR rowsort @@ -222,7 +222,7 @@ FROM generate_series(20000) AS t1(v1) GROUP BY round(v1, -4), v1 % 5000 ---- Plan with Metrics -01)AggregateExec: mode=Single,aggr=[min(t1.v1 * Int64(2)), avg(t1.v1)], group_completion_mode=Partial([0]), metrics=[spilled_bytes= KB,] +01)AggregateExec: mode=Single,aggr=[min(t1.v1 * Int64(2)), avg(t1.v1)], group_clustering_mode=Partial([0]), metrics=[spilled_bytes= KB,] # Same result hash as the no-spill round above diff --git a/datafusion/sqllogictest/test_files/preserve_file_partitioning.slt b/datafusion/sqllogictest/test_files/preserve_file_partitioning.slt index 7e187669ce56d..f4aab94ea627d 100644 --- a/datafusion/sqllogictest/test_files/preserve_file_partitioning.slt +++ b/datafusion/sqllogictest/test_files/preserve_file_partitioning.slt @@ -287,9 +287,9 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, count(Int64(1))@1 as count(*), avg(fact_table_ordered.value)@2 as avg(fact_table_ordered.value)] -03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], group_completion_mode=Full +03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], group_clustering_mode=Full 04)------RepartitionExec: partitioning=Hash([f_dkey@0], 3), input_partitions=3, preserve_order=true, sort_exprs=f_dkey@0 ASC NULLS LAST -05)--------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], group_completion_mode=Full +05)--------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], group_clustering_mode=Full 06)----------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=C/data.parquet]]}, projection=[value, f_dkey], output_ordering=[f_dkey@1 ASC NULLS LAST], file_type=parquet # Verify results without optimization @@ -319,7 +319,7 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, count(Int64(1))@1 as count(*), avg(fact_table_ordered.value)@2 as avg(fact_table_ordered.value)] -03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], group_completion_mode=Full +03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], group_clustering_mode=Full 04)------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=C/data.parquet]]}, projection=[value, f_dkey], output_ordering=[f_dkey@1 ASC NULLS LAST], output_partitioning=Hash([f_dkey@1], 3), file_type=parquet query TIR @@ -359,9 +359,9 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, max(d.env)@1 as max(d.env), max(d.service)@2 as max(d.service), count(Int64(1))@3 as count(*), sum(f.value)@4 as sum(f.value)] -03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[max(d.env), max(d.service), count(Int64(1)), sum(f.value)], group_completion_mode=Full +03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[max(d.env), max(d.service), count(Int64(1)), sum(f.value)], group_clustering_mode=Full 04)------RepartitionExec: partitioning=Hash([f_dkey@0], 3), input_partitions=3, preserve_order=true, sort_exprs=f_dkey@0 ASC NULLS LAST -05)--------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[max(d.env), max(d.service), count(Int64(1)), sum(f.value)], group_completion_mode=Full +05)--------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[max(d.env), max(d.service), count(Int64(1)), sum(f.value)], group_clustering_mode=Full 06)----------HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(d_dkey@0, f_dkey@1)], projection=[value@3, f_dkey@4, env@1, service@2] 07)------------CoalescePartitionsExec 08)--------------FilterExec: service@2 = log @@ -412,7 +412,7 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, max(d.env)@1 as max(d.env), max(d.service)@2 as max(d.service), count(Int64(1))@3 as count(*), sum(f.value)@4 as sum(f.value)] -03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[max(d.env), max(d.service), count(Int64(1)), sum(f.value)], group_completion_mode=Full +03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[max(d.env), max(d.service), count(Int64(1)), sum(f.value)], group_clustering_mode=Full 04)------HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(d_dkey@0, f_dkey@1)], projection=[value@3, f_dkey@4, env@1, service@2] 05)--------CoalescePartitionsExec 06)----------FilterExec: service@2 = log diff --git a/datafusion/sqllogictest/test_files/range_sorted_time_bin_agg.slt b/datafusion/sqllogictest/test_files/range_sorted_time_bin_agg.slt index 058d728f505cb..400bbab1a3592 100644 --- a/datafusion/sqllogictest/test_files/range_sorted_time_bin_agg.slt +++ b/datafusion/sqllogictest/test_files/range_sorted_time_bin_agg.slt @@ -35,9 +35,9 @@ # single streaming SinglePartitioned step with no hash shuffle. # # Today's plan still hash-repartitions: -# Partial AggregateExec (group_completion_mode=Full) +# Partial AggregateExec (group_clustering_mode=Full) # -> RepartitionExec Hash([key, date_bin(...)]) -# -> FinalPartitioned AggregateExec (group_completion_mode=Full) +# -> FinalPartitioned AggregateExec (group_clustering_mode=Full) statement ok set datafusion.explain.physical_plan_only = true; @@ -86,7 +86,7 @@ physical_plan DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion # overlap across the two 60-minute streams. # # Today this is still Partial + hash RepartitionExec + Final, even though -# group_completion_mode=Full is already recognized. +# group_clustering_mode=Full is already recognized. ########## query TT @@ -97,9 +97,9 @@ GROUP BY key, time_bin; ---- physical_plan 01)ProjectionExec: expr=[key@0 as key, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1 as time_bin, sum(range_sorted_time_bin.value)@2 as sum(range_sorted_time_bin.value)] -02)--AggregateExec: mode=FinalPartitioned, gby=[key@0 as key, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)], aggr=[sum(range_sorted_time_bin.value)], group_completion_mode=Full +02)--AggregateExec: mode=FinalPartitioned, gby=[key@0 as key, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)], aggr=[sum(range_sorted_time_bin.value)], group_clustering_mode=Full 03)----RepartitionExec: partitioning=Hash([key@0, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1], 2), input_partitions=2, preserve_order=true, sort_exprs=key@0 ASC, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1 ASC -04)------AggregateExec: mode=Partial, gby=[key@0 as key, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }, timestamp@1) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)], aggr=[sum(range_sorted_time_bin.value)], group_completion_mode=Full +04)------AggregateExec: mode=Partial, gby=[key@0 as key, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }, timestamp@1) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)], aggr=[sum(range_sorted_time_bin.value)], group_clustering_mode=Full 05)--------FilterExec: col4@1 = a, projection=[key@0, timestamp@2, value@3] 06)----------DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch_range_partitioning/range_sorted_time_bin/part-0.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch_range_partitioning/range_sorted_time_bin/part-1.parquet]]}, projection=[key, col4, timestamp, value], output_ordering=[key@0 ASC, timestamp@2 ASC], output_partitioning=Range([timestamp@2 ASC], [(1704070800000000000)], 2), file_type=parquet, predicate=col4@4 = a, pruning_predicate=col4_null_count@2 != row_count@3 AND col4_min@0 <= a AND a <= col4_max@1, required_guarantees=[col4 in (a)] @@ -128,9 +128,9 @@ GROUP BY key, time_bin; ---- physical_plan 01)ProjectionExec: expr=[key@0 as key, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1 as time_bin, sum(range_sorted_time_bin.value)@2 as sum(range_sorted_time_bin.value)] -02)--AggregateExec: mode=FinalPartitioned, gby=[key@0 as key, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)], aggr=[sum(range_sorted_time_bin.value)], group_completion_mode=Full +02)--AggregateExec: mode=FinalPartitioned, gby=[key@0 as key, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)], aggr=[sum(range_sorted_time_bin.value)], group_clustering_mode=Full 03)----RepartitionExec: partitioning=Hash([key@0, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1], 2), input_partitions=2, preserve_order=true, sort_exprs=key@0 ASC, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1 ASC -04)------AggregateExec: mode=Partial, gby=[key@0 as key, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }, timestamp@1) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)], aggr=[sum(range_sorted_time_bin.value)], group_completion_mode=Full +04)------AggregateExec: mode=Partial, gby=[key@0 as key, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }, timestamp@1) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)], aggr=[sum(range_sorted_time_bin.value)], group_clustering_mode=Full 05)--------DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch_range_partitioning/range_sorted_time_bin/part-0.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch_range_partitioning/range_sorted_time_bin/part-1.parquet]]}, projection=[key, timestamp, value], output_ordering=[key@0 ASC, timestamp@1 ASC], output_partitioning=Range([timestamp@1 ASC], [(1704070800000000000)], 2), file_type=parquet query TPI diff --git a/datafusion/sqllogictest/test_files/repartition_subset_satisfaction.slt b/datafusion/sqllogictest/test_files/repartition_subset_satisfaction.slt index 5459fc9512ab6..09bf131e5e95f 100644 --- a/datafusion/sqllogictest/test_files/repartition_subset_satisfaction.slt +++ b/datafusion/sqllogictest/test_files/repartition_subset_satisfaction.slt @@ -161,9 +161,9 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST, time_bin@1 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1 as time_bin, count(Int64(1))@2 as count(*), avg(fact_table_ordered.value)@3 as avg(fact_table_ordered.value)] -03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], group_completion_mode=Full +03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], group_clustering_mode=Full 04)------RepartitionExec: partitioning=Hash([f_dkey@0, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1], 3), input_partitions=3, preserve_order=true, sort_exprs=f_dkey@0 ASC NULLS LAST, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1 ASC NULLS LAST -05)--------AggregateExec: mode=Partial, gby=[f_dkey@2 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@0) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], group_completion_mode=Full +05)--------AggregateExec: mode=Partial, gby=[f_dkey@2 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@0) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], group_clustering_mode=Full 06)----------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=C/data.parquet]]}, projection=[timestamp, value, f_dkey], output_ordering=[f_dkey@2 ASC NULLS LAST, timestamp@0 ASC NULLS LAST], output_partitioning=Hash([f_dkey@2], 3), file_type=parquet # Verify results without subset satisfaction @@ -203,7 +203,7 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST, time_bin@1 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1 as time_bin, count(Int64(1))@2 as count(*), avg(fact_table_ordered.value)@3 as avg(fact_table_ordered.value)] -03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@2 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@0) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], group_completion_mode=Full +03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@2 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@0) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], group_clustering_mode=Full 04)------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=C/data.parquet]]}, projection=[timestamp, value, f_dkey], output_ordering=[f_dkey@2 ASC NULLS LAST, timestamp@0 ASC NULLS LAST], output_partitioning=Hash([f_dkey@2], 3), file_type=parquet # Verify results match with subset satisfaction @@ -373,9 +373,9 @@ physical_plan 05)--------RepartitionExec: partitioning=Hash([env@0, time_bin@1], 3), input_partitions=3 06)----------AggregateExec: mode=Partial, gby=[env@1 as env, time_bin@0 as time_bin], aggr=[avg(a.max_bin_value)] 07)------------ProjectionExec: expr=[date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp)@1 as time_bin, env@2 as env, max(j.value)@3 as max_bin_value] -08)--------------AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp), env@2 as env], aggr=[max(j.value)], group_completion_mode=Partial([0, 1]) +08)--------------AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp), env@2 as env], aggr=[max(j.value)], group_clustering_mode=Partial([0, 1]) 09)----------------RepartitionExec: partitioning=Hash([f_dkey@0, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp)@1, env@2], 3), input_partitions=3, preserve_order=true, sort_exprs=f_dkey@0 ASC NULLS LAST, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp)@1 ASC NULLS LAST -10)------------------AggregateExec: mode=Partial, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@2) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp), env@1 as env], aggr=[max(j.value)], group_completion_mode=Partial([0, 1]) +10)------------------AggregateExec: mode=Partial, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@2) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp), env@1 as env], aggr=[max(j.value)], group_clustering_mode=Partial([0, 1]) 11)--------------------HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(d_dkey@1, f_dkey@2)], projection=[f_dkey@4, env@0, timestamp@2, value@3] 12)----------------------CoalescePartitionsExec 13)------------------------FilterExec: service@1 = log, projection=[env@0, d_dkey@2] @@ -470,7 +470,7 @@ physical_plan 05)--------RepartitionExec: partitioning=Hash([env@0, time_bin@1], 3), input_partitions=3 06)----------AggregateExec: mode=Partial, gby=[env@1 as env, time_bin@0 as time_bin], aggr=[avg(a.max_bin_value)] 07)------------ProjectionExec: expr=[date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp)@1 as time_bin, env@2 as env, max(j.value)@3 as max_bin_value] -08)--------------AggregateExec: mode=SinglePartitioned, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@2) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp), env@1 as env], aggr=[max(j.value)], group_completion_mode=Partial([0, 1]) +08)--------------AggregateExec: mode=SinglePartitioned, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@2) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp), env@1 as env], aggr=[max(j.value)], group_clustering_mode=Partial([0, 1]) 09)----------------HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(d_dkey@1, f_dkey@2)], projection=[f_dkey@4, env@0, timestamp@2, value@3] 10)------------------CoalescePartitionsExec 11)--------------------FilterExec: service@1 = log, projection=[env@0, d_dkey@2] diff --git a/datafusion/sqllogictest/test_files/sort_pushdown.slt b/datafusion/sqllogictest/test_files/sort_pushdown.slt index 20422dd95a614..da583f127c2b7 100644 --- a/datafusion/sqllogictest/test_files/sort_pushdown.slt +++ b/datafusion/sqllogictest/test_files/sort_pushdown.slt @@ -912,7 +912,7 @@ logical_plan 03)----TableScan: agg_expr_parquet projection=[x, y, v] physical_plan 01)SortExec: expr=[x@0 ASC NULLS LAST, agg_expr_parquet.y % Int64(2)@1 ASC NULLS LAST], preserve_partitioning=[false] -02)--AggregateExec: mode=Single, gby=[x@0 as x, CAST(y@1 AS Int64) % 2 as agg_expr_parquet.y % Int64(2)], aggr=[sum(agg_expr_parquet.v)], group_completion_mode=Partial([0]) +02)--AggregateExec: mode=Single, gby=[x@0 as x, CAST(y@1 AS Int64) % 2 as agg_expr_parquet.y % Int64(2)], aggr=[sum(agg_expr_parquet.v)], group_clustering_mode=Partial([0]) 03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/sort_pushdown/agg_expr_sorted.parquet]]}, projection=[x, y, v], output_ordering=[x@0 ASC NULLS LAST, y@1 ASC NULLS LAST], file_type=parquet # Expected output pattern from ORDER BY [x, bucket]: @@ -946,7 +946,7 @@ logical_plan 02)--Aggregate: groupBy=[[agg_expr_parquet.x, CAST(agg_expr_parquet.y AS Int64)]], aggr=[[sum(CAST(agg_expr_parquet.v AS Int64))]] 03)----TableScan: agg_expr_parquet projection=[x, y, v] physical_plan -01)AggregateExec: mode=Single, gby=[x@0 as x, CAST(y@1 AS Int64) as agg_expr_parquet.y], aggr=[sum(agg_expr_parquet.v)], group_completion_mode=Full +01)AggregateExec: mode=Single, gby=[x@0 as x, CAST(y@1 AS Int64) as agg_expr_parquet.y], aggr=[sum(agg_expr_parquet.v)], group_clustering_mode=Full 02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/sort_pushdown/agg_expr_sorted.parquet]]}, projection=[x, y, v], output_ordering=[x@0 ASC NULLS LAST, y@1 ASC NULLS LAST], file_type=parquet query III @@ -978,7 +978,7 @@ logical_plan 03)----TableScan: agg_expr_parquet projection=[x, v] physical_plan 01)SortExec: expr=[sum(agg_expr_parquet.v)@1 ASC NULLS LAST], preserve_partitioning=[false] -02)--AggregateExec: mode=Single, gby=[x@0 as x], aggr=[sum(agg_expr_parquet.v)], group_completion_mode=Full +02)--AggregateExec: mode=Single, gby=[x@0 as x], aggr=[sum(agg_expr_parquet.v)], group_clustering_mode=Full 03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/sort_pushdown/agg_expr_sorted.parquet]]}, projection=[x, v], output_ordering=[x@0 ASC NULLS LAST], file_type=parquet query II @@ -1034,7 +1034,7 @@ logical_plan 03)----TableScan: agg_expr_parquet projection=[x, v] physical_plan 01)SortExec: expr=[CAST(x@0 AS Int64) + 1 DESC], preserve_partitioning=[false] -02)--AggregateExec: mode=Single, gby=[x@0 as x], aggr=[sum(agg_expr_parquet.v)], group_completion_mode=Full +02)--AggregateExec: mode=Single, gby=[x@0 as x], aggr=[sum(agg_expr_parquet.v)], group_clustering_mode=Full 03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/sort_pushdown/agg_expr_sorted.parquet]]}, projection=[x, v], output_ordering=[x@0 ASC NULLS LAST], file_type=parquet query II @@ -1060,7 +1060,7 @@ logical_plan 03)----TableScan: agg_expr_parquet projection=[x, v] physical_plan 01)SortExec: expr=[2 * CAST(x@0 AS Int64) ASC NULLS LAST], preserve_partitioning=[false] -02)--AggregateExec: mode=Single, gby=[x@0 as x], aggr=[sum(agg_expr_parquet.v)], group_completion_mode=Full +02)--AggregateExec: mode=Single, gby=[x@0 as x], aggr=[sum(agg_expr_parquet.v)], group_clustering_mode=Full 03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/sort_pushdown/agg_expr_sorted.parquet]]}, projection=[x, v], output_ordering=[x@0 ASC NULLS LAST], file_type=parquet query II diff --git a/datafusion/sqllogictest/test_files/unnest.slt b/datafusion/sqllogictest/test_files/unnest.slt index c746adaa18e1c..39ed64a526755 100644 --- a/datafusion/sqllogictest/test_files/unnest.slt +++ b/datafusion/sqllogictest/test_files/unnest.slt @@ -988,9 +988,9 @@ logical_plan 08)--------------TableScan: range() projection=[value] physical_plan 01)ProjectionExec: expr=[array_agg(unnested.ar)@1 as array_agg(unnested.ar)] -02)--AggregateExec: mode=FinalPartitioned, gby=[generated_id@0 as generated_id], aggr=[array_agg(unnested.ar)], group_completion_mode=Full +02)--AggregateExec: mode=FinalPartitioned, gby=[generated_id@0 as generated_id], aggr=[array_agg(unnested.ar)], group_clustering_mode=Full 03)----RepartitionExec: partitioning=Hash([generated_id@0], 4), input_partitions=4, preserve_order=true, sort_exprs=generated_id@0 ASC NULLS LAST -04)------AggregateExec: mode=Partial, gby=[generated_id@0 as generated_id], aggr=[array_agg(unnested.ar)], group_completion_mode=Full +04)------AggregateExec: mode=Partial, gby=[generated_id@0 as generated_id], aggr=[array_agg(unnested.ar)], group_clustering_mode=Full 05)--------ProjectionExec: expr=[generated_id@0 as generated_id, __unnest_placeholder(make_array(range().value),depth=1)@1 as ar] 06)----------UnnestExec 07)------------ProjectionExec: expr=[row_number() ROWS BETWEEN UNBOUNDED PRECEDING AND UNBOUNDED FOLLOWING@1 as generated_id, make_array(value@0) as __unnest_placeholder(make_array(range().value))] diff --git a/datafusion/sqllogictest/test_files/window.slt b/datafusion/sqllogictest/test_files/window.slt index 36811abf0ad00..254fdf20e5124 100644 --- a/datafusion/sqllogictest/test_files/window.slt +++ b/datafusion/sqllogictest/test_files/window.slt @@ -357,7 +357,7 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [b@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[b@0 as b, max(d.a)@1 as max_a, max(d.seq)@2 as max(d.seq)] -03)----AggregateExec: mode=SinglePartitioned, gby=[b@2 as b], aggr=[max(d.a), max(d.seq)], group_completion_mode=Full +03)----AggregateExec: mode=SinglePartitioned, gby=[b@2 as b], aggr=[max(d.a), max(d.seq)], group_clustering_mode=Full 04)------ProjectionExec: expr=[row_number() PARTITION BY [s.b] ORDER BY [s.a ASC NULLS LAST] RANGE BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW@2 as seq, a@0 as a, b@1 as b] 05)--------BoundedWindowAggExec: wdw=[row_number() PARTITION BY [s.b] ORDER BY [s.a ASC NULLS LAST] RANGE BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW: Field { "row_number() PARTITION BY [s.b] ORDER BY [s.a ASC NULLS LAST] RANGE BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW": UInt64 }, frame: RANGE BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW], mode=[Sorted] 06)----------SortExec: expr=[b@1 ASC NULLS LAST, a@0 ASC NULLS LAST], preserve_partitioning=[true] diff --git a/docs/source/library-user-guide/upgrading/56.0.0.md b/docs/source/library-user-guide/upgrading/56.0.0.md index 8163ab3fc182d..be1a3007d94c1 100644 --- a/docs/source/library-user-guide/upgrading/56.0.0.md +++ b/docs/source/library-user-guide/upgrading/56.0.0.md @@ -75,25 +75,25 @@ The Minimum Supported Rust Version (MSRV) has been updated to [`1.95.0`]. [`1.95.0`]: https://releases.rs/docs/1.95.0/ -### Aggregate group-completion APIs +### Aggregate group-clustering APIs `AggregateExec::input_order_mode()` has been replaced by -`AggregateExec::group_completion_mode()`. It returns a `GroupCompletionMode`: +`AggregateExec::group_clustering_mode()`. It returns a `GroupClusteringMode`: `None`, `Partial(indices)`, or `Full`, corresponding to the previous aggregate modes `Linear`, `PartiallySorted(indices)`, and `Sorted`. -`AggregateExec::compute_properties` now accepts `&GroupCompletionMode` in place of +`AggregateExec::compute_properties` now accepts `&GroupClusteringMode` in place of `&InputOrderMode`. -`GroupCompletionMode` describes the input's group-clustering guarantees, which +`GroupClusteringMode` describes the input's group-clustering guarantees, which determine when groups can be emitted during aggregation. In `datafusion_physical_plan::aggregates::order`, `GroupOrdering`, `GroupOrderingPartial`, and `GroupOrderingFull` have been renamed to -`GroupCompletion`, `GroupCompletionPartial`, and `GroupCompletionFull`. -`GroupCompletion::try_new` accepts `&GroupCompletionMode`. +`GroupClustering`, `GroupClusteringPartial`, and `GroupClusteringFull`. +`GroupClustering::try_new` accepts `&GroupClusteringMode`. -Aggregate plans now display `group_completion_mode=Full` or -`group_completion_mode=Partial(indices)` instead of `ordering_mode=Sorted` or +Aggregate plans now display `group_clustering_mode=Full` or +`group_clustering_mode=Partial(indices)` instead of `ordering_mode=Sorted` or `ordering_mode=PartiallySorted(indices)`. ### Upgrade arrow/parquet to 60.0.0 and object_store to 0.14.2 From e31886e0ae4da9f900beb3dc6296253d9e84be77 Mon Sep 17 00:00:00 2001 From: "xavier.lee" Date: Fri, 4 Sep 2026 14:05:14 -0400 Subject: [PATCH 5/5] feat: add grouped equivalence properties --- datafusion/expr-common/src/sort_properties.rs | 44 +++- datafusion/expr/src/udf.rs | 21 +- datafusion/ffi/src/expr/expr_properties.rs | 3 + datafusion/ffi/src/tests/udf_udaf_udwf.rs | 7 +- datafusion/ffi/src/udf/mod.rs | 10 + datafusion/ffi/tests/ffi_udf.rs | 5 +- datafusion/functions/src/datetime/date_bin.rs | 28 ++- .../physical-expr/src/equivalence/grouping.rs | 203 +++++++++++++++++ .../physical-expr/src/equivalence/mod.rs | 200 ++++++++++++++++- .../src/equivalence/properties/mod.rs | 210 ++++++++++++++++-- .../physical-expr/src/expressions/cast.rs | 26 ++- .../physical-optimizer/src/join_selection.rs | 7 +- .../physical-plan/src/coalesce_partitions.rs | 33 +++ datafusion/physical-plan/src/limit.rs | 36 ++- .../physical-plan/src/repartition/mod.rs | 37 ++- .../src/sorts/sort_preserving_merge.rs | 24 ++ datafusion/physical-plan/src/test.rs | 55 ++++- 17 files changed, 908 insertions(+), 41 deletions(-) create mode 100644 datafusion/physical-expr/src/equivalence/grouping.rs diff --git a/datafusion/expr-common/src/sort_properties.rs b/datafusion/expr-common/src/sort_properties.rs index af417ed16d2d5..53dc8bd8c5568 100644 --- a/datafusion/expr-common/src/sort_properties.rs +++ b/datafusion/expr-common/src/sort_properties.rs @@ -37,16 +37,24 @@ use arrow::datatypes::DataType; pub enum SortProperties { /// Use the ordinary [`SortOptions`] struct to represent ordered data: Ordered(SortOptions), - // This alternative represents unordered data: + /// Within each partition, all rows with the same value for this expression + /// form one contiguous run. The runs may occur in any order. + Grouped, + /// This alternative represents unordered data: #[default] Unordered, - // Singleton is used for single-valued literal numbers: + /// Singleton is used for single-valued literal numbers: Singleton, } impl SortProperties { pub fn add(&self, rhs: &Self) -> Self { match (self, rhs) { + // Addition can collapse distinct values (for example through + // floating-point rounding), which may join non-adjacent groups. + (Self::Grouped, Self::Singleton) | (Self::Singleton, Self::Grouped) => { + Self::Unordered + } (Self::Singleton, _) => *rhs, (_, Self::Singleton) => *self, (Self::Ordered(lhs), Self::Ordered(rhs)) @@ -65,6 +73,11 @@ impl SortProperties { pub fn sub(&self, rhs: &Self) -> Self { match (self, rhs) { (Self::Singleton, Self::Singleton) => Self::Singleton, + // Subtraction can collapse distinct values (for example through + // floating-point rounding), which may join non-adjacent groups. + (Self::Grouped, Self::Singleton) | (Self::Singleton, Self::Grouped) => { + Self::Unordered + } (Self::Singleton, Self::Ordered(rhs)) => Self::Ordered(SortOptions { descending: !rhs.descending, nulls_first: rhs.nulls_first, @@ -89,6 +102,9 @@ impl SortProperties { descending: !rhs.descending, nulls_first: rhs.nulls_first, }), + // Comparisons can map several non-adjacent grouped values to the + // same boolean value, so they do not preserve grouping. + (Self::Grouped, Self::Singleton) => Self::Unordered, (_, Self::Singleton) => *self, (Self::Ordered(lhs), Self::Ordered(rhs)) if lhs.descending != rhs.descending @@ -147,6 +163,7 @@ mod sort_properties_test { const ASC_NL: SortProperties = ordered(false, false); const DESC_NF: SortProperties = ordered(true, true); const DESC_NL: SortProperties = ordered(true, false); + const GROUPED: SortProperties = SortProperties::Grouped; const UNORDERED: SortProperties = SortProperties::Unordered; const SINGLETON: SortProperties = SortProperties::Singleton; @@ -197,6 +214,27 @@ mod sort_properties_test { SINGLETON, SINGLETON, ), + ( + "add: may collapse grouped values", + SortProperties::add, + GROUPED, + SINGLETON, + UNORDERED, + ), + ( + "sub: may collapse grouped values", + SortProperties::sub, + GROUPED, + SINGLETON, + UNORDERED, + ), + ( + "comparison does not preserve grouping", + SortProperties::gt_or_gteq, + GROUPED, + SINGLETON, + UNORDERED, + ), // `and` keeps ASC NULLS LAST / DESC NULLS FIRST, `or` keeps ASC // NULLS FIRST / DESC NULLS LAST. Both are commutative. ( @@ -482,7 +520,7 @@ impl Neg for SortProperties { #[derive(Debug, Clone)] pub struct ExprProperties { /// Properties that describe the sorting behavior of the expression, - /// such as whether it is ordered, unordered, or a singleton value. + /// such as whether it is ordered, grouped, unordered, or a singleton value. pub sort_properties: SortProperties, /// A closed interval representing the range of possible values for /// the expression. Used to compute reliable bounds. diff --git a/datafusion/expr/src/udf.rs b/datafusion/expr/src/udf.rs index d937a54029398..fad9611287fb8 100644 --- a/datafusion/expr/src/udf.rs +++ b/datafusion/expr/src/udf.rs @@ -360,8 +360,20 @@ impl ScalarUDF { /// Calculates the [`SortProperties`] of this function based on its /// children's properties. + /// + /// [`SortProperties::Grouped`] is retained only when the implementation + /// also reports that the transformation is strictly order-preserving. A + /// many-to-one function can otherwise make an output value occur in + /// multiple non-adjacent runs. pub fn output_ordering(&self, inputs: &[ExprProperties]) -> Result { - self.inner.output_ordering(inputs) + let sort_properties = self.inner.output_ordering(inputs)?; + if sort_properties == SortProperties::Grouped + && !self.inner.strictly_order_preserving(inputs)? + { + Ok(SortProperties::Unordered) + } else { + Ok(sort_properties) + } } pub fn preserves_lex_ordering(&self, inputs: &[ExprProperties]) -> Result { @@ -957,7 +969,12 @@ pub trait ScalarUDFImpl: Debug + DynEq + DynHash + Send + Sync + Any { Ok(Some(vec![])) } - /// Calculates the [`SortProperties`] of this function based on its children's properties. + /// Calculates the [`SortProperties`] of this function based on its children's + /// properties. + /// + /// The [`ScalarUDF`] wrapper retains a [`SortProperties::Grouped`] result + /// only when [`Self::strictly_order_preserving`] also returns `true` for + /// the same inputs. fn output_ordering(&self, inputs: &[ExprProperties]) -> Result { if !self.preserves_lex_ordering(inputs)? { return Ok(SortProperties::Unordered); diff --git a/datafusion/ffi/src/expr/expr_properties.rs b/datafusion/ffi/src/expr/expr_properties.rs index 584f774c7b26e..0a9c0bc73aead 100644 --- a/datafusion/ffi/src/expr/expr_properties.rs +++ b/datafusion/ffi/src/expr/expr_properties.rs @@ -67,6 +67,7 @@ pub enum FFI_SortProperties { Ordered(FFI_SortOptions), Unordered, Singleton, + Grouped, } impl From<&SortProperties> for FFI_SortProperties { @@ -74,6 +75,7 @@ impl From<&SortProperties> for FFI_SortProperties { match value { SortProperties::Unordered => FFI_SortProperties::Unordered, SortProperties::Singleton => FFI_SortProperties::Singleton, + SortProperties::Grouped => FFI_SortProperties::Grouped, SortProperties::Ordered(o) => FFI_SortProperties::Ordered(o.into()), } } @@ -84,6 +86,7 @@ impl From<&FFI_SortProperties> for SortProperties { match value { FFI_SortProperties::Unordered => SortProperties::Unordered, FFI_SortProperties::Singleton => SortProperties::Singleton, + FFI_SortProperties::Grouped => SortProperties::Grouped, FFI_SortProperties::Ordered(o) => SortProperties::Ordered(o.into()), } } diff --git a/datafusion/ffi/src/tests/udf_udaf_udwf.rs b/datafusion/ffi/src/tests/udf_udaf_udwf.rs index ea7c03b5d6487..98f874a23fc75 100644 --- a/datafusion/ffi/src/tests/udf_udaf_udwf.rs +++ b/datafusion/ffi/src/tests/udf_udaf_udwf.rs @@ -21,7 +21,7 @@ use arrow_schema::DataType; use datafusion_catalog::TableFunctionImpl; use datafusion_common::ScalarValue; use datafusion_common::config::ConfigOptions; -use datafusion_expr::sort_properties::ExprProperties; +use datafusion_expr::sort_properties::{ExprProperties, SortProperties}; use datafusion_expr::{ AggregateUDF, ColumnarValue, ExpressionPlacement, ScalarFunctionArgs, ScalarUDF, ScalarUDFImpl, Signature, Volatility, WindowUDF, @@ -172,6 +172,11 @@ impl ScalarUDFImpl for PlacementUDF { &self, inputs: &[ExprProperties], ) -> datafusion_common::Result { + // This test-only sentinel verifies that the new `Grouped` variant + // survives the cross-library `ExprProperties` conversion. + if matches!(inputs, [input] if input.sort_properties == SortProperties::Grouped) { + return Ok(true); + } Ok(inputs.iter().all(|input| input.preserves_lex_ordering)) } } diff --git a/datafusion/ffi/src/udf/mod.rs b/datafusion/ffi/src/udf/mod.rs index d14614f1474a3..72e5b59c426e8 100644 --- a/datafusion/ffi/src/udf/mod.rs +++ b/datafusion/ffi/src/udf/mod.rs @@ -553,6 +553,7 @@ impl ScalarUDFImpl for ForeignScalarUDF { #[cfg(test)] mod tests { use super::*; + use datafusion_expr::sort_properties::SortProperties; #[derive(Debug, PartialEq, Eq, Hash)] struct PlacementUDF { @@ -594,6 +595,13 @@ mod tests { return internal_err!("preserves_lex_ordering requires an input"); } + // This test-only sentinel verifies that the new `Grouped` variant + // travels through the foreign path. + if matches!(inputs, [input] if input.sort_properties == SortProperties::Grouped) + { + return Ok(true); + } + Ok(inputs.iter().all(|input| input.preserves_lex_ordering)) } @@ -692,6 +700,8 @@ mod tests { .preserves_lex_ordering(&[preserves, does_not_preserve]) .unwrap() ); + let grouped = ExprProperties::new_unknown().with_order(SortProperties::Grouped); + assert!(foreign_udf.preserves_lex_ordering(&[grouped]).unwrap()); assert!(foreign_udf.preserves_lex_ordering(&[]).is_err()); let updated = foreign_udf diff --git a/datafusion/ffi/tests/ffi_udf.rs b/datafusion/ffi/tests/ffi_udf.rs index f97153526f733..2b83ccf013f28 100644 --- a/datafusion/ffi/tests/ffi_udf.rs +++ b/datafusion/ffi/tests/ffi_udf.rs @@ -27,7 +27,7 @@ mod tests { use datafusion::prelude::{SessionContext, col}; use datafusion_execution::config::SessionConfig; use datafusion_expr::lit; - use datafusion_expr::sort_properties::ExprProperties; + use datafusion_expr::sort_properties::{ExprProperties, SortProperties}; use datafusion_ffi::tests::create_record_batch; use datafusion_ffi::tests::utils::get_module; use std::sync::Arc; @@ -119,6 +119,9 @@ mod tests { assert!(foreign_func.preserves_lex_ordering(std::slice::from_ref(&preserves))?); assert!(!foreign_func.preserves_lex_ordering(&[preserves, does_not_preserve])?); + let grouped = ExprProperties::new_unknown().with_order(SortProperties::Grouped); + assert!(foreign_func.preserves_lex_ordering(&[grouped])?); + Ok(()) } diff --git a/datafusion/functions/src/datetime/date_bin.rs b/datafusion/functions/src/datetime/date_bin.rs index dc3da309dab82..58582ce75362e 100644 --- a/datafusion/functions/src/datetime/date_bin.rs +++ b/datafusion/functions/src/datetime/date_bin.rs @@ -822,12 +822,15 @@ mod tests { use crate::datetime::date_bin::{DateBinFunc, date_bin_nanos_interval}; use arrow::array::types::TimestampNanosecondType; use arrow::array::{Array, IntervalDayTimeArray, TimestampNanosecondArray}; + use arrow::compute::SortOptions; use arrow::compute::kernels::cast_utils::string_to_timestamp_nanos; use arrow::datatypes::{DataType, Field, FieldRef, TimeUnit}; use arrow_buffer::{IntervalDayTime, IntervalMonthDayNano}; use datafusion_common::{DataFusionError, ScalarValue}; - use datafusion_expr::{ColumnarValue, ScalarFunctionArgs, ScalarUDFImpl}; + use datafusion_expr::interval_arithmetic::Interval; + use datafusion_expr::sort_properties::{ExprProperties, SortProperties}; + use datafusion_expr::{ColumnarValue, ScalarFunctionArgs, ScalarUDF, ScalarUDFImpl}; use chrono::TimeDelta; use datafusion_common::config::ConfigOptions; @@ -877,6 +880,29 @@ mod tests { ); } + #[test] + fn date_bin_does_not_propagate_grouping() -> Result<(), DataFusionError> { + let udf = ScalarUDF::from(DateBinFunc::new()); + let step = ExprProperties::new_unknown().with_order(SortProperties::Singleton); + let timestamp = ExprProperties::new_unknown().with_range( + Interval::make_unbounded(&DataType::Timestamp(TimeUnit::Nanosecond, None))?, + ); + let grouped = timestamp.clone().with_order(SortProperties::Grouped); + + assert_eq!( + udf.output_ordering(&[step.clone(), grouped])?, + SortProperties::Unordered + ); + + let ordered = + timestamp.with_order(SortProperties::Ordered(SortOptions::default())); + assert_eq!( + udf.output_ordering(&[step, ordered])?, + SortProperties::Ordered(SortOptions::default()) + ); + Ok(()) + } + #[test] fn test_date_bin() { let return_field = &Arc::new(Field::new( diff --git a/datafusion/physical-expr/src/equivalence/grouping.rs b/datafusion/physical-expr/src/equivalence/grouping.rs new file mode 100644 index 0000000000000..ca3570ece05b8 --- /dev/null +++ b/datafusion/physical-expr/src/equivalence/grouping.rs @@ -0,0 +1,203 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +use std::fmt::Display; +use std::ops::Deref; +use std::sync::Arc; +use std::vec::IntoIter; + +use crate::PhysicalExpr; +use crate::expressions::with_new_schema; + +use arrow::datatypes::SchemaRef; +use datafusion_common::{HashSet, Result}; +use datafusion_physical_expr_common::physical_expr::format_physical_expr_list; + +/// A set of grouping tuples that are known to describe contiguous rows. +/// +/// Each entry is a complete grouping tuple. For example, `[a, b]` means that, +/// within each output partition, all rows having the same values for both `a` +/// and `b` occur in one contiguous run. The runs themselves may occur in any +/// order. Contiguity applies to the complete partition stream, not separately +/// to each record batch. +/// +/// Expression order within an entry is not significant: `[a, b]` and `[b, a]` +/// describe the same grouping. Entries do not imply properties for subsets; +/// `[a, b]` alone says nothing about whether all rows with the same `a` are +/// contiguous. +#[derive(Clone, Debug, Default)] +pub struct GroupingEquivalenceClass { + groupings: Vec>>, +} + +impl GroupingEquivalenceClass { + /// Clears all groupings in this equivalence class. + pub fn clear(&mut self) { + self.groupings.clear(); + } + + /// Creates a grouping equivalence class, discarding empty and duplicate + /// entries and duplicate expressions within each entry. + pub fn new( + groupings: impl IntoIterator>>, + ) -> Self { + let mut result = Self::default(); + result.add_groupings(groupings); + result + } + + /// Adds grouping tuples to this equivalence class. + pub fn add_groupings( + &mut self, + groupings: impl IntoIterator>>, + ) { + for grouping in groupings { + let mut seen = HashSet::new(); + let grouping = grouping + .into_iter() + .filter(|expr| seen.insert(Arc::clone(expr))) + .collect::>(); + if !grouping.is_empty() && !self.contains(&grouping) { + self.groupings.push(grouping); + } + } + } + + /// Returns whether this class contains the complete grouping tuple. + pub fn contains(&self, grouping: &[Arc]) -> bool { + self.groupings + .iter() + .any(|candidate| same_grouping(candidate, grouping)) + } + + /// Rewrites all expressions to reference an aligned schema. + pub fn with_new_schema(self, schema: &SchemaRef) -> Result { + let groupings = self.groupings.into_iter().map(|grouping| { + grouping + .into_iter() + .map(|expr| with_new_schema(expr, schema)) + .collect::>>() + }); + Ok(Self::new(groupings.collect::>>()?)) + } +} + +fn same_grouping(lhs: &[Arc], rhs: &[Arc]) -> bool { + lhs.iter() + .all(|lhs_expr| rhs.iter().any(|rhs_expr| lhs_expr.eq(rhs_expr))) + && rhs + .iter() + .all(|rhs_expr| lhs.iter().any(|lhs_expr| rhs_expr.eq(lhs_expr))) +} + +impl PartialEq for GroupingEquivalenceClass { + fn eq(&self, other: &Self) -> bool { + self.groupings.len() == other.groupings.len() + && self + .groupings + .iter() + .all(|grouping| other.contains(grouping)) + } +} + +impl Eq for GroupingEquivalenceClass {} + +impl Deref for GroupingEquivalenceClass { + type Target = [Vec>]; + + fn deref(&self) -> &Self::Target { + self.groupings.as_slice() + } +} + +impl From>>> for GroupingEquivalenceClass { + fn from(groupings: Vec>>) -> Self { + Self::new(groupings) + } +} + +/// Converts the grouping equivalence class into an iterator of complete +/// grouping tuples. +impl IntoIterator for GroupingEquivalenceClass { + type Item = Vec>; + type IntoIter = IntoIter; + + fn into_iter(self) -> Self::IntoIter { + self.groupings.into_iter() + } +} + +impl From for Vec>> { + fn from(geq_class: GroupingEquivalenceClass) -> Self { + geq_class.groupings + } +} + +impl Display for GroupingEquivalenceClass { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "[")?; + let mut groupings = self.groupings.iter(); + if let Some(grouping) = groupings.next() { + write!(f, "{}", format_physical_expr_list(grouping))?; + } + for grouping in groupings { + write!(f, ", {}", format_physical_expr_list(grouping))?; + } + write!(f, "]") + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::expressions::Column; + + #[test] + fn grouping_equality_ignores_expression_and_entry_order() { + let a = Arc::new(Column::new("a", 0)) as Arc; + let b = Arc::new(Column::new("b", 1)) as Arc; + let c = Arc::new(Column::new("c", 2)) as Arc; + + let lhs = GroupingEquivalenceClass::new([ + vec![Arc::clone(&a), Arc::clone(&b)], + vec![Arc::clone(&c)], + ]); + let rhs = GroupingEquivalenceClass::new([ + vec![Arc::clone(&c)], + vec![Arc::clone(&b), Arc::clone(&a)], + ]); + + assert_eq!(lhs, rhs); + } + + #[test] + fn grouping_deduplicates_entries_and_expressions() { + let a = Arc::new(Column::new("a", 0)) as Arc; + let b = Arc::new(Column::new("b", 1)) as Arc; + + let groupings = GroupingEquivalenceClass::new([ + vec![Arc::clone(&a), Arc::clone(&a), Arc::clone(&b)], + vec![Arc::clone(&b), Arc::clone(&a)], + vec![], + ]); + + assert_eq!(groupings.len(), 1); + assert_eq!(groupings[0].len(), 2); + assert!(groupings.contains(&[Arc::clone(&a), Arc::clone(&a), Arc::clone(&b),])); + assert_eq!(groupings.to_string(), "[[a@0, b@1]]"); + } +} diff --git a/datafusion/physical-expr/src/equivalence/mod.rs b/datafusion/physical-expr/src/equivalence/mod.rs index 64bb62901310f..ab4aacdd4c2fb 100644 --- a/datafusion/physical-expr/src/equivalence/mod.rs +++ b/datafusion/physical-expr/src/equivalence/mod.rs @@ -24,10 +24,12 @@ use arrow::compute::SortOptions; use datafusion_physical_expr_common::sort_expr::{LexOrdering, PhysicalSortExpr}; mod class; +mod grouping; mod ordering; mod properties; pub use class::{AcrossPartitions, ConstExpr, EquivalenceClass, EquivalenceGroup}; +pub use grouping::GroupingEquivalenceClass; pub use ordering::OrderingEquivalenceClass; // Re-export for backwards compatibility, we recommend importing from // datafusion_physical_expr::projection instead @@ -57,11 +59,12 @@ pub fn convert_to_orderings>>( #[cfg(test)] mod tests { use super::*; - use crate::expressions::{Column, col}; + use crate::expressions::{BinaryExpr, Column, cast, col, lit}; use crate::{LexRequirement, PhysicalSortExpr}; use arrow::datatypes::{DataType, Field, Schema, SchemaRef}; use datafusion_common::Result; + use datafusion_expr::Operator; use datafusion_physical_expr_common::sort_expr::PhysicalSortRequirement; /// Converts a string to a physical sort expression @@ -220,4 +223,199 @@ mod tests { Ok(()) } + + #[test] + fn grouping_satisfy_requires_the_complete_tuple() -> Result<()> { + let schema = create_test_schema()?; + let a = col("a", &schema)?; + let b = col("b", &schema)?; + let c = col("c", &schema)?; + let mut properties = EquivalenceProperties::new(schema); + properties.add_grouping([Arc::clone(&a), Arc::clone(&b)]); + + assert!(properties.grouping_satisfy([Arc::clone(&a), Arc::clone(&b)])?); + assert!(properties.grouping_satisfy([Arc::clone(&b), Arc::clone(&a)])?); + assert!(!properties.grouping_satisfy([Arc::clone(&a)])?); + assert!(!properties.grouping_satisfy([Arc::clone(&b)])?); + assert!(!properties.grouping_satisfy([a, b, c])?); + Ok(()) + } + + #[test] + fn ordering_implies_grouped_prefixes() -> Result<()> { + let schema = create_test_schema()?; + let a = col("a", &schema)?; + let b = col("b", &schema)?; + let c = col("c", &schema)?; + let properties = EquivalenceProperties::new_with_orderings( + schema, + [[ + PhysicalSortExpr::new_default(Arc::clone(&a)), + PhysicalSortExpr::new_default(Arc::clone(&b)), + PhysicalSortExpr::new_default(Arc::clone(&c)), + ]], + ); + + assert!(properties.grouping_satisfy([Arc::clone(&a)])?); + assert!(properties.grouping_satisfy([Arc::clone(&a), Arc::clone(&b)])?); + assert!(properties.grouping_satisfy([Arc::clone(&b), Arc::clone(&a)])?); + assert!(!properties.grouping_satisfy([Arc::clone(&b)])?); + assert!(!properties.grouping_satisfy([a, c])?); + Ok(()) + } + + #[test] + fn equivalent_orderings_can_jointly_satisfy_grouping() -> Result<()> { + let schema = create_test_schema()?; + let a = col("a", &schema)?; + let b = col("b", &schema)?; + let properties = EquivalenceProperties::new_with_orderings( + schema, + [ + [PhysicalSortExpr::new_default(Arc::clone(&a))], + [PhysicalSortExpr::new_default(Arc::clone(&b))], + ], + ); + + assert!(properties.grouping_satisfy([b, a])?); + Ok(()) + } + + #[test] + fn reorder_clears_explicit_grouping() -> Result<()> { + let schema = create_test_schema()?; + let a = col("a", &schema)?; + let b = col("b", &schema)?; + let ordering = [PhysicalSortExpr::new_default(Arc::clone(&b))]; + let mut properties = + EquivalenceProperties::new_with_orderings(schema, [ordering.clone()]); + properties.add_grouping([Arc::clone(&a)]); + + // The ordering is already satisfied, but a physical sort may still + // rearrange equal `b` values and separate an `a` group. + assert!(!properties.reorder(ordering)?); + assert!(!properties.grouping_satisfy([a])?); + assert!(properties.grouping_satisfy([b])?); + Ok(()) + } + + #[test] + fn grouping_normalizes_equivalences_and_constants() -> Result<()> { + let schema = create_test_schema()?; + let a = col("a", &schema)?; + let b = col("b", &schema)?; + let c = col("c", &schema)?; + let mut properties = EquivalenceProperties::new(schema); + properties.add_equal_conditions(Arc::clone(&a), Arc::clone(&c))?; + properties.add_constants([ConstExpr::from(Arc::clone(&a))])?; + properties.add_grouping([Arc::clone(&a), Arc::clone(&b)]); + + assert!(properties.grouping_satisfy([Arc::clone(&b)])?); + assert!(properties.grouping_satisfy([Arc::clone(&b), Arc::clone(&c)])?); + + properties.clear_per_partition_constants(); + assert!(!properties.grouping_satisfy([Arc::clone(&b)])?); + assert!(properties.grouping_satisfy([b, c])?); + Ok(()) + } + + #[test] + fn project_grouping_preserves_complete_tuple() -> Result<()> { + let schema = create_test_schema()?; + let a = col("a", &schema)?; + let b = col("b", &schema)?; + let c = col("c", &schema)?; + let mut properties = EquivalenceProperties::new(Arc::clone(&schema)); + properties.add_grouping([Arc::clone(&a), Arc::clone(&b)]); + + let output_schema = Arc::new(Schema::new(vec![ + Field::new("z", DataType::Int32, true), + Field::new("x", DataType::Int32, true), + Field::new("y", DataType::Int32, true), + ])); + let mapping = ProjectionMapping::try_new( + [ + (c, "z".to_string()), + (Arc::clone(&b), "x".to_string()), + (Arc::clone(&a), "y".to_string()), + ], + &schema, + )?; + let projected = properties.project(&mapping, Arc::clone(&output_schema)); + let x = col("x", &output_schema)?; + let y = col("y", &output_schema)?; + assert!(projected.grouping_satisfy([x, y])?); + Ok(()) + } + + #[test] + fn project_grouping_drops_incomplete_tuple() -> Result<()> { + let schema = create_test_schema()?; + let a = col("a", &schema)?; + let b = col("b", &schema)?; + let c = col("c", &schema)?; + let mut properties = EquivalenceProperties::new(Arc::clone(&schema)); + properties.add_grouping([Arc::clone(&a), b]); + + let output_schema = Arc::new(Schema::new(vec![ + Field::new("x", DataType::Int32, true), + Field::new("z", DataType::Int32, true), + ])); + let mapping = ProjectionMapping::try_new( + [(a, "x".to_string()), (c, "z".to_string())], + &schema, + )?; + let projected = properties.project(&mapping, Arc::clone(&output_schema)); + assert!(projected.geq_class().is_empty()); + assert!(!projected.grouping_satisfy([col("x", &output_schema)?])?); + Ok(()) + } + + #[test] + fn project_grouping_omits_constant_members() -> Result<()> { + let schema = create_test_schema()?; + let a = col("a", &schema)?; + let b = col("b", &schema)?; + let mut properties = EquivalenceProperties::new(Arc::clone(&schema)); + properties.add_constants([ConstExpr::from(Arc::clone(&a))])?; + properties.add_grouping([a, Arc::clone(&b)]); + + let output_schema = + Arc::new(Schema::new(vec![Field::new("x", DataType::Int32, true)])); + let mapping = ProjectionMapping::try_new([(b, "x".to_string())], &schema)?; + let projected = properties.project(&mapping, Arc::clone(&output_schema)); + + assert!(projected.grouping_satisfy([col("x", &output_schema)?])?); + Ok(()) + } + + #[test] + fn project_grouped_expression_requires_an_injective_mapping() -> Result<()> { + let schema = create_test_schema()?; + let a = col("a", &schema)?; + let mut properties = EquivalenceProperties::new(Arc::clone(&schema)); + properties.add_grouping([Arc::clone(&a)]); + + let widened = cast(Arc::clone(&a), &schema, DataType::Int64)?; + let output_schema = + Arc::new(Schema::new(vec![Field::new("x", DataType::Int64, true)])); + let mapping = ProjectionMapping::try_new([(widened, "x".to_string())], &schema)?; + let projected = properties.project(&mapping, Arc::clone(&output_schema)); + assert!(projected.grouping_satisfy([col("x", &output_schema)?])?); + + let greater_than_one = Arc::new(BinaryExpr::new(a, Operator::Gt, lit(1_i32))) + as Arc; + let output_schema = Arc::new(Schema::new(vec![Field::new( + "is_greater", + DataType::Boolean, + true, + )])); + let mapping = ProjectionMapping::try_new( + [(greater_than_one, "is_greater".to_string())], + &schema, + )?; + let projected = properties.project(&mapping, Arc::clone(&output_schema)); + assert!(projected.geq_class().is_empty()); + Ok(()) + } } diff --git a/datafusion/physical-expr/src/equivalence/properties/mod.rs b/datafusion/physical-expr/src/equivalence/properties/mod.rs index c68157fecbd8c..df58bba6d7dc3 100644 --- a/datafusion/physical-expr/src/equivalence/properties/mod.rs +++ b/datafusion/physical-expr/src/equivalence/properties/mod.rs @@ -31,7 +31,8 @@ use self::dependency::{ generate_dependency_orderings, referred_dependencies, }; use crate::equivalence::{ - AcrossPartitions, EquivalenceGroup, OrderingEquivalenceClass, ProjectionMapping, + AcrossPartitions, EquivalenceGroup, GroupingEquivalenceClass, + OrderingEquivalenceClass, ProjectionMapping, }; use crate::expressions::{Column, Literal, with_new_schema}; use crate::{ @@ -41,7 +42,7 @@ use crate::{ use arrow::datatypes::SchemaRef; use datafusion_common::tree_node::{Transformed, TransformedResult, TreeNode}; -use datafusion_common::{Constraint, Constraints, HashMap, Result, plan_err}; +use datafusion_common::{Constraint, Constraints, HashMap, HashSet, Result, plan_err}; use datafusion_expr::interval_arithmetic::Interval; use datafusion_expr::sort_properties::{ExprProperties, SortProperties}; use datafusion_physical_expr_common::sort_expr::options_compatible; @@ -53,6 +54,7 @@ use itertools::Itertools; /// `EquivalenceProperties` stores information about the output of a plan node /// that can be used to optimize the plan. Currently, it keeps track of: /// - Sort expressions (orderings), +/// - Complete expression tuples whose values are contiguous, /// - Equivalent expressions; i.e. expressions known to have the same value. /// - Constants expressions; i.e. expressions known to contain a single constant /// value. @@ -138,10 +140,16 @@ pub struct EquivalenceProperties { eq_group: EquivalenceGroup, /// Equivalent sort expressions (i.e. those define the same ordering). oeq_class: OrderingEquivalenceClass, + /// Complete expression tuples whose values occupy one contiguous run + /// in each output partition. + geq_class: GroupingEquivalenceClass, /// Cache storing equivalent sort expressions in normal form (i.e. without /// constants/duplicates and in standard form) and a map associating leading /// terms with full sort expressions. oeq_cache: OrderingEquivalenceCache, + /// Grouping expressions in normal form (i.e. without constants, + /// duplicates, or non-canonical equivalent expressions). + geq_cache: GroupingEquivalenceClass, /// Table constraints that factor in equivalence calculations. constraints: Constraints, /// Schema associated with this object. @@ -252,7 +260,9 @@ impl EquivalenceProperties { Self { eq_group: EquivalenceGroup::default(), oeq_class: OrderingEquivalenceClass::default(), + geq_class: GroupingEquivalenceClass::default(), oeq_cache: OrderingEquivalenceCache::default(), + geq_cache: GroupingEquivalenceClass::default(), constraints: Constraints::default(), schema, } @@ -285,6 +295,8 @@ impl EquivalenceProperties { Self { oeq_cache: OrderingEquivalenceCache::new(normal_orderings), oeq_class, + geq_class: GroupingEquivalenceClass::default(), + geq_cache: GroupingEquivalenceClass::default(), eq_group, constraints: Constraints::default(), schema, @@ -301,6 +313,11 @@ impl EquivalenceProperties { &self.oeq_class } + /// Returns a reference to the grouping equivalence class within. + pub fn geq_class(&self) -> &GroupingEquivalenceClass { + &self.geq_class + } + /// Returns a reference to the equivalence group within. pub fn eq_group(&self) -> &EquivalenceGroup { &self.eq_group @@ -336,6 +353,7 @@ impl EquivalenceProperties { self.constraints.extend(other.constraints); self.add_equivalence_group(other.eq_group)?; self.add_orderings(other.oeq_class); + self.add_groupings(other.geq_class); Ok(self) } @@ -346,6 +364,13 @@ impl EquivalenceProperties { self.oeq_cache.clear(); } + /// Clears grouping information invalidated by an operation that changes + /// row sequence or combines input partitions. + pub fn clear_groupings(&mut self) { + self.geq_class.clear(); + self.geq_cache.clear(); + } + /// Removes constant expressions that may change across partitions. /// This method should be used when merging data from different partitions. pub fn clear_per_partition_constants(&mut self) { @@ -357,6 +382,7 @@ impl EquivalenceProperties { .cloned() .map(|o| self.eq_group.normalize_sort_exprs(o)); self.oeq_cache = OrderingEquivalenceCache::new(normal_orderings); + self.update_geq_cache(); } } @@ -387,6 +413,61 @@ impl EquivalenceProperties { self.add_orderings(std::iter::once(ordering)); } + /// Adds complete grouping tuples whose values are contiguous within + /// every output partition. + /// + /// This is a correctness assertion. DataFusion does not verify the row + /// layout, and consumers may act on a tuple as soon as its values change. + /// Callers must therefore only add tuples that hold for every output + /// partition. + /// + /// Expression order within a tuple is not significant. A tuple only + /// describes its complete set of expressions: grouping `[a, b]` does not + /// imply grouping `[a]` or `[b]`. + pub fn add_groupings( + &mut self, + groupings: impl IntoIterator>>, + ) { + for grouping in GroupingEquivalenceClass::new(groupings) { + let normal_grouping = self.normalize_grouping(grouping.iter().cloned()); + self.geq_class.add_groupings(std::iter::once(grouping)); + if !normal_grouping.is_empty() { + self.geq_cache + .add_groupings(std::iter::once(normal_grouping)); + } + } + } + + /// Adds one complete grouping tuple. + pub fn add_grouping( + &mut self, + grouping: impl IntoIterator>, + ) { + self.add_groupings(std::iter::once(grouping)); + } + + fn normalize_grouping( + &self, + grouping: impl IntoIterator>, + ) -> Vec> { + let mut seen = HashSet::new(); + grouping + .into_iter() + .map(|expr| self.eq_group.normalize_expr(expr)) + .filter(|expr| self.eq_group.is_expr_constant(expr).is_none()) + .filter(|expr| seen.insert(Arc::clone(expr))) + .collect() + } + + fn update_geq_cache(&mut self) { + let groupings = self + .geq_class + .iter() + .map(|grouping| self.normalize_grouping(grouping.iter().cloned())) + .collect::>(); + self.geq_cache = GroupingEquivalenceClass::new(groupings); + } + fn update_oeq_cache(&mut self) -> Result<()> { // Renormalize orderings if the equivalence group changes: let normal_cls = mem::take(&mut self.oeq_cache.normal_cls); @@ -412,6 +493,7 @@ impl EquivalenceProperties { if !other_eq_group.is_empty() { self.eq_group.extend(other_eq_group); self.update_oeq_cache()?; + self.update_geq_cache(); } Ok(()) } @@ -428,6 +510,11 @@ impl EquivalenceProperties { .into() } + /// Returns the grouping equivalence class within in normal form. + pub fn normalized_geq_class(&self) -> GroupingEquivalenceClass { + self.geq_cache.clone() + } + /// Adds a new equality condition into the existing equivalence group. /// If the given equality defines a new equivalence class, adds this new /// equivalence class to the equivalence group. @@ -441,6 +528,7 @@ impl EquivalenceProperties { self.update_oeq_cache()?; } self.update_oeq_cache()?; + self.update_geq_cache(); Ok(()) } @@ -463,6 +551,7 @@ impl EquivalenceProperties { }); self.oeq_cache.normal_cls = OrderingEquivalenceClass::new(normal_orderings); self.oeq_cache.update_map(); + self.update_geq_cache(); // Discover any new orderings based on the constants: let leading_exprs: Vec<_> = self.oeq_cache.leading_map.keys().cloned().collect(); for expr in leading_exprs { @@ -539,15 +628,21 @@ impl EquivalenceProperties { Ok(()) } - /// Updates the ordering equivalence class within assuming that the table - /// is re-sorted according to the argument `ordering`, and returns whether - /// this operation resulted in any change. Note that equivalence classes - /// (and constants) do not change as they are unaffected by a re-sort. If - /// the given ordering is already satisfied, the function does nothing. + /// Updates the ordering equivalence class assuming that the table is + /// re-sorted according to `ordering`, and returns whether the ordering + /// class changed. Equivalence classes and constants are unaffected by a + /// re-sort. Explicit grouping assertions are cleared because a sort may + /// rearrange rows that compare equally, even when `ordering` was already + /// satisfied. pub fn reorder( &mut self, ordering: impl IntoIterator, ) -> Result { + // A sort may reorder rows that compare equally under `ordering`, so + // explicit grouping assertions do not necessarily survive even when + // the requested ordering is already satisfied. Groupings implied by + // the output ordering remain discoverable through `oeq_class`. + self.clear_groupings(); let (ordering, ordering_tee) = ordering.into_iter().tee(); // First, standardize the given ordering: let Some(normal_ordering) = self.normalize_sort_exprs(ordering) else { @@ -585,6 +680,27 @@ impl EquivalenceProperties { LexRequirement::new(self.eq_group.normalize_sort_requirements(sort_reqs)) } + /// Returns whether the given complete expression tuple is known to be + /// grouped within every output partition. + /// + /// A tuple is grouped when all rows with the same values for the complete + /// tuple occur in one contiguous run. The order of expressions within a + /// grouping tuple does not matter: `[a, b]` and `[b, a]` describe the same + /// groups. A lexicographical ordering also satisfies grouping for each of + /// its prefixes, regardless of sort direction. + pub fn grouping_satisfy( + &self, + given: impl IntoIterator>, + ) -> Result { + let normal_grouping = self.normalize_grouping(given); + if normal_grouping.is_empty() || self.geq_cache.contains(&normal_grouping) { + return Ok(true); + } + + let (_, indices) = self.find_longest_permutation(&normal_grouping)?; + Ok(indices.len() == normal_grouping.len()) + } + /// Iteratively checks whether the given ordering is satisfied by any of /// the existing orderings. See [`Self::ordering_satisfy_requirement`] for /// more details and examples. @@ -658,7 +774,7 @@ impl EquivalenceProperties { }), // Singleton expressions satisfy any requirement. SortProperties::Singleton => true, - SortProperties::Unordered => false, + SortProperties::Grouped | SortProperties::Unordered => false, }; if !satisfy { return Ok(false); @@ -747,7 +863,7 @@ impl EquivalenceProperties { ), // Singleton expressions satisfy any ordering. SortProperties::Singleton => true, - SortProperties::Unordered => false, + SortProperties::Grouped | SortProperties::Unordered => false, }; if !satisfy { // As soon as one sort expression is unsatisfied, return how @@ -1184,6 +1300,34 @@ impl EquivalenceProperties { orderings.chain(projected_orderings).collect() } + /// Projects grouping tuples through `mapping`, dropping a tuple unless all + /// of its expressions can be represented by the projection. + fn projected_groupings( + &self, + mapping: &ProjectionMapping, + ) -> Vec>> { + let mut groupings = self + .geq_cache + .iter() + .filter_map(|grouping| { + self.project_expressions(grouping.iter(), mapping) + .collect::>>() + }) + .collect::>(); + + // A projection expression that safely preserves a one-expression + // grouping establishes the same property for its output column. + for (source, targets) in mapping.iter() { + if self.get_expr_properties(Arc::clone(source)).sort_properties + == SortProperties::Grouped + { + groupings + .extend(targets.iter().map(|(target, _)| vec![Arc::clone(target)])); + } + } + groupings + } + /// Projects constraints according to the given projection mapping. /// /// This function takes a projection mapping and extracts column indices of @@ -1227,6 +1371,8 @@ impl EquivalenceProperties { /// preserved through `c + 1` but dropped through `abs(c)`. Orderings /// implied by the mapping are also derived, e.g. an ordering on `a + b` /// yields one on the projected `a_new + b_new`. + /// - Groupings: a complete grouping tuple is carried only when every + /// expression in the tuple can be represented in the output. /// - Equivalence group: each class is re-expressed on the output columns. /// - Constraints: projected onto the surviving column indices. /// @@ -1278,17 +1424,22 @@ impl EquivalenceProperties { ) -> Self { let orderings = self.projected_orderings(mapping, self.oeq_cache.normal_cls.clone()); + let groupings = self.projected_groupings(mapping); let normal_orderings = orderings .iter() .cloned() .map(|o| eq_group.normalize_sort_exprs(o)); - Self { + let mut result = Self { oeq_cache: OrderingEquivalenceCache::new(normal_orderings), oeq_class: OrderingEquivalenceClass::new(orderings), + geq_class: GroupingEquivalenceClass::default(), + geq_cache: GroupingEquivalenceClass::default(), constraints: self.projected_constraints(mapping).unwrap_or_default(), schema: output_schema, eq_group, - } + }; + result.add_groupings(groupings); + result } /// Returns the longest (potentially partial) permutation satisfying the @@ -1335,7 +1486,7 @@ impl EquivalenceProperties { let expr = Arc::clone(&exprs[idx]); Some((PhysicalSortExpr::new_default(expr), idx)) } - SortProperties::Unordered => None, + SortProperties::Grouped | SortProperties::Unordered => None, } }) .collect::>(); @@ -1468,6 +1619,11 @@ impl EquivalenceProperties { self.oeq_class = self.oeq_class.with_new_schema(&schema)?; self.oeq_cache.normal_cls = self.oeq_cache.normal_cls.with_new_schema(&schema)?; + // Rewrite grouping expressions according to new schema and rebuild + // their normalized form against the rewritten equivalence group. + self.geq_class = self.geq_class.with_new_schema(&schema)?; + self.update_geq_cache(); + // Update the schema: self.schema = schema; @@ -1485,21 +1641,28 @@ impl From for OrderingEquivalenceClass { /// /// Format: /// ```text -/// order: [[b@1 ASC NULLS LAST]], eq: [{members: [a@0], constant: (heterogeneous)}] +/// order: [[b@1 ASC NULLS LAST]], group: [[a@0, b@1]], eq: [{members: [a@0], constant: (heterogeneous)}] /// ``` impl Display for EquivalenceProperties { fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { let empty_eq_group = self.eq_group.is_empty(); let empty_oeq_class = self.oeq_class.is_empty(); - if empty_oeq_class && empty_eq_group { - write!(f, "No properties")?; - } else if !empty_oeq_class { + let empty_geq_class = self.geq_class.is_empty(); + if empty_oeq_class && empty_geq_class && empty_eq_group { + return write!(f, "No properties"); + } + + let mut separator = ""; + if !empty_oeq_class { write!(f, "order: {}", self.oeq_class)?; - if !empty_eq_group { - write!(f, ", eq: {}", self.eq_group)?; - } - } else { - write!(f, "eq: {}", self.eq_group)?; + separator = ", "; + } + if !empty_geq_class { + write!(f, "{separator}group: {}", self.geq_class)?; + separator = ", "; + } + if !empty_eq_group { + write!(f, "{separator}eq: {}", self.eq_group)?; } Ok(()) } @@ -1556,6 +1719,11 @@ fn update_properties( node.data.sort_properties = SortProperties::Singleton; } else if let Some(options) = oeq_class.get_options(&normal_expr) { node.data.sort_properties = SortProperties::Ordered(options); + } else if eq_properties + .geq_cache + .contains(std::slice::from_ref(&normal_expr)) + { + node.data.sort_properties = SortProperties::Grouped; } Ok(Transformed::yes(node)) } diff --git a/datafusion/physical-expr/src/expressions/cast.rs b/datafusion/physical-expr/src/expressions/cast.rs index 792687fe6acaa..3c13cbcf00a66 100644 --- a/datafusion/physical-expr/src/expressions/cast.rs +++ b/datafusion/physical-expr/src/expressions/cast.rs @@ -33,7 +33,7 @@ use datafusion_common::nested_struct::{ use datafusion_common::{Result, not_impl_err}; use datafusion_expr_common::columnar_value::ColumnarValue; use datafusion_expr_common::interval_arithmetic::Interval; -use datafusion_expr_common::sort_properties::ExprProperties; +use datafusion_expr_common::sort_properties::{ExprProperties, SortProperties}; const DEFAULT_CAST_OPTIONS: CastOptions<'static> = CastOptions { safe: false, @@ -341,12 +341,16 @@ pub(crate) fn cast_expr_properties( if bigger_cast || (!null_on_failure && is_order_preserving_cast(&source_type, target_type)) { + let strictly_order_preserving = child.strictly_order_preserving && bigger_cast; + let sort_properties = match child.sort_properties { + SortProperties::Grouped if !bigger_cast => SortProperties::Unordered, + sort_properties => sort_properties, + }; Ok(child .clone() + .with_order(sort_properties) .with_range(unbounded) - .with_strictly_order_preserving( - child.strictly_order_preserving && bigger_cast, - )) + .with_strictly_order_preserving(strictly_order_preserving)) } else { Ok(ExprProperties::new_unknown().with_range(unbounded)) } @@ -2094,6 +2098,20 @@ mod tests { assert!(!CastExpr::check_bigger_cast(&Int64, &UInt64)); assert!(!CastExpr::check_bigger_cast(&Int8, &UInt16)); } + + #[test] + fn grouped_cast_requires_an_injective_conversion() -> Result<()> { + let grouped = ExprProperties::new_unknown() + .with_order(SortProperties::Grouped) + .with_range(Interval::make_unbounded(&Int32)?); + + let widened = cast_expr_properties(&grouped, &Int64, false)?; + assert_eq!(widened.sort_properties, SortProperties::Grouped); + + let narrowed = cast_expr_properties(&grouped, &Int8, false)?; + assert_eq!(narrowed.sort_properties, SortProperties::Unordered); + Ok(()) + } } /// Tests for the `try_to_proto` / `try_from_proto` hooks. diff --git a/datafusion/physical-optimizer/src/join_selection.rs b/datafusion/physical-optimizer/src/join_selection.rs index 2b342bece7040..a8a5ee3749579 100644 --- a/datafusion/physical-optimizer/src/join_selection.rs +++ b/datafusion/physical-optimizer/src/join_selection.rs @@ -441,8 +441,11 @@ fn hash_join_convert_symmetric_subrule( let name = schema.field(*index).name(); let col = Arc::new(Column::new(name, *index)) as _; // Check if the column is ordered. - equivalence.get_expr_properties(col).sort_properties - != SortProperties::Unordered + matches!( + equivalence.get_expr_properties(col).sort_properties, + SortProperties::Ordered(_) + | SortProperties::Singleton + ) }, ) }) diff --git a/datafusion/physical-plan/src/coalesce_partitions.rs b/datafusion/physical-plan/src/coalesce_partitions.rs index 9e3811e0ada76..4ce40ce43c964 100644 --- a/datafusion/physical-plan/src/coalesce_partitions.rs +++ b/datafusion/physical-plan/src/coalesce_partitions.rs @@ -98,6 +98,9 @@ impl CoalescePartitionsExec { // Coalescing partitions loses existing orderings: let mut eq_properties = input.equivalence_properties().clone(); eq_properties.clear_orderings(); + if input_partitions > 1 { + eq_properties.clear_groupings(); + } eq_properties.clear_per_partition_constants(); PlanProperties::new( eq_properties, // Equivalence Properties @@ -486,9 +489,39 @@ mod tests { use arrow::array::RecordBatch; use arrow::datatypes::{DataType, Field, Schema}; + use datafusion_physical_expr::expressions::col; use futures::FutureExt; + #[test] + fn grouping_is_cleared_only_when_partitions_are_merged() -> Result<()> { + let input = test::mem_exec(2); + let grouping = col("i", &input.schema())?; + let input = Arc::new(input.try_with_grouping_information(vec![vec![grouping]])?); + let merge = CoalescePartitionsExec::new(input); + assert!( + merge + .properties() + .equivalence_properties() + .geq_class() + .is_empty() + ); + + let input = test::mem_exec(1); + let grouping = col("i", &input.schema())?; + let input = Arc::new(input.try_with_grouping_information(vec![vec![grouping]])?); + let merge = CoalescePartitionsExec::new(input); + assert_eq!( + merge + .properties() + .equivalence_properties() + .geq_class() + .len(), + 1 + ); + Ok(()) + } + #[tokio::test] async fn merge() -> Result<()> { let task_ctx = Arc::new(TaskContext::default()); diff --git a/datafusion/physical-plan/src/limit.rs b/datafusion/physical-plan/src/limit.rs index ef504a40ea832..a7655c4febe9f 100644 --- a/datafusion/physical-plan/src/limit.rs +++ b/datafusion/physical-plan/src/limit.rs @@ -107,9 +107,13 @@ impl GlobalLimitExec { /// This function creates the cache object that stores the plan properties such as schema, equivalence properties, ordering, partitioning, etc. fn compute_properties(input: &Arc) -> PlanProperties { + let mut eq_properties = input.equivalence_properties().clone(); + if input.output_partitioning().partition_count() > 1 { + eq_properties.clear_groupings(); + } PlanProperties::new( - input.equivalence_properties().clone(), // Equivalence Properties - Partitioning::UnknownPartitioning(1), // Output Partitioning + eq_properties, // Equivalence Properties + Partitioning::UnknownPartitioning(1), // Output Partitioning input.pipeline_behavior(), // Limit operations are always bounded since they output a finite number of rows Boundedness::Bounded, @@ -787,6 +791,34 @@ mod tests { use datafusion_physical_expr::expressions::col; use datafusion_physical_expr::{PhysicalExpr, PhysicalSortExpr}; + #[test] + fn limits_preserve_grouping_without_merging_partitions() -> Result<()> { + let input = test::mem_exec(2); + let grouping = col("i", &input.schema())?; + let input: Arc = + Arc::new(input.try_with_grouping_information(vec![vec![grouping]])?); + + let local = LocalLimitExec::new(Arc::clone(&input), 10); + assert_eq!( + local + .properties() + .equivalence_properties() + .geq_class() + .len(), + 1 + ); + + let global = GlobalLimitExec::new(input, 0, Some(10)); + assert!( + global + .properties() + .equivalence_properties() + .geq_class() + .is_empty() + ); + Ok(()) + } + #[tokio::test] async fn limit() -> Result<()> { let task_ctx = Arc::new(TaskContext::default()); diff --git a/datafusion/physical-plan/src/repartition/mod.rs b/datafusion/physical-plan/src/repartition/mod.rs index 3df29b577cd1a..1d36e4e2efd26 100644 --- a/datafusion/physical-plan/src/repartition/mod.rs +++ b/datafusion/physical-plan/src/repartition/mod.rs @@ -2199,8 +2199,10 @@ impl RepartitionExec { eq_properties.clear_orderings(); } // When there are more than one input partitions, they will be fused at the output. - // Therefore, remove per partition constants. + // Therefore, remove per-partition properties that may not hold after + // rows from different inputs are combined. if input.output_partitioning().partition_count() > 1 { + eq_properties.clear_groupings(); eq_properties.clear_per_partition_constants(); } eq_properties @@ -2666,6 +2668,39 @@ mod tests { }; use insta::assert_snapshot; + #[test] + fn grouping_is_cleared_only_when_input_partitions_are_merged() -> Result<()> { + let input = crate::test::mem_exec(2); + let grouping = col("i", &input.schema())?; + let input: Arc = + Arc::new(input.try_with_grouping_information(vec![vec![grouping]])?); + let repartition = + RepartitionExec::try_new(input, Partitioning::RoundRobinBatch(2))?; + assert!( + repartition + .properties() + .equivalence_properties() + .geq_class() + .is_empty() + ); + + let input = crate::test::mem_exec(1); + let grouping = col("i", &input.schema())?; + let input: Arc = + Arc::new(input.try_with_grouping_information(vec![vec![grouping]])?); + let repartition = + RepartitionExec::try_new(input, Partitioning::RoundRobinBatch(2))?; + assert_eq!( + repartition + .properties() + .equivalence_properties() + .geq_class() + .len(), + 1 + ); + Ok(()) + } + #[derive(Debug)] struct UnboundedTestPartition { schema: SchemaRef, diff --git a/datafusion/physical-plan/src/sorts/sort_preserving_merge.rs b/datafusion/physical-plan/src/sorts/sort_preserving_merge.rs index 7c924ff32bb04..f6901d59c594a 100644 --- a/datafusion/physical-plan/src/sorts/sort_preserving_merge.rs +++ b/datafusion/physical-plan/src/sorts/sort_preserving_merge.rs @@ -174,6 +174,9 @@ impl SortPreservingMergeExec { }; let mut eq_properties = input.equivalence_properties().clone(); + if input_partitions > 1 { + eq_properties.clear_groupings(); + } eq_properties.clear_per_partition_constants(); eq_properties.add_ordering(ordering); PlanProperties::new( @@ -901,6 +904,27 @@ mod tests { use insta::assert_snapshot; use tokio::time::timeout; + #[test] + fn merging_sorted_partitions_clears_explicit_grouping() -> Result<()> { + let input = test::mem_exec(2); + let grouping = col("i", &input.schema())?; + let ordering: LexOrdering = + [PhysicalSortExpr::new_default(Arc::clone(&grouping))].into(); + let input = input + .try_with_sort_information(vec![ordering.clone()])? + .try_with_grouping_information(vec![vec![grouping]])?; + let merge = SortPreservingMergeExec::new(ordering, Arc::new(input)); + + assert!( + merge + .properties() + .equivalence_properties() + .geq_class() + .is_empty() + ); + Ok(()) + } + // The number in the function is highly related to the memory limit we are testing // any change of the constant should be aware of fn generate_task_ctx_for_round_robin_tie_breaker( diff --git a/datafusion/physical-plan/src/test.rs b/datafusion/physical-plan/src/test.rs index b38a46d160755..3103196675f86 100644 --- a/datafusion/physical-plan/src/test.rs +++ b/datafusion/physical-plan/src/test.rs @@ -73,6 +73,8 @@ pub struct TestMemoryExec { projection: Option>, /// Sort information: one or more equivalent orderings sort_information: Vec, + /// Complete expression tuples whose values are contiguous. + grouping_information: Vec>>, /// if partition sizes should be displayed show_sizes: bool, /// The maximum number of records to read from this plan. If `None`, @@ -233,10 +235,12 @@ impl TestMemoryExec { } fn eq_properties(&self) -> EquivalenceProperties { - EquivalenceProperties::new_with_orderings( + let mut properties = EquivalenceProperties::new_with_orderings( Arc::clone(&self.projected_schema), self.sort_information.clone(), - ) + ); + properties.add_groupings(self.grouping_information.clone()); + properties } fn statistics_inner(&self) -> Result { @@ -268,6 +272,7 @@ impl TestMemoryExec { projected_schema, projection, sort_information: vec![], + grouping_information: vec![], show_sizes: true, fetch: None, }) @@ -363,6 +368,52 @@ impl TestMemoryExec { Ok(self) } + /// Adds grouping information to this source. + pub fn try_with_grouping_information( + mut self, + mut grouping_information: Vec>>, + ) -> Result { + // All grouping expressions must refer to the original schema. + let fields = self.schema.fields(); + let ambiguous_column = grouping_information + .iter() + .flatten() + .flat_map(collect_columns) + .find(|col| { + fields + .get(col.index()) + .map(|field| field.name() != col.name()) + .unwrap_or(true) + }); + assert_or_internal_err!( + ambiguous_column.is_none(), + "Column {:?} is not found in the original schema of the TestMemoryExec", + ambiguous_column.as_ref().unwrap() + ); + + if let Some(projection) = &self.projection { + let base_schema = self.original_schema(); + let proj_exprs = projection.iter().map(|idx| { + let name = base_schema.field(*idx).name(); + (Arc::new(Column::new(name, *idx)) as _, name.to_string()) + }); + let projection_mapping = + ProjectionMapping::try_new(proj_exprs, &base_schema)?; + let mut base_eqp = EquivalenceProperties::new(base_schema); + base_eqp.add_groupings(grouping_information); + grouping_information = base_eqp + .project(&projection_mapping, Arc::clone(&self.projected_schema)) + .geq_class() + .iter() + .cloned() + .collect(); + } + + self.grouping_information = grouping_information; + self.cache = Arc::new(self.compute_properties()); + Ok(self) + } + /// Arc clone of ref to original schema pub fn original_schema(&self) -> SchemaRef { Arc::clone(&self.schema)