diff --git a/datafusion/core/tests/dataframe/mod.rs b/datafusion/core/tests/dataframe/mod.rs index 43ceee3444ced..d011a25273fdf 100644 --- a/datafusion/core/tests/dataframe/mod.rs +++ b/datafusion/core/tests/dataframe/mod.rs @@ -3422,9 +3422,9 @@ async fn union_with_mix_of_presorted_and_explicitly_resorted_inputs_with_reparti assert_snapshot!( union_with_mix_of_presorted_and_explicitly_resorted_inputs_impl(false).await?, @r" - AggregateExec: mode=Final, gby=[id@0 as id], aggr=[], ordering_mode=Sorted + AggregateExec: mode=Final, gby=[id@0 as id], aggr=[], group_clustering_mode=Full SortPreservingMergeExec: [id@0 ASC NULLS LAST] - AggregateExec: mode=Partial, gby=[id@0 as id], aggr=[], ordering_mode=Sorted + AggregateExec: mode=Partial, gby=[id@0 as id], aggr=[], group_clustering_mode=Full UnionExec DataSourceExec: file_groups={1 group: [[{testdata}/alltypes_tiny_pages.parquet]]}, projection=[id], output_ordering=[id@0 ASC NULLS LAST], file_type=parquet SortExec: expr=[id@0 ASC NULLS LAST], preserve_partitioning=[false] @@ -3440,9 +3440,9 @@ async fn union_with_mix_of_presorted_and_explicitly_resorted_inputs_with_reparti assert_snapshot!( union_with_mix_of_presorted_and_explicitly_resorted_inputs_impl(true).await?, @r" - AggregateExec: mode=Final, gby=[id@0 as id], aggr=[], ordering_mode=Sorted + AggregateExec: mode=Final, gby=[id@0 as id], aggr=[], group_clustering_mode=Full SortPreservingMergeExec: [id@0 ASC NULLS LAST] - AggregateExec: mode=Partial, gby=[id@0 as id], aggr=[], ordering_mode=Sorted + AggregateExec: mode=Partial, gby=[id@0 as id], aggr=[], group_clustering_mode=Full UnionExec DataSourceExec: file_groups={1 group: [[{testdata}/alltypes_tiny_pages.parquet]]}, projection=[id], output_ordering=[id@0 ASC NULLS LAST], file_type=parquet SortExec: expr=[id@0 ASC NULLS LAST], preserve_partitioning=[false] diff --git a/datafusion/core/tests/fuzz_cases/aggregate_fuzz.rs b/datafusion/core/tests/fuzz_cases/aggregate_fuzz.rs index afbb1b21d0856..74cc881c22baf 100644 --- a/datafusion/core/tests/fuzz_cases/aggregate_fuzz.rs +++ b/datafusion/core/tests/fuzz_cases/aggregate_fuzz.rs @@ -41,7 +41,6 @@ use datafusion_common_runtime::JoinSet; use datafusion_functions_aggregate::sum::sum_udaf; use datafusion_physical_expr::PhysicalSortExpr; use datafusion_physical_expr::expressions::{Column, col, lit}; -use datafusion_physical_plan::InputOrderMode; use test_utils::{StringBatchGenerator, add_empty_batches}; use datafusion_execution::TaskContext; @@ -49,7 +48,7 @@ use datafusion_execution::memory_pool::FairSpillPool; use datafusion_execution::runtime_env::RuntimeEnvBuilder; use datafusion_physical_expr::aggregate::AggregateExprBuilder; use datafusion_physical_plan::aggregates::{ - AggregateExec, AggregateMode, PhysicalGroupBy, + AggregateExec, AggregateMode, GroupClusteringMode, PhysicalGroupBy, }; use datafusion_physical_plan::metrics::MetricValue; use datafusion_physical_plan::{ExecutionPlan, collect, displayable}; @@ -303,7 +302,7 @@ async fn streaming_aggregate_test() { /// two `AggregateExec` variants produce the same result: the pipeline breaking /// one over unordered input (`PartialHashAggregateStream`) and the /// non-pipeline breaking one over ordered input -/// (`OrderedPartialAggregateStream`). +/// (`ClusteredPartialAggregateStream`). async fn run_aggregate_test(input1: Vec, group_by_columns: Vec<&str>) { let schema = input1[0].schema(); let session_config = SessionConfig::new().with_batch_size(50); @@ -354,8 +353,8 @@ async fn run_aggregate_test(input1: Vec, group_by_columns: Vec<&str .unwrap(), ); assert_ne!( - aggregate_exec_running.input_order_mode(), - &InputOrderMode::Linear, + aggregate_exec_running.group_clustering_mode(), + &GroupClusteringMode::None, "running aggregate should observe ordered input for group_by: {group_by:?}" ); @@ -555,13 +554,17 @@ async fn verify_ordered_aggregate(frame: &DataFrame, expected_sort: bool) { fn f_down(&mut self, node: &'n Self::Node) -> Result { if let Some(exec) = node.downcast_ref::() { + assert_eq!( + exec.properties().output_ordering().is_some(), + self.expected_sort + ); if self.expected_sort { assert!(matches!( - exec.input_order_mode(), - InputOrderMode::PartiallySorted(_) | InputOrderMode::Sorted + exec.group_clustering_mode(), + GroupClusteringMode::Partial(_) | GroupClusteringMode::Full )); } else { - assert_eq!(*exec.input_order_mode(), InputOrderMode::Linear); + assert_eq!(*exec.group_clustering_mode(), GroupClusteringMode::None); } } Ok(TreeNodeRecursion::Continue) diff --git a/datafusion/core/tests/fuzz_cases/aggregation_fuzzer/query_builder.rs b/datafusion/core/tests/fuzz_cases/aggregation_fuzzer/query_builder.rs index 2b2d2483d73f9..b4df25a00660c 100644 --- a/datafusion/core/tests/fuzz_cases/aggregation_fuzzer/query_builder.rs +++ b/datafusion/core/tests/fuzz_cases/aggregation_fuzzer/query_builder.rs @@ -93,9 +93,9 @@ pub struct QueryBuilder { /// ... /// ``` /// - /// More details can see [`GroupOrdering`]. + /// More details can see [`GroupClustering`]. /// - /// [`GroupOrdering`]: datafusion_physical_plan::aggregates::order::GroupOrdering + /// [`GroupClustering`]: datafusion_physical_plan::aggregates::order::GroupClustering dataset_sort_keys: Vec>, /// If we will also test the no grouping case like: diff --git a/datafusion/core/tests/physical_optimizer/enforce_distribution.rs b/datafusion/core/tests/physical_optimizer/enforce_distribution.rs index c280a0aa5ef80..7fee51f528035 100644 --- a/datafusion/core/tests/physical_optimizer/enforce_distribution.rs +++ b/datafusion/core/tests/physical_optimizer/enforce_distribution.rs @@ -4867,9 +4867,9 @@ fn preserve_ordering_for_streaming_sorted_aggregate() -> Result<()> { let plan_distrib = test_config.to_plan(physical_plan.clone(), &DISTRIB_DISTRIB_SORT); assert_plan!(plan_distrib, @r" - AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[COUNT(b)], ordering_mode=Sorted + AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[COUNT(b)], group_clustering_mode=Full RepartitionExec: partitioning=Hash([a@0], 2), input_partitions=2, preserve_order=true, sort_exprs=a@0 ASC - AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[COUNT(b)], ordering_mode=Sorted + AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[COUNT(b)], group_clustering_mode=Full DataSourceExec: file_groups={2 groups: [[x], [y]]}, projection=[a, b, c, d, e], output_ordering=[a@0 ASC], file_type=parquet "); @@ -4901,9 +4901,9 @@ fn preserve_ordering_for_streaming_partially_sorted_aggregate() -> Result<()> { let plan_distrib = test_config.to_plan(physical_plan.clone(), &DISTRIB_DISTRIB_SORT); assert_plan!(plan_distrib, @r" - AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b], aggr=[COUNT(c)], ordering_mode=PartiallySorted([0]) + AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b], aggr=[COUNT(c)], group_clustering_mode=Partial([0]) RepartitionExec: partitioning=Hash([a@0, b@1], 2), input_partitions=2, preserve_order=true, sort_exprs=a@0 ASC - AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b], aggr=[COUNT(c)], ordering_mode=PartiallySorted([0]) + AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b], aggr=[COUNT(c)], group_clustering_mode=Partial([0]) DataSourceExec: file_groups={2 groups: [[x], [y]]}, projection=[a, b, c, d, e], output_ordering=[a@0 ASC], file_type=parquet "); diff --git a/datafusion/core/tests/physical_optimizer/limited_distinct_aggregation.rs b/datafusion/core/tests/physical_optimizer/limited_distinct_aggregation.rs index 1d5737d2431ab..833c9485f4cc3 100644 --- a/datafusion/core/tests/physical_optimizer/limited_distinct_aggregation.rs +++ b/datafusion/core/tests/physical_optimizer/limited_distinct_aggregation.rs @@ -520,7 +520,7 @@ fn test_has_order_by() -> Result<()> { actual, @r" LocalLimitExec: fetch=10 - AggregateExec: mode=Single, gby=[a@0 as a], aggr=[], ordering_mode=Sorted + AggregateExec: mode=Single, gby=[a@0 as a], aggr=[], group_clustering_mode=Full DataSourceExec: file_groups={1 group: [[x]]}, projection=[a, b, c, d, e], output_ordering=[a@0 ASC], file_type=parquet " ); diff --git a/datafusion/core/tests/physical_optimizer/pushdown_sort.rs b/datafusion/core/tests/physical_optimizer/pushdown_sort.rs index b72563a942ae3..6cf05ecfd8fc4 100644 --- a/datafusion/core/tests/physical_optimizer/pushdown_sort.rs +++ b/datafusion/core/tests/physical_optimizer/pushdown_sort.rs @@ -731,13 +731,13 @@ fn test_pushdown_through_blocking_node() { OptimizationTest: input: - SortExec: expr=[a@0 ASC], preserve_partitioning=[false] - - AggregateExec: mode=Final, gby=[a@0 as a], aggr=[COUNT(b)], ordering_mode=Sorted + - AggregateExec: mode=Final, gby=[a@0 as a], aggr=[COUNT(b)], group_clustering_mode=Full - SortExec: expr=[a@0 DESC NULLS LAST], preserve_partitioning=[false] - DataSourceExec: file_groups={1 group: [[x]]}, projection=[a, b, c, d, e], output_ordering=[a@0 ASC], file_type=parquet output: Ok: - SortExec: expr=[a@0 ASC], preserve_partitioning=[false] - - AggregateExec: mode=Final, gby=[a@0 as a], aggr=[COUNT(b)], ordering_mode=Sorted + - AggregateExec: mode=Final, gby=[a@0 as a], aggr=[COUNT(b)], group_clustering_mode=Full - SortExec: expr=[a@0 DESC NULLS LAST], preserve_partitioning=[false] - DataSourceExec: file_groups={1 group: [[x]]}, projection=[a, b, c, d, e], file_type=parquet, sort_order_for_reorder=[a@0 DESC NULLS LAST], reverse_row_groups=true " diff --git a/datafusion/expr-common/src/sort_properties.rs b/datafusion/expr-common/src/sort_properties.rs index af417ed16d2d5..53dc8bd8c5568 100644 --- a/datafusion/expr-common/src/sort_properties.rs +++ b/datafusion/expr-common/src/sort_properties.rs @@ -37,16 +37,24 @@ use arrow::datatypes::DataType; pub enum SortProperties { /// Use the ordinary [`SortOptions`] struct to represent ordered data: Ordered(SortOptions), - // This alternative represents unordered data: + /// Within each partition, all rows with the same value for this expression + /// form one contiguous run. The runs may occur in any order. + Grouped, + /// This alternative represents unordered data: #[default] Unordered, - // Singleton is used for single-valued literal numbers: + /// Singleton is used for single-valued literal numbers: Singleton, } impl SortProperties { pub fn add(&self, rhs: &Self) -> Self { match (self, rhs) { + // Addition can collapse distinct values (for example through + // floating-point rounding), which may join non-adjacent groups. + (Self::Grouped, Self::Singleton) | (Self::Singleton, Self::Grouped) => { + Self::Unordered + } (Self::Singleton, _) => *rhs, (_, Self::Singleton) => *self, (Self::Ordered(lhs), Self::Ordered(rhs)) @@ -65,6 +73,11 @@ impl SortProperties { pub fn sub(&self, rhs: &Self) -> Self { match (self, rhs) { (Self::Singleton, Self::Singleton) => Self::Singleton, + // Subtraction can collapse distinct values (for example through + // floating-point rounding), which may join non-adjacent groups. + (Self::Grouped, Self::Singleton) | (Self::Singleton, Self::Grouped) => { + Self::Unordered + } (Self::Singleton, Self::Ordered(rhs)) => Self::Ordered(SortOptions { descending: !rhs.descending, nulls_first: rhs.nulls_first, @@ -89,6 +102,9 @@ impl SortProperties { descending: !rhs.descending, nulls_first: rhs.nulls_first, }), + // Comparisons can map several non-adjacent grouped values to the + // same boolean value, so they do not preserve grouping. + (Self::Grouped, Self::Singleton) => Self::Unordered, (_, Self::Singleton) => *self, (Self::Ordered(lhs), Self::Ordered(rhs)) if lhs.descending != rhs.descending @@ -147,6 +163,7 @@ mod sort_properties_test { const ASC_NL: SortProperties = ordered(false, false); const DESC_NF: SortProperties = ordered(true, true); const DESC_NL: SortProperties = ordered(true, false); + const GROUPED: SortProperties = SortProperties::Grouped; const UNORDERED: SortProperties = SortProperties::Unordered; const SINGLETON: SortProperties = SortProperties::Singleton; @@ -197,6 +214,27 @@ mod sort_properties_test { SINGLETON, SINGLETON, ), + ( + "add: may collapse grouped values", + SortProperties::add, + GROUPED, + SINGLETON, + UNORDERED, + ), + ( + "sub: may collapse grouped values", + SortProperties::sub, + GROUPED, + SINGLETON, + UNORDERED, + ), + ( + "comparison does not preserve grouping", + SortProperties::gt_or_gteq, + GROUPED, + SINGLETON, + UNORDERED, + ), // `and` keeps ASC NULLS LAST / DESC NULLS FIRST, `or` keeps ASC // NULLS FIRST / DESC NULLS LAST. Both are commutative. ( @@ -482,7 +520,7 @@ impl Neg for SortProperties { #[derive(Debug, Clone)] pub struct ExprProperties { /// Properties that describe the sorting behavior of the expression, - /// such as whether it is ordered, unordered, or a singleton value. + /// such as whether it is ordered, grouped, unordered, or a singleton value. pub sort_properties: SortProperties, /// A closed interval representing the range of possible values for /// the expression. Used to compute reliable bounds. diff --git a/datafusion/expr/src/udf.rs b/datafusion/expr/src/udf.rs index d937a54029398..fad9611287fb8 100644 --- a/datafusion/expr/src/udf.rs +++ b/datafusion/expr/src/udf.rs @@ -360,8 +360,20 @@ impl ScalarUDF { /// Calculates the [`SortProperties`] of this function based on its /// children's properties. + /// + /// [`SortProperties::Grouped`] is retained only when the implementation + /// also reports that the transformation is strictly order-preserving. A + /// many-to-one function can otherwise make an output value occur in + /// multiple non-adjacent runs. pub fn output_ordering(&self, inputs: &[ExprProperties]) -> Result { - self.inner.output_ordering(inputs) + let sort_properties = self.inner.output_ordering(inputs)?; + if sort_properties == SortProperties::Grouped + && !self.inner.strictly_order_preserving(inputs)? + { + Ok(SortProperties::Unordered) + } else { + Ok(sort_properties) + } } pub fn preserves_lex_ordering(&self, inputs: &[ExprProperties]) -> Result { @@ -957,7 +969,12 @@ pub trait ScalarUDFImpl: Debug + DynEq + DynHash + Send + Sync + Any { Ok(Some(vec![])) } - /// Calculates the [`SortProperties`] of this function based on its children's properties. + /// Calculates the [`SortProperties`] of this function based on its children's + /// properties. + /// + /// The [`ScalarUDF`] wrapper retains a [`SortProperties::Grouped`] result + /// only when [`Self::strictly_order_preserving`] also returns `true` for + /// the same inputs. fn output_ordering(&self, inputs: &[ExprProperties]) -> Result { if !self.preserves_lex_ordering(inputs)? { return Ok(SortProperties::Unordered); diff --git a/datafusion/ffi/src/expr/expr_properties.rs b/datafusion/ffi/src/expr/expr_properties.rs index 584f774c7b26e..0a9c0bc73aead 100644 --- a/datafusion/ffi/src/expr/expr_properties.rs +++ b/datafusion/ffi/src/expr/expr_properties.rs @@ -67,6 +67,7 @@ pub enum FFI_SortProperties { Ordered(FFI_SortOptions), Unordered, Singleton, + Grouped, } impl From<&SortProperties> for FFI_SortProperties { @@ -74,6 +75,7 @@ impl From<&SortProperties> for FFI_SortProperties { match value { SortProperties::Unordered => FFI_SortProperties::Unordered, SortProperties::Singleton => FFI_SortProperties::Singleton, + SortProperties::Grouped => FFI_SortProperties::Grouped, SortProperties::Ordered(o) => FFI_SortProperties::Ordered(o.into()), } } @@ -84,6 +86,7 @@ impl From<&FFI_SortProperties> for SortProperties { match value { FFI_SortProperties::Unordered => SortProperties::Unordered, FFI_SortProperties::Singleton => SortProperties::Singleton, + FFI_SortProperties::Grouped => SortProperties::Grouped, FFI_SortProperties::Ordered(o) => SortProperties::Ordered(o.into()), } } diff --git a/datafusion/ffi/src/tests/udf_udaf_udwf.rs b/datafusion/ffi/src/tests/udf_udaf_udwf.rs index ea7c03b5d6487..98f874a23fc75 100644 --- a/datafusion/ffi/src/tests/udf_udaf_udwf.rs +++ b/datafusion/ffi/src/tests/udf_udaf_udwf.rs @@ -21,7 +21,7 @@ use arrow_schema::DataType; use datafusion_catalog::TableFunctionImpl; use datafusion_common::ScalarValue; use datafusion_common::config::ConfigOptions; -use datafusion_expr::sort_properties::ExprProperties; +use datafusion_expr::sort_properties::{ExprProperties, SortProperties}; use datafusion_expr::{ AggregateUDF, ColumnarValue, ExpressionPlacement, ScalarFunctionArgs, ScalarUDF, ScalarUDFImpl, Signature, Volatility, WindowUDF, @@ -172,6 +172,11 @@ impl ScalarUDFImpl for PlacementUDF { &self, inputs: &[ExprProperties], ) -> datafusion_common::Result { + // This test-only sentinel verifies that the new `Grouped` variant + // survives the cross-library `ExprProperties` conversion. + if matches!(inputs, [input] if input.sort_properties == SortProperties::Grouped) { + return Ok(true); + } Ok(inputs.iter().all(|input| input.preserves_lex_ordering)) } } diff --git a/datafusion/ffi/src/udf/mod.rs b/datafusion/ffi/src/udf/mod.rs index d14614f1474a3..72e5b59c426e8 100644 --- a/datafusion/ffi/src/udf/mod.rs +++ b/datafusion/ffi/src/udf/mod.rs @@ -553,6 +553,7 @@ impl ScalarUDFImpl for ForeignScalarUDF { #[cfg(test)] mod tests { use super::*; + use datafusion_expr::sort_properties::SortProperties; #[derive(Debug, PartialEq, Eq, Hash)] struct PlacementUDF { @@ -594,6 +595,13 @@ mod tests { return internal_err!("preserves_lex_ordering requires an input"); } + // This test-only sentinel verifies that the new `Grouped` variant + // travels through the foreign path. + if matches!(inputs, [input] if input.sort_properties == SortProperties::Grouped) + { + return Ok(true); + } + Ok(inputs.iter().all(|input| input.preserves_lex_ordering)) } @@ -692,6 +700,8 @@ mod tests { .preserves_lex_ordering(&[preserves, does_not_preserve]) .unwrap() ); + let grouped = ExprProperties::new_unknown().with_order(SortProperties::Grouped); + assert!(foreign_udf.preserves_lex_ordering(&[grouped]).unwrap()); assert!(foreign_udf.preserves_lex_ordering(&[]).is_err()); let updated = foreign_udf diff --git a/datafusion/ffi/tests/ffi_udf.rs b/datafusion/ffi/tests/ffi_udf.rs index f97153526f733..2b83ccf013f28 100644 --- a/datafusion/ffi/tests/ffi_udf.rs +++ b/datafusion/ffi/tests/ffi_udf.rs @@ -27,7 +27,7 @@ mod tests { use datafusion::prelude::{SessionContext, col}; use datafusion_execution::config::SessionConfig; use datafusion_expr::lit; - use datafusion_expr::sort_properties::ExprProperties; + use datafusion_expr::sort_properties::{ExprProperties, SortProperties}; use datafusion_ffi::tests::create_record_batch; use datafusion_ffi::tests::utils::get_module; use std::sync::Arc; @@ -119,6 +119,9 @@ mod tests { assert!(foreign_func.preserves_lex_ordering(std::slice::from_ref(&preserves))?); assert!(!foreign_func.preserves_lex_ordering(&[preserves, does_not_preserve])?); + let grouped = ExprProperties::new_unknown().with_order(SortProperties::Grouped); + assert!(foreign_func.preserves_lex_ordering(&[grouped])?); + Ok(()) } diff --git a/datafusion/functions/src/datetime/date_bin.rs b/datafusion/functions/src/datetime/date_bin.rs index dc3da309dab82..58582ce75362e 100644 --- a/datafusion/functions/src/datetime/date_bin.rs +++ b/datafusion/functions/src/datetime/date_bin.rs @@ -822,12 +822,15 @@ mod tests { use crate::datetime::date_bin::{DateBinFunc, date_bin_nanos_interval}; use arrow::array::types::TimestampNanosecondType; use arrow::array::{Array, IntervalDayTimeArray, TimestampNanosecondArray}; + use arrow::compute::SortOptions; use arrow::compute::kernels::cast_utils::string_to_timestamp_nanos; use arrow::datatypes::{DataType, Field, FieldRef, TimeUnit}; use arrow_buffer::{IntervalDayTime, IntervalMonthDayNano}; use datafusion_common::{DataFusionError, ScalarValue}; - use datafusion_expr::{ColumnarValue, ScalarFunctionArgs, ScalarUDFImpl}; + use datafusion_expr::interval_arithmetic::Interval; + use datafusion_expr::sort_properties::{ExprProperties, SortProperties}; + use datafusion_expr::{ColumnarValue, ScalarFunctionArgs, ScalarUDF, ScalarUDFImpl}; use chrono::TimeDelta; use datafusion_common::config::ConfigOptions; @@ -877,6 +880,29 @@ mod tests { ); } + #[test] + fn date_bin_does_not_propagate_grouping() -> Result<(), DataFusionError> { + let udf = ScalarUDF::from(DateBinFunc::new()); + let step = ExprProperties::new_unknown().with_order(SortProperties::Singleton); + let timestamp = ExprProperties::new_unknown().with_range( + Interval::make_unbounded(&DataType::Timestamp(TimeUnit::Nanosecond, None))?, + ); + let grouped = timestamp.clone().with_order(SortProperties::Grouped); + + assert_eq!( + udf.output_ordering(&[step.clone(), grouped])?, + SortProperties::Unordered + ); + + let ordered = + timestamp.with_order(SortProperties::Ordered(SortOptions::default())); + assert_eq!( + udf.output_ordering(&[step, ordered])?, + SortProperties::Ordered(SortOptions::default()) + ); + Ok(()) + } + #[test] fn test_date_bin() { let return_field = &Arc::new(Field::new( diff --git a/datafusion/physical-expr/src/equivalence/grouping.rs b/datafusion/physical-expr/src/equivalence/grouping.rs new file mode 100644 index 0000000000000..ca3570ece05b8 --- /dev/null +++ b/datafusion/physical-expr/src/equivalence/grouping.rs @@ -0,0 +1,203 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +use std::fmt::Display; +use std::ops::Deref; +use std::sync::Arc; +use std::vec::IntoIter; + +use crate::PhysicalExpr; +use crate::expressions::with_new_schema; + +use arrow::datatypes::SchemaRef; +use datafusion_common::{HashSet, Result}; +use datafusion_physical_expr_common::physical_expr::format_physical_expr_list; + +/// A set of grouping tuples that are known to describe contiguous rows. +/// +/// Each entry is a complete grouping tuple. For example, `[a, b]` means that, +/// within each output partition, all rows having the same values for both `a` +/// and `b` occur in one contiguous run. The runs themselves may occur in any +/// order. Contiguity applies to the complete partition stream, not separately +/// to each record batch. +/// +/// Expression order within an entry is not significant: `[a, b]` and `[b, a]` +/// describe the same grouping. Entries do not imply properties for subsets; +/// `[a, b]` alone says nothing about whether all rows with the same `a` are +/// contiguous. +#[derive(Clone, Debug, Default)] +pub struct GroupingEquivalenceClass { + groupings: Vec>>, +} + +impl GroupingEquivalenceClass { + /// Clears all groupings in this equivalence class. + pub fn clear(&mut self) { + self.groupings.clear(); + } + + /// Creates a grouping equivalence class, discarding empty and duplicate + /// entries and duplicate expressions within each entry. + pub fn new( + groupings: impl IntoIterator>>, + ) -> Self { + let mut result = Self::default(); + result.add_groupings(groupings); + result + } + + /// Adds grouping tuples to this equivalence class. + pub fn add_groupings( + &mut self, + groupings: impl IntoIterator>>, + ) { + for grouping in groupings { + let mut seen = HashSet::new(); + let grouping = grouping + .into_iter() + .filter(|expr| seen.insert(Arc::clone(expr))) + .collect::>(); + if !grouping.is_empty() && !self.contains(&grouping) { + self.groupings.push(grouping); + } + } + } + + /// Returns whether this class contains the complete grouping tuple. + pub fn contains(&self, grouping: &[Arc]) -> bool { + self.groupings + .iter() + .any(|candidate| same_grouping(candidate, grouping)) + } + + /// Rewrites all expressions to reference an aligned schema. + pub fn with_new_schema(self, schema: &SchemaRef) -> Result { + let groupings = self.groupings.into_iter().map(|grouping| { + grouping + .into_iter() + .map(|expr| with_new_schema(expr, schema)) + .collect::>>() + }); + Ok(Self::new(groupings.collect::>>()?)) + } +} + +fn same_grouping(lhs: &[Arc], rhs: &[Arc]) -> bool { + lhs.iter() + .all(|lhs_expr| rhs.iter().any(|rhs_expr| lhs_expr.eq(rhs_expr))) + && rhs + .iter() + .all(|rhs_expr| lhs.iter().any(|lhs_expr| rhs_expr.eq(lhs_expr))) +} + +impl PartialEq for GroupingEquivalenceClass { + fn eq(&self, other: &Self) -> bool { + self.groupings.len() == other.groupings.len() + && self + .groupings + .iter() + .all(|grouping| other.contains(grouping)) + } +} + +impl Eq for GroupingEquivalenceClass {} + +impl Deref for GroupingEquivalenceClass { + type Target = [Vec>]; + + fn deref(&self) -> &Self::Target { + self.groupings.as_slice() + } +} + +impl From>>> for GroupingEquivalenceClass { + fn from(groupings: Vec>>) -> Self { + Self::new(groupings) + } +} + +/// Converts the grouping equivalence class into an iterator of complete +/// grouping tuples. +impl IntoIterator for GroupingEquivalenceClass { + type Item = Vec>; + type IntoIter = IntoIter; + + fn into_iter(self) -> Self::IntoIter { + self.groupings.into_iter() + } +} + +impl From for Vec>> { + fn from(geq_class: GroupingEquivalenceClass) -> Self { + geq_class.groupings + } +} + +impl Display for GroupingEquivalenceClass { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "[")?; + let mut groupings = self.groupings.iter(); + if let Some(grouping) = groupings.next() { + write!(f, "{}", format_physical_expr_list(grouping))?; + } + for grouping in groupings { + write!(f, ", {}", format_physical_expr_list(grouping))?; + } + write!(f, "]") + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::expressions::Column; + + #[test] + fn grouping_equality_ignores_expression_and_entry_order() { + let a = Arc::new(Column::new("a", 0)) as Arc; + let b = Arc::new(Column::new("b", 1)) as Arc; + let c = Arc::new(Column::new("c", 2)) as Arc; + + let lhs = GroupingEquivalenceClass::new([ + vec![Arc::clone(&a), Arc::clone(&b)], + vec![Arc::clone(&c)], + ]); + let rhs = GroupingEquivalenceClass::new([ + vec![Arc::clone(&c)], + vec![Arc::clone(&b), Arc::clone(&a)], + ]); + + assert_eq!(lhs, rhs); + } + + #[test] + fn grouping_deduplicates_entries_and_expressions() { + let a = Arc::new(Column::new("a", 0)) as Arc; + let b = Arc::new(Column::new("b", 1)) as Arc; + + let groupings = GroupingEquivalenceClass::new([ + vec![Arc::clone(&a), Arc::clone(&a), Arc::clone(&b)], + vec![Arc::clone(&b), Arc::clone(&a)], + vec![], + ]); + + assert_eq!(groupings.len(), 1); + assert_eq!(groupings[0].len(), 2); + assert!(groupings.contains(&[Arc::clone(&a), Arc::clone(&a), Arc::clone(&b),])); + assert_eq!(groupings.to_string(), "[[a@0, b@1]]"); + } +} diff --git a/datafusion/physical-expr/src/equivalence/mod.rs b/datafusion/physical-expr/src/equivalence/mod.rs index 64bb62901310f..ab4aacdd4c2fb 100644 --- a/datafusion/physical-expr/src/equivalence/mod.rs +++ b/datafusion/physical-expr/src/equivalence/mod.rs @@ -24,10 +24,12 @@ use arrow::compute::SortOptions; use datafusion_physical_expr_common::sort_expr::{LexOrdering, PhysicalSortExpr}; mod class; +mod grouping; mod ordering; mod properties; pub use class::{AcrossPartitions, ConstExpr, EquivalenceClass, EquivalenceGroup}; +pub use grouping::GroupingEquivalenceClass; pub use ordering::OrderingEquivalenceClass; // Re-export for backwards compatibility, we recommend importing from // datafusion_physical_expr::projection instead @@ -57,11 +59,12 @@ pub fn convert_to_orderings>>( #[cfg(test)] mod tests { use super::*; - use crate::expressions::{Column, col}; + use crate::expressions::{BinaryExpr, Column, cast, col, lit}; use crate::{LexRequirement, PhysicalSortExpr}; use arrow::datatypes::{DataType, Field, Schema, SchemaRef}; use datafusion_common::Result; + use datafusion_expr::Operator; use datafusion_physical_expr_common::sort_expr::PhysicalSortRequirement; /// Converts a string to a physical sort expression @@ -220,4 +223,199 @@ mod tests { Ok(()) } + + #[test] + fn grouping_satisfy_requires_the_complete_tuple() -> Result<()> { + let schema = create_test_schema()?; + let a = col("a", &schema)?; + let b = col("b", &schema)?; + let c = col("c", &schema)?; + let mut properties = EquivalenceProperties::new(schema); + properties.add_grouping([Arc::clone(&a), Arc::clone(&b)]); + + assert!(properties.grouping_satisfy([Arc::clone(&a), Arc::clone(&b)])?); + assert!(properties.grouping_satisfy([Arc::clone(&b), Arc::clone(&a)])?); + assert!(!properties.grouping_satisfy([Arc::clone(&a)])?); + assert!(!properties.grouping_satisfy([Arc::clone(&b)])?); + assert!(!properties.grouping_satisfy([a, b, c])?); + Ok(()) + } + + #[test] + fn ordering_implies_grouped_prefixes() -> Result<()> { + let schema = create_test_schema()?; + let a = col("a", &schema)?; + let b = col("b", &schema)?; + let c = col("c", &schema)?; + let properties = EquivalenceProperties::new_with_orderings( + schema, + [[ + PhysicalSortExpr::new_default(Arc::clone(&a)), + PhysicalSortExpr::new_default(Arc::clone(&b)), + PhysicalSortExpr::new_default(Arc::clone(&c)), + ]], + ); + + assert!(properties.grouping_satisfy([Arc::clone(&a)])?); + assert!(properties.grouping_satisfy([Arc::clone(&a), Arc::clone(&b)])?); + assert!(properties.grouping_satisfy([Arc::clone(&b), Arc::clone(&a)])?); + assert!(!properties.grouping_satisfy([Arc::clone(&b)])?); + assert!(!properties.grouping_satisfy([a, c])?); + Ok(()) + } + + #[test] + fn equivalent_orderings_can_jointly_satisfy_grouping() -> Result<()> { + let schema = create_test_schema()?; + let a = col("a", &schema)?; + let b = col("b", &schema)?; + let properties = EquivalenceProperties::new_with_orderings( + schema, + [ + [PhysicalSortExpr::new_default(Arc::clone(&a))], + [PhysicalSortExpr::new_default(Arc::clone(&b))], + ], + ); + + assert!(properties.grouping_satisfy([b, a])?); + Ok(()) + } + + #[test] + fn reorder_clears_explicit_grouping() -> Result<()> { + let schema = create_test_schema()?; + let a = col("a", &schema)?; + let b = col("b", &schema)?; + let ordering = [PhysicalSortExpr::new_default(Arc::clone(&b))]; + let mut properties = + EquivalenceProperties::new_with_orderings(schema, [ordering.clone()]); + properties.add_grouping([Arc::clone(&a)]); + + // The ordering is already satisfied, but a physical sort may still + // rearrange equal `b` values and separate an `a` group. + assert!(!properties.reorder(ordering)?); + assert!(!properties.grouping_satisfy([a])?); + assert!(properties.grouping_satisfy([b])?); + Ok(()) + } + + #[test] + fn grouping_normalizes_equivalences_and_constants() -> Result<()> { + let schema = create_test_schema()?; + let a = col("a", &schema)?; + let b = col("b", &schema)?; + let c = col("c", &schema)?; + let mut properties = EquivalenceProperties::new(schema); + properties.add_equal_conditions(Arc::clone(&a), Arc::clone(&c))?; + properties.add_constants([ConstExpr::from(Arc::clone(&a))])?; + properties.add_grouping([Arc::clone(&a), Arc::clone(&b)]); + + assert!(properties.grouping_satisfy([Arc::clone(&b)])?); + assert!(properties.grouping_satisfy([Arc::clone(&b), Arc::clone(&c)])?); + + properties.clear_per_partition_constants(); + assert!(!properties.grouping_satisfy([Arc::clone(&b)])?); + assert!(properties.grouping_satisfy([b, c])?); + Ok(()) + } + + #[test] + fn project_grouping_preserves_complete_tuple() -> Result<()> { + let schema = create_test_schema()?; + let a = col("a", &schema)?; + let b = col("b", &schema)?; + let c = col("c", &schema)?; + let mut properties = EquivalenceProperties::new(Arc::clone(&schema)); + properties.add_grouping([Arc::clone(&a), Arc::clone(&b)]); + + let output_schema = Arc::new(Schema::new(vec![ + Field::new("z", DataType::Int32, true), + Field::new("x", DataType::Int32, true), + Field::new("y", DataType::Int32, true), + ])); + let mapping = ProjectionMapping::try_new( + [ + (c, "z".to_string()), + (Arc::clone(&b), "x".to_string()), + (Arc::clone(&a), "y".to_string()), + ], + &schema, + )?; + let projected = properties.project(&mapping, Arc::clone(&output_schema)); + let x = col("x", &output_schema)?; + let y = col("y", &output_schema)?; + assert!(projected.grouping_satisfy([x, y])?); + Ok(()) + } + + #[test] + fn project_grouping_drops_incomplete_tuple() -> Result<()> { + let schema = create_test_schema()?; + let a = col("a", &schema)?; + let b = col("b", &schema)?; + let c = col("c", &schema)?; + let mut properties = EquivalenceProperties::new(Arc::clone(&schema)); + properties.add_grouping([Arc::clone(&a), b]); + + let output_schema = Arc::new(Schema::new(vec![ + Field::new("x", DataType::Int32, true), + Field::new("z", DataType::Int32, true), + ])); + let mapping = ProjectionMapping::try_new( + [(a, "x".to_string()), (c, "z".to_string())], + &schema, + )?; + let projected = properties.project(&mapping, Arc::clone(&output_schema)); + assert!(projected.geq_class().is_empty()); + assert!(!projected.grouping_satisfy([col("x", &output_schema)?])?); + Ok(()) + } + + #[test] + fn project_grouping_omits_constant_members() -> Result<()> { + let schema = create_test_schema()?; + let a = col("a", &schema)?; + let b = col("b", &schema)?; + let mut properties = EquivalenceProperties::new(Arc::clone(&schema)); + properties.add_constants([ConstExpr::from(Arc::clone(&a))])?; + properties.add_grouping([a, Arc::clone(&b)]); + + let output_schema = + Arc::new(Schema::new(vec![Field::new("x", DataType::Int32, true)])); + let mapping = ProjectionMapping::try_new([(b, "x".to_string())], &schema)?; + let projected = properties.project(&mapping, Arc::clone(&output_schema)); + + assert!(projected.grouping_satisfy([col("x", &output_schema)?])?); + Ok(()) + } + + #[test] + fn project_grouped_expression_requires_an_injective_mapping() -> Result<()> { + let schema = create_test_schema()?; + let a = col("a", &schema)?; + let mut properties = EquivalenceProperties::new(Arc::clone(&schema)); + properties.add_grouping([Arc::clone(&a)]); + + let widened = cast(Arc::clone(&a), &schema, DataType::Int64)?; + let output_schema = + Arc::new(Schema::new(vec![Field::new("x", DataType::Int64, true)])); + let mapping = ProjectionMapping::try_new([(widened, "x".to_string())], &schema)?; + let projected = properties.project(&mapping, Arc::clone(&output_schema)); + assert!(projected.grouping_satisfy([col("x", &output_schema)?])?); + + let greater_than_one = Arc::new(BinaryExpr::new(a, Operator::Gt, lit(1_i32))) + as Arc; + let output_schema = Arc::new(Schema::new(vec![Field::new( + "is_greater", + DataType::Boolean, + true, + )])); + let mapping = ProjectionMapping::try_new( + [(greater_than_one, "is_greater".to_string())], + &schema, + )?; + let projected = properties.project(&mapping, Arc::clone(&output_schema)); + assert!(projected.geq_class().is_empty()); + Ok(()) + } } diff --git a/datafusion/physical-expr/src/equivalence/properties/mod.rs b/datafusion/physical-expr/src/equivalence/properties/mod.rs index c68157fecbd8c..df58bba6d7dc3 100644 --- a/datafusion/physical-expr/src/equivalence/properties/mod.rs +++ b/datafusion/physical-expr/src/equivalence/properties/mod.rs @@ -31,7 +31,8 @@ use self::dependency::{ generate_dependency_orderings, referred_dependencies, }; use crate::equivalence::{ - AcrossPartitions, EquivalenceGroup, OrderingEquivalenceClass, ProjectionMapping, + AcrossPartitions, EquivalenceGroup, GroupingEquivalenceClass, + OrderingEquivalenceClass, ProjectionMapping, }; use crate::expressions::{Column, Literal, with_new_schema}; use crate::{ @@ -41,7 +42,7 @@ use crate::{ use arrow::datatypes::SchemaRef; use datafusion_common::tree_node::{Transformed, TransformedResult, TreeNode}; -use datafusion_common::{Constraint, Constraints, HashMap, Result, plan_err}; +use datafusion_common::{Constraint, Constraints, HashMap, HashSet, Result, plan_err}; use datafusion_expr::interval_arithmetic::Interval; use datafusion_expr::sort_properties::{ExprProperties, SortProperties}; use datafusion_physical_expr_common::sort_expr::options_compatible; @@ -53,6 +54,7 @@ use itertools::Itertools; /// `EquivalenceProperties` stores information about the output of a plan node /// that can be used to optimize the plan. Currently, it keeps track of: /// - Sort expressions (orderings), +/// - Complete expression tuples whose values are contiguous, /// - Equivalent expressions; i.e. expressions known to have the same value. /// - Constants expressions; i.e. expressions known to contain a single constant /// value. @@ -138,10 +140,16 @@ pub struct EquivalenceProperties { eq_group: EquivalenceGroup, /// Equivalent sort expressions (i.e. those define the same ordering). oeq_class: OrderingEquivalenceClass, + /// Complete expression tuples whose values occupy one contiguous run + /// in each output partition. + geq_class: GroupingEquivalenceClass, /// Cache storing equivalent sort expressions in normal form (i.e. without /// constants/duplicates and in standard form) and a map associating leading /// terms with full sort expressions. oeq_cache: OrderingEquivalenceCache, + /// Grouping expressions in normal form (i.e. without constants, + /// duplicates, or non-canonical equivalent expressions). + geq_cache: GroupingEquivalenceClass, /// Table constraints that factor in equivalence calculations. constraints: Constraints, /// Schema associated with this object. @@ -252,7 +260,9 @@ impl EquivalenceProperties { Self { eq_group: EquivalenceGroup::default(), oeq_class: OrderingEquivalenceClass::default(), + geq_class: GroupingEquivalenceClass::default(), oeq_cache: OrderingEquivalenceCache::default(), + geq_cache: GroupingEquivalenceClass::default(), constraints: Constraints::default(), schema, } @@ -285,6 +295,8 @@ impl EquivalenceProperties { Self { oeq_cache: OrderingEquivalenceCache::new(normal_orderings), oeq_class, + geq_class: GroupingEquivalenceClass::default(), + geq_cache: GroupingEquivalenceClass::default(), eq_group, constraints: Constraints::default(), schema, @@ -301,6 +313,11 @@ impl EquivalenceProperties { &self.oeq_class } + /// Returns a reference to the grouping equivalence class within. + pub fn geq_class(&self) -> &GroupingEquivalenceClass { + &self.geq_class + } + /// Returns a reference to the equivalence group within. pub fn eq_group(&self) -> &EquivalenceGroup { &self.eq_group @@ -336,6 +353,7 @@ impl EquivalenceProperties { self.constraints.extend(other.constraints); self.add_equivalence_group(other.eq_group)?; self.add_orderings(other.oeq_class); + self.add_groupings(other.geq_class); Ok(self) } @@ -346,6 +364,13 @@ impl EquivalenceProperties { self.oeq_cache.clear(); } + /// Clears grouping information invalidated by an operation that changes + /// row sequence or combines input partitions. + pub fn clear_groupings(&mut self) { + self.geq_class.clear(); + self.geq_cache.clear(); + } + /// Removes constant expressions that may change across partitions. /// This method should be used when merging data from different partitions. pub fn clear_per_partition_constants(&mut self) { @@ -357,6 +382,7 @@ impl EquivalenceProperties { .cloned() .map(|o| self.eq_group.normalize_sort_exprs(o)); self.oeq_cache = OrderingEquivalenceCache::new(normal_orderings); + self.update_geq_cache(); } } @@ -387,6 +413,61 @@ impl EquivalenceProperties { self.add_orderings(std::iter::once(ordering)); } + /// Adds complete grouping tuples whose values are contiguous within + /// every output partition. + /// + /// This is a correctness assertion. DataFusion does not verify the row + /// layout, and consumers may act on a tuple as soon as its values change. + /// Callers must therefore only add tuples that hold for every output + /// partition. + /// + /// Expression order within a tuple is not significant. A tuple only + /// describes its complete set of expressions: grouping `[a, b]` does not + /// imply grouping `[a]` or `[b]`. + pub fn add_groupings( + &mut self, + groupings: impl IntoIterator>>, + ) { + for grouping in GroupingEquivalenceClass::new(groupings) { + let normal_grouping = self.normalize_grouping(grouping.iter().cloned()); + self.geq_class.add_groupings(std::iter::once(grouping)); + if !normal_grouping.is_empty() { + self.geq_cache + .add_groupings(std::iter::once(normal_grouping)); + } + } + } + + /// Adds one complete grouping tuple. + pub fn add_grouping( + &mut self, + grouping: impl IntoIterator>, + ) { + self.add_groupings(std::iter::once(grouping)); + } + + fn normalize_grouping( + &self, + grouping: impl IntoIterator>, + ) -> Vec> { + let mut seen = HashSet::new(); + grouping + .into_iter() + .map(|expr| self.eq_group.normalize_expr(expr)) + .filter(|expr| self.eq_group.is_expr_constant(expr).is_none()) + .filter(|expr| seen.insert(Arc::clone(expr))) + .collect() + } + + fn update_geq_cache(&mut self) { + let groupings = self + .geq_class + .iter() + .map(|grouping| self.normalize_grouping(grouping.iter().cloned())) + .collect::>(); + self.geq_cache = GroupingEquivalenceClass::new(groupings); + } + fn update_oeq_cache(&mut self) -> Result<()> { // Renormalize orderings if the equivalence group changes: let normal_cls = mem::take(&mut self.oeq_cache.normal_cls); @@ -412,6 +493,7 @@ impl EquivalenceProperties { if !other_eq_group.is_empty() { self.eq_group.extend(other_eq_group); self.update_oeq_cache()?; + self.update_geq_cache(); } Ok(()) } @@ -428,6 +510,11 @@ impl EquivalenceProperties { .into() } + /// Returns the grouping equivalence class within in normal form. + pub fn normalized_geq_class(&self) -> GroupingEquivalenceClass { + self.geq_cache.clone() + } + /// Adds a new equality condition into the existing equivalence group. /// If the given equality defines a new equivalence class, adds this new /// equivalence class to the equivalence group. @@ -441,6 +528,7 @@ impl EquivalenceProperties { self.update_oeq_cache()?; } self.update_oeq_cache()?; + self.update_geq_cache(); Ok(()) } @@ -463,6 +551,7 @@ impl EquivalenceProperties { }); self.oeq_cache.normal_cls = OrderingEquivalenceClass::new(normal_orderings); self.oeq_cache.update_map(); + self.update_geq_cache(); // Discover any new orderings based on the constants: let leading_exprs: Vec<_> = self.oeq_cache.leading_map.keys().cloned().collect(); for expr in leading_exprs { @@ -539,15 +628,21 @@ impl EquivalenceProperties { Ok(()) } - /// Updates the ordering equivalence class within assuming that the table - /// is re-sorted according to the argument `ordering`, and returns whether - /// this operation resulted in any change. Note that equivalence classes - /// (and constants) do not change as they are unaffected by a re-sort. If - /// the given ordering is already satisfied, the function does nothing. + /// Updates the ordering equivalence class assuming that the table is + /// re-sorted according to `ordering`, and returns whether the ordering + /// class changed. Equivalence classes and constants are unaffected by a + /// re-sort. Explicit grouping assertions are cleared because a sort may + /// rearrange rows that compare equally, even when `ordering` was already + /// satisfied. pub fn reorder( &mut self, ordering: impl IntoIterator, ) -> Result { + // A sort may reorder rows that compare equally under `ordering`, so + // explicit grouping assertions do not necessarily survive even when + // the requested ordering is already satisfied. Groupings implied by + // the output ordering remain discoverable through `oeq_class`. + self.clear_groupings(); let (ordering, ordering_tee) = ordering.into_iter().tee(); // First, standardize the given ordering: let Some(normal_ordering) = self.normalize_sort_exprs(ordering) else { @@ -585,6 +680,27 @@ impl EquivalenceProperties { LexRequirement::new(self.eq_group.normalize_sort_requirements(sort_reqs)) } + /// Returns whether the given complete expression tuple is known to be + /// grouped within every output partition. + /// + /// A tuple is grouped when all rows with the same values for the complete + /// tuple occur in one contiguous run. The order of expressions within a + /// grouping tuple does not matter: `[a, b]` and `[b, a]` describe the same + /// groups. A lexicographical ordering also satisfies grouping for each of + /// its prefixes, regardless of sort direction. + pub fn grouping_satisfy( + &self, + given: impl IntoIterator>, + ) -> Result { + let normal_grouping = self.normalize_grouping(given); + if normal_grouping.is_empty() || self.geq_cache.contains(&normal_grouping) { + return Ok(true); + } + + let (_, indices) = self.find_longest_permutation(&normal_grouping)?; + Ok(indices.len() == normal_grouping.len()) + } + /// Iteratively checks whether the given ordering is satisfied by any of /// the existing orderings. See [`Self::ordering_satisfy_requirement`] for /// more details and examples. @@ -658,7 +774,7 @@ impl EquivalenceProperties { }), // Singleton expressions satisfy any requirement. SortProperties::Singleton => true, - SortProperties::Unordered => false, + SortProperties::Grouped | SortProperties::Unordered => false, }; if !satisfy { return Ok(false); @@ -747,7 +863,7 @@ impl EquivalenceProperties { ), // Singleton expressions satisfy any ordering. SortProperties::Singleton => true, - SortProperties::Unordered => false, + SortProperties::Grouped | SortProperties::Unordered => false, }; if !satisfy { // As soon as one sort expression is unsatisfied, return how @@ -1184,6 +1300,34 @@ impl EquivalenceProperties { orderings.chain(projected_orderings).collect() } + /// Projects grouping tuples through `mapping`, dropping a tuple unless all + /// of its expressions can be represented by the projection. + fn projected_groupings( + &self, + mapping: &ProjectionMapping, + ) -> Vec>> { + let mut groupings = self + .geq_cache + .iter() + .filter_map(|grouping| { + self.project_expressions(grouping.iter(), mapping) + .collect::>>() + }) + .collect::>(); + + // A projection expression that safely preserves a one-expression + // grouping establishes the same property for its output column. + for (source, targets) in mapping.iter() { + if self.get_expr_properties(Arc::clone(source)).sort_properties + == SortProperties::Grouped + { + groupings + .extend(targets.iter().map(|(target, _)| vec![Arc::clone(target)])); + } + } + groupings + } + /// Projects constraints according to the given projection mapping. /// /// This function takes a projection mapping and extracts column indices of @@ -1227,6 +1371,8 @@ impl EquivalenceProperties { /// preserved through `c + 1` but dropped through `abs(c)`. Orderings /// implied by the mapping are also derived, e.g. an ordering on `a + b` /// yields one on the projected `a_new + b_new`. + /// - Groupings: a complete grouping tuple is carried only when every + /// expression in the tuple can be represented in the output. /// - Equivalence group: each class is re-expressed on the output columns. /// - Constraints: projected onto the surviving column indices. /// @@ -1278,17 +1424,22 @@ impl EquivalenceProperties { ) -> Self { let orderings = self.projected_orderings(mapping, self.oeq_cache.normal_cls.clone()); + let groupings = self.projected_groupings(mapping); let normal_orderings = orderings .iter() .cloned() .map(|o| eq_group.normalize_sort_exprs(o)); - Self { + let mut result = Self { oeq_cache: OrderingEquivalenceCache::new(normal_orderings), oeq_class: OrderingEquivalenceClass::new(orderings), + geq_class: GroupingEquivalenceClass::default(), + geq_cache: GroupingEquivalenceClass::default(), constraints: self.projected_constraints(mapping).unwrap_or_default(), schema: output_schema, eq_group, - } + }; + result.add_groupings(groupings); + result } /// Returns the longest (potentially partial) permutation satisfying the @@ -1335,7 +1486,7 @@ impl EquivalenceProperties { let expr = Arc::clone(&exprs[idx]); Some((PhysicalSortExpr::new_default(expr), idx)) } - SortProperties::Unordered => None, + SortProperties::Grouped | SortProperties::Unordered => None, } }) .collect::>(); @@ -1468,6 +1619,11 @@ impl EquivalenceProperties { self.oeq_class = self.oeq_class.with_new_schema(&schema)?; self.oeq_cache.normal_cls = self.oeq_cache.normal_cls.with_new_schema(&schema)?; + // Rewrite grouping expressions according to new schema and rebuild + // their normalized form against the rewritten equivalence group. + self.geq_class = self.geq_class.with_new_schema(&schema)?; + self.update_geq_cache(); + // Update the schema: self.schema = schema; @@ -1485,21 +1641,28 @@ impl From for OrderingEquivalenceClass { /// /// Format: /// ```text -/// order: [[b@1 ASC NULLS LAST]], eq: [{members: [a@0], constant: (heterogeneous)}] +/// order: [[b@1 ASC NULLS LAST]], group: [[a@0, b@1]], eq: [{members: [a@0], constant: (heterogeneous)}] /// ``` impl Display for EquivalenceProperties { fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { let empty_eq_group = self.eq_group.is_empty(); let empty_oeq_class = self.oeq_class.is_empty(); - if empty_oeq_class && empty_eq_group { - write!(f, "No properties")?; - } else if !empty_oeq_class { + let empty_geq_class = self.geq_class.is_empty(); + if empty_oeq_class && empty_geq_class && empty_eq_group { + return write!(f, "No properties"); + } + + let mut separator = ""; + if !empty_oeq_class { write!(f, "order: {}", self.oeq_class)?; - if !empty_eq_group { - write!(f, ", eq: {}", self.eq_group)?; - } - } else { - write!(f, "eq: {}", self.eq_group)?; + separator = ", "; + } + if !empty_geq_class { + write!(f, "{separator}group: {}", self.geq_class)?; + separator = ", "; + } + if !empty_eq_group { + write!(f, "{separator}eq: {}", self.eq_group)?; } Ok(()) } @@ -1556,6 +1719,11 @@ fn update_properties( node.data.sort_properties = SortProperties::Singleton; } else if let Some(options) = oeq_class.get_options(&normal_expr) { node.data.sort_properties = SortProperties::Ordered(options); + } else if eq_properties + .geq_cache + .contains(std::slice::from_ref(&normal_expr)) + { + node.data.sort_properties = SortProperties::Grouped; } Ok(Transformed::yes(node)) } diff --git a/datafusion/physical-expr/src/expressions/cast.rs b/datafusion/physical-expr/src/expressions/cast.rs index 792687fe6acaa..3c13cbcf00a66 100644 --- a/datafusion/physical-expr/src/expressions/cast.rs +++ b/datafusion/physical-expr/src/expressions/cast.rs @@ -33,7 +33,7 @@ use datafusion_common::nested_struct::{ use datafusion_common::{Result, not_impl_err}; use datafusion_expr_common::columnar_value::ColumnarValue; use datafusion_expr_common::interval_arithmetic::Interval; -use datafusion_expr_common::sort_properties::ExprProperties; +use datafusion_expr_common::sort_properties::{ExprProperties, SortProperties}; const DEFAULT_CAST_OPTIONS: CastOptions<'static> = CastOptions { safe: false, @@ -341,12 +341,16 @@ pub(crate) fn cast_expr_properties( if bigger_cast || (!null_on_failure && is_order_preserving_cast(&source_type, target_type)) { + let strictly_order_preserving = child.strictly_order_preserving && bigger_cast; + let sort_properties = match child.sort_properties { + SortProperties::Grouped if !bigger_cast => SortProperties::Unordered, + sort_properties => sort_properties, + }; Ok(child .clone() + .with_order(sort_properties) .with_range(unbounded) - .with_strictly_order_preserving( - child.strictly_order_preserving && bigger_cast, - )) + .with_strictly_order_preserving(strictly_order_preserving)) } else { Ok(ExprProperties::new_unknown().with_range(unbounded)) } @@ -2094,6 +2098,20 @@ mod tests { assert!(!CastExpr::check_bigger_cast(&Int64, &UInt64)); assert!(!CastExpr::check_bigger_cast(&Int8, &UInt16)); } + + #[test] + fn grouped_cast_requires_an_injective_conversion() -> Result<()> { + let grouped = ExprProperties::new_unknown() + .with_order(SortProperties::Grouped) + .with_range(Interval::make_unbounded(&Int32)?); + + let widened = cast_expr_properties(&grouped, &Int64, false)?; + assert_eq!(widened.sort_properties, SortProperties::Grouped); + + let narrowed = cast_expr_properties(&grouped, &Int8, false)?; + assert_eq!(narrowed.sort_properties, SortProperties::Unordered); + Ok(()) + } } /// Tests for the `try_to_proto` / `try_from_proto` hooks. diff --git a/datafusion/physical-optimizer/src/join_selection.rs b/datafusion/physical-optimizer/src/join_selection.rs index 2b342bece7040..a8a5ee3749579 100644 --- a/datafusion/physical-optimizer/src/join_selection.rs +++ b/datafusion/physical-optimizer/src/join_selection.rs @@ -441,8 +441,11 @@ fn hash_join_convert_symmetric_subrule( let name = schema.field(*index).name(); let col = Arc::new(Column::new(name, *index)) as _; // Check if the column is ordered. - equivalence.get_expr_properties(col).sort_properties - != SortProperties::Unordered + matches!( + equivalence.get_expr_properties(col).sort_properties, + SortProperties::Ordered(_) + | SortProperties::Singleton + ) }, ) }) diff --git a/datafusion/physical-plan/benches/dictionary_group_values.rs b/datafusion/physical-plan/benches/dictionary_group_values.rs index 502869787ee40..51363c0bdffe3 100644 --- a/datafusion/physical-plan/benches/dictionary_group_values.rs +++ b/datafusion/physical-plan/benches/dictionary_group_values.rs @@ -29,7 +29,7 @@ use criterion::{ }; use datafusion_expr::EmitTo; use datafusion_physical_plan::aggregates::group_values::new_group_values; -use datafusion_physical_plan::aggregates::order::{GroupOrdering, GroupOrderingFull}; +use datafusion_physical_plan::aggregates::order::{GroupClustering, GroupClusteringFull}; use rand::rngs::StdRng; use rand::seq::SliceRandom; use rand::{Rng, SeedableRng}; @@ -106,7 +106,7 @@ fn bench_intern_emit(c: &mut Criterion) { b.iter_batched_ref( || { ( - new_group_values(schema.clone(), &GroupOrdering::None) + new_group_values(schema.clone(), &GroupClustering::None) .unwrap(), Vec::::with_capacity(size), ) @@ -151,7 +151,7 @@ fn bench_repeated_intern_emit(c: &mut Criterion) { b.iter_batched_ref( || { ( - new_group_values(schema.clone(), &GroupOrdering::None) + new_group_values(schema.clone(), &GroupClustering::None) .unwrap(), Vec::::with_capacity(size), ) @@ -172,7 +172,7 @@ fn bench_repeated_intern_emit(c: &mut Criterion) { group.finish(); } -// GroupOrdering::Full -> GroupValuesColumn::: scalar append_val/equal_to path. +// GroupClustering::Full -> GroupValuesColumn::: scalar append_val/equal_to path. fn bench_scalar_append_equal(c: &mut Criterion) { let mut group = c.benchmark_group("dict_scalar_append_equal"); let schema = dict_schema(); @@ -192,7 +192,7 @@ fn bench_scalar_append_equal(c: &mut Criterion) { ( new_group_values( schema.clone(), - &GroupOrdering::Full(GroupOrderingFull::new()), + &GroupClustering::Full(GroupClusteringFull::new()), ) .unwrap(), Vec::::with_capacity(size), @@ -227,7 +227,7 @@ fn bench_take_n(c: &mut Criterion) { b.iter_batched_ref( || { ( - new_group_values(schema.clone(), &GroupOrdering::None).unwrap(), + new_group_values(schema.clone(), &GroupClustering::None).unwrap(), Vec::::with_capacity(size), ) }, @@ -298,7 +298,7 @@ fn bench_shared_values_arc(c: &mut Criterion) { b.iter_batched_ref( || { ( - new_group_values(schema.clone(), &GroupOrdering::None) + new_group_values(schema.clone(), &GroupClustering::None) .unwrap(), Vec::::with_capacity(size), ) diff --git a/datafusion/physical-plan/benches/ordered_group_values.rs b/datafusion/physical-plan/benches/ordered_group_values.rs index 30af725cddcfc..fc60423c8be92 100644 --- a/datafusion/physical-plan/benches/ordered_group_values.rs +++ b/datafusion/physical-plan/benches/ordered_group_values.rs @@ -38,12 +38,12 @@ use datafusion_physical_expr::expressions::col; use datafusion_physical_expr::{LexOrdering, PhysicalSortExpr}; use datafusion_physical_plan::aggregates::group_values::multi_group_by::GroupValuesColumn; use datafusion_physical_plan::aggregates::group_values::{GroupValues, new_group_values}; -use datafusion_physical_plan::aggregates::order::GroupOrdering; +use datafusion_physical_plan::aggregates::order::GroupClustering; use datafusion_physical_plan::aggregates::{ - AggregateExec, AggregateMode, PhysicalGroupBy, + AggregateExec, AggregateMode, GroupClusteringMode, PhysicalGroupBy, }; use datafusion_physical_plan::test::TestMemoryExec; -use datafusion_physical_plan::{ExecutionPlan, InputOrderMode, collect}; +use datafusion_physical_plan::{ExecutionPlan, collect}; use tokio::runtime::Runtime; const ROWS: usize = 131_072; @@ -112,8 +112,10 @@ fn grouping(c: &mut Criterion) { let values: Box = if selected { new_group_values( Arc::clone(&schema), - &GroupOrdering::try_new(&InputOrderMode::Sorted) - .unwrap(), + &GroupClustering::try_new( + &GroupClusteringMode::Full, + ) + .unwrap(), ) .unwrap() } else { @@ -158,7 +160,7 @@ fn check_case(schema: &SchemaRef, batches: &[Vec]) -> (usize, usize) { let mut hashed = GroupValuesColumn::::try_new(Arc::clone(schema)).unwrap(); let mut selected = new_group_values( Arc::clone(schema), - &GroupOrdering::try_new(&InputOrderMode::Sorted).unwrap(), + &GroupClustering::try_new(&GroupClusteringMode::Full).unwrap(), ) .unwrap(); let mut expected = Vec::new(); @@ -188,7 +190,7 @@ fn aggregate_plan( schema: &SchemaRef, keys: Vec>, sort_columns: &[&str], - input_order_mode: &InputOrderMode, + group_clustering_mode: &GroupClusteringMode, ) -> Arc { let mut fields = schema.fields().to_vec(); fields.push(Arc::new(Field::new("v", DataType::Int64, false))); @@ -232,7 +234,7 @@ fn aggregate_plan( schema, ) .unwrap(); - assert_eq!(plan.input_order_mode(), input_order_mode); + assert_eq!(plan.group_clustering_mode(), group_clustering_mode); Arc::new(plan) } @@ -246,7 +248,7 @@ fn aggregation(c: &mut Criterion) { for run_length in [1, 8, 128, 8192] { let (schema, keys) = inputs(run_length, 8192, strings); let plan = - aggregate_plan(&schema, keys, &["a", "b"], &InputOrderMode::Sorted); + aggregate_plan(&schema, keys, &["a", "b"], &GroupClusteringMode::Full); let name = format!("{}_run{run_length}", if strings { "string" } else { "int" }); group.bench_function(name, |b| { @@ -291,7 +293,7 @@ fn partially_ordered_aggregation(c: &mut Criterion) { &schema, keys, &["a"], - &InputOrderMode::PartiallySorted(vec![0]), + &GroupClusteringMode::Partial(vec![0]), ); let runtime = Runtime::new().unwrap(); let mut group = c.benchmark_group("partially_ordered_aggregate_exec"); diff --git a/datafusion/physical-plan/benches/partial_ordering.rs b/datafusion/physical-plan/benches/partial_ordering.rs index bdadd6274b75e..967e84d78a88b 100644 --- a/datafusion/physical-plan/benches/partial_ordering.rs +++ b/datafusion/physical-plan/benches/partial_ordering.rs @@ -18,7 +18,7 @@ use std::sync::Arc; use arrow::array::{ArrayRef, Int32Array}; -use datafusion_physical_plan::aggregates::order::GroupOrderingPartial; +use datafusion_physical_plan::aggregates::order::GroupClusteringPartial; use criterion::{Criterion, criterion_group, criterion_main}; @@ -34,20 +34,20 @@ fn create_test_arrays(num_columns: usize) -> Vec { .collect() } fn bench_new_groups(c: &mut Criterion) { - let mut group = c.benchmark_group("group_ordering_partial"); + let mut group = c.benchmark_group("group_clustering_partial"); - // Test with 1, 2, 4, and 8 order indices + // Test with 1, 2, 4, and 8 grouping indices for num_columns in [1, 2, 4, 8] { - let order_indices: Vec = (0..num_columns).collect(); + let grouping_indices: Vec = (0..num_columns).collect(); - group.bench_function(format!("order_indices_{num_columns}"), |b| { + group.bench_function(format!("grouping_indices_{num_columns}"), |b| { let batch_group_values = create_test_arrays(num_columns); let group_indices: Vec = (0..BATCH_SIZE).collect(); b.iter(|| { - let mut ordering = - GroupOrderingPartial::try_new(order_indices.clone()).unwrap(); - ordering + let mut clustering = + GroupClusteringPartial::try_new(grouping_indices.clone()).unwrap(); + clustering .new_groups(&batch_group_values, &group_indices, BATCH_SIZE) .unwrap(); }); diff --git a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/ordered_final_table.rs b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_final_table.rs similarity index 76% rename from datafusion/physical-plan/src/aggregates/aggregate_hash_table/ordered_final_table.rs rename to datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_final_table.rs index 9cf6497e4843e..e2bad901f0760 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/ordered_final_table.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_final_table.rs @@ -15,9 +15,9 @@ // specific language governing permissions and limitations // under the License. -//! Aggregate table for final aggregation when partial-state input is ordered. +//! Aggregate table for final aggregation when partial-state input is clustered. //! -//! See comments in [`super::ordered_partial_table`] for details. +//! See comments in [`super::clustered_partial_table`] for details. use std::sync::Arc; @@ -25,12 +25,12 @@ use arrow::datatypes::SchemaRef; use arrow::record_batch::RecordBatch; use datafusion_common::Result; -use crate::InputOrderMode; use crate::aggregates::aggregate_hash_table::FinalMarker; +use crate::aggregates::order::GroupClusteringMode; use crate::aggregates::{AggregateExec, AggregateMode, group_values::AccumulatorPhase}; use super::common::HashAggregateAccumulator; -use super::common_ordered::{OrderedAggregateTable, OrderedAggregateTableMetrics}; +use super::common_clustered::{ClusteredAggregateTable, ClusteredAggregateTableMetrics}; /// Implementation specific to final aggregation, where the table stores partial /// aggregate states and the input rows are also partial states. @@ -40,28 +40,28 @@ use super::common_ordered::{OrderedAggregateTable, OrderedAggregateTableMetrics} /// - Aggregate table stores: `k, sum(x), count(x)` /// - Input rows: `k, sum(x), count(x)` /// -/// See comments at [`OrderedAggregateTable`] for details. -impl OrderedAggregateTable { - pub(in crate::aggregates) fn new_with_input_order( +/// See comments at [`ClusteredAggregateTable`] for details. +impl ClusteredAggregateTable { + pub(in crate::aggregates) fn new_with_group_clustering( agg: &AggregateExec, input_schema: &SchemaRef, output_schema: SchemaRef, - input_order_mode: &InputOrderMode, - metrics: OrderedAggregateTableMetrics, + group_clustering_mode: &GroupClusteringMode, + metrics: ClusteredAggregateTableMetrics, ) -> Result { Self::new_for_mode( agg, input_schema, output_schema, Arc::clone(input_schema), - input_order_mode, + group_clustering_mode, &AggregateMode::Final, vec![None; agg.aggr_expr().len()], metrics, ) } - /// Merges one partial-state input batch and updates ordering information for + /// Merges one partial-state input batch and updates completion state for /// any newly observed groups. pub(in crate::aggregates) fn aggregate_batch( &mut self, @@ -69,7 +69,7 @@ impl OrderedAggregateTable { ) -> Result<()> { let evaluated_batch = self.evaluate_batch(batch)?; // `PhysicalGroupBy::as_final()` removes grouping sets while planning - // final aggregation, so final ordered aggregation sees one grouping. + // final aggregation, so clustered final aggregation sees one grouping. debug_assert_eq!(evaluated_batch.grouping_set_args.len(), 1); self.aggregate_evaluated_batch( &evaluated_batch, @@ -78,8 +78,8 @@ impl OrderedAggregateTable { ) } - /// Materializes final results for all groups proven complete by the input - /// ordering, leaving the active ordered-key range in the table. + /// Materializes final results for all completed groups, leaving + /// the active contiguous-key range in the table. /// /// Returns None if there are no completed groups. pub(in crate::aggregates) fn take_completed_result_batch( @@ -88,7 +88,7 @@ impl OrderedAggregateTable { if self.is_empty() { return Ok(None); } - let Some(emit_to) = self.group_ordering().emit_to() else { + let Some(emit_to) = self.group_clustering().emit_to() else { return Ok(None); }; self.materialize_groups( diff --git a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/ordered_partial_table.rs b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_partial_table.rs similarity index 76% rename from datafusion/physical-plan/src/aggregates/aggregate_hash_table/ordered_partial_table.rs rename to datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_partial_table.rs index 3756c4b8e7868..fef9659fc2165 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/ordered_partial_table.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_partial_table.rs @@ -15,13 +15,13 @@ // specific language governing permissions and limitations // under the License. -//! Aggregate table for partial aggregation when input is ordered by group keys. +//! Aggregate table for partial aggregation when input is clustered by group keys. //! -//! See the [`super::common_ordered`] comments for the high-level ideas. +//! See the [`super::common_clustered`] comments for the high-level ideas. //! -//! This operator handles input that is ordered by group keys: -//! - Fully ordered: `GROUP BY a, b`, input is `ORDER BY a, b` -//! - Partially ordered: `GROUP BY a, b`, input is `ORDER BY a` +//! Ordering can establish either group-clustering mode: +//! - Full: `GROUP BY a, b`, input is `ORDER BY a, b` +//! - Partial: `GROUP BY a, b`, input is `ORDER BY a` //! //! When a group key combination is exhausted, this table eagerly flushes the //! completed groups to improve memory efficiency. @@ -41,7 +41,7 @@ use crate::aggregates::{ }; use super::common::HashAggregateAccumulator; -use super::common_ordered::{OrderedAggregateTable, OrderedAggregateTableMetrics}; +use super::common_clustered::{ClusteredAggregateTable, ClusteredAggregateTableMetrics}; /// Implementation specific to partial aggregation, where the table stores /// partial aggregate states and the input rows are raw rows. @@ -51,8 +51,8 @@ use super::common_ordered::{OrderedAggregateTable, OrderedAggregateTableMetrics} /// - Aggregate table stores: `k, sum(x), count(x)` /// - Input rows: `k, x` /// -/// See comments at [`OrderedAggregateTable`] for details. -impl OrderedAggregateTable { +/// See comments at [`ClusteredAggregateTable`] for details. +impl ClusteredAggregateTable { pub(in crate::aggregates) fn new( agg: &AggregateExec, partition: usize, @@ -60,20 +60,20 @@ impl OrderedAggregateTable { ) -> Result { let input_schema = agg.input().schema(); let state_schema = Arc::clone(&output_schema); - let metrics = OrderedAggregateTableMetrics::new(agg, partition); + let metrics = ClusteredAggregateTableMetrics::new(agg, partition); Self::new_for_mode( agg, &input_schema, output_schema, state_schema, - &agg.input_order_mode, + &agg.group_clustering_mode, &AggregateMode::Partial, agg.filter_expr().to_vec(), metrics, ) } - /// Aggregates one raw input batch and updates ordering information for any + /// Aggregates one raw input batch and updates completion state for any /// newly observed groups. pub(in crate::aggregates) fn aggregate_batch( &mut self, @@ -87,15 +87,15 @@ impl OrderedAggregateTable { ) } - /// Materializes all groups proven complete by the input ordering, leaving - /// the active ordered-key range in the table. + /// Materializes all completed groups, leaving + /// the active contiguous-key range in the table. pub(in crate::aggregates) fn take_completed_state_batch( &mut self, ) -> Result> { if self.is_empty() { return Ok(None); } - let Some(emit_to) = self.group_ordering().emit_to() else { + let Some(emit_to) = self.group_clustering().emit_to() else { return Ok(None); }; self.materialize_groups( diff --git a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/ordered_single_table.rs b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_single_table.rs similarity index 79% rename from datafusion/physical-plan/src/aggregates/aggregate_hash_table/ordered_single_table.rs rename to datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_single_table.rs index 89859bc036d83..c8ecbf4f7f154 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/ordered_single_table.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/clustered_single_table.rs @@ -15,9 +15,9 @@ // specific language governing permissions and limitations // under the License. -//! Aggregate table for single aggregation when raw input is ordered. +//! Aggregate table for single aggregation when raw input is clustered. //! -//! See comments in [`super::ordered_partial_table`] for details. +//! See comments in [`super::clustered_partial_table`] for details. use arrow::datatypes::SchemaRef; use arrow::record_batch::RecordBatch; @@ -27,7 +27,7 @@ use crate::aggregates::aggregate_hash_table::SingleMarker; use crate::aggregates::{AggregateExec, AggregateMode, group_values::AccumulatorPhase}; use super::common::HashAggregateAccumulator; -use super::common_ordered::{OrderedAggregateTable, OrderedAggregateTableMetrics}; +use super::common_clustered::{ClusteredAggregateTable, ClusteredAggregateTableMetrics}; /// Implementation specific to single aggregation, where the table stores final /// aggregate values and the input rows are raw rows. @@ -37,8 +37,8 @@ use super::common_ordered::{OrderedAggregateTable, OrderedAggregateTableMetrics} /// - Aggregate table stores: `k, avg(x)` /// - Input rows: `k, x` /// -/// See comments at [`OrderedAggregateTable`] for details. -impl OrderedAggregateTable { +/// See comments at [`ClusteredAggregateTable`] for details. +impl ClusteredAggregateTable { pub(in crate::aggregates) fn new( agg: &AggregateExec, partition: usize, @@ -51,20 +51,20 @@ impl OrderedAggregateTable { )); let input_schema = agg.input().schema(); - let metrics = OrderedAggregateTableMetrics::new(agg, partition); + let metrics = ClusteredAggregateTableMetrics::new(agg, partition); Self::new_for_mode( agg, &input_schema, output_schema, state_schema, - &agg.input_order_mode, + &agg.group_clustering_mode, &agg.mode, agg.filter_expr().to_vec(), metrics, ) } - /// Aggregates one raw input batch and updates ordering information for any + /// Aggregates one raw input batch and updates completion state for any /// newly observed groups. pub(in crate::aggregates) fn aggregate_batch( &mut self, @@ -78,8 +78,8 @@ impl OrderedAggregateTable { ) } - /// Materializes final results for all groups proven complete by the input - /// ordering, leaving the active ordered-key range in the table. + /// Materializes final results for all completed groups, leaving + /// the active contiguous-key range in the table. /// /// Returns None if there are no completed groups. pub(in crate::aggregates) fn take_completed_result_batch( @@ -88,7 +88,7 @@ impl OrderedAggregateTable { if self.is_empty() { return Ok(None); } - let Some(emit_to) = self.group_ordering().emit_to() else { + let Some(emit_to) = self.group_clustering().emit_to() else { return Ok(None); }; self.materialize_groups( diff --git a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common.rs b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common.rs index b25caf815eed1..a40c5e5fae693 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common.rs @@ -38,7 +38,7 @@ use crate::aggregates::group_values::{ AccumulatorPhase, AggregateAccumulatorMetrics, AggregateArgumentMetrics, GroupByMetrics, GroupValues, new_group_values, }; -use crate::aggregates::order::GroupOrdering; +use crate::aggregates::order::GroupClustering; use crate::aggregates::{ AggregateExec, PhysicalGroupBy, aggregate_expressions, evaluate_group_by, group_id_array, max_duplicate_ordinal, @@ -180,7 +180,7 @@ impl AggregateHashTable { .collect::>()?; let group_schema = agg.group_by().group_schema(&input_schema)?; - let group_values = new_group_values(group_schema, &GroupOrdering::None)?; + let group_values = new_group_values(group_schema, &GroupClustering::None)?; Ok(Self { group_by_metrics: metrics.group_by, diff --git a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common_ordered.rs b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common_clustered.rs similarity index 84% rename from datafusion/physical-plan/src/aggregates/aggregate_hash_table/common_ordered.rs rename to datafusion/physical-plan/src/aggregates/aggregate_hash_table/common_clustered.rs index 86c052009faea..4cb5796a54f4c 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common_ordered.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common_clustered.rs @@ -15,8 +15,7 @@ // specific language governing permissions and limitations // under the License. -//! Common utilities for aggregate tables used in aggregations that inputs are ordered -//! by the groups. +//! Common utilities for aggregate tables with input clustered by group keys. use std::marker::PhantomData; use std::sync::Arc; @@ -27,13 +26,12 @@ use datafusion_common::Result; use datafusion_execution::memory_pool::proxy::VecAllocExt; use datafusion_expr::{AggregateMetrics, EmitTo}; -use crate::InputOrderMode; use crate::PhysicalExpr; use crate::aggregates::group_values::{ AccumulatorPhase, AggregateAccumulatorMetrics, AggregateArgumentMetrics, GroupByMetrics, GroupValues, new_group_values, }; -use crate::aggregates::order::GroupOrdering; +use crate::aggregates::order::{GroupClustering, GroupClusteringMode}; use crate::aggregates::{ AggregateExec, AggregateMode, PhysicalGroupBy, aggregate_expressions, evaluate_group_by, @@ -46,14 +44,14 @@ use super::common::{ }; #[derive(Clone)] -pub(in crate::aggregates) struct OrderedAggregateTableMetrics { +pub(in crate::aggregates) struct ClusteredAggregateTableMetrics { pub(super) group_by: GroupByMetrics, pub(super) aggregate_arguments: AggregateArgumentMetrics, pub(super) accumulator: Arc, pub(super) submetrics: Vec>, } -impl OrderedAggregateTableMetrics { +impl ClusteredAggregateTableMetrics { pub(in crate::aggregates) fn new(agg: &AggregateExec, partition: usize) -> Self { let metrics = AggregateTableMetrics::new(agg, partition); Self { @@ -76,20 +74,20 @@ impl OrderedAggregateTableMetrics { } } -/// Aggregate table shared by the ordered single, partial and final paths. +/// Aggregate table shared by the clustered single, partial and final paths. /// -/// # Ordering optimization +/// # Group clustering optimization /// -/// The table consumes input batches while `GroupOrdering` tracks which groups +/// The table consumes input batches while [`GroupClustering`] tracks which groups /// are proven complete. Completed groups can be emitted before the input stream -/// ends, which keeps memory bounded by the active ordered key range. +/// ends, so completed groups no longer occupy the table. /// /// # Single, partial and final variant difference /// /// The partial and final aggregate tables implement the two stages of grouped /// aggregation, while the single aggregate table implements both stages in one /// table. See -/// [`OrderedPartialAggregateStream`](crate::aggregates::ordered_partial_stream::OrderedPartialAggregateStream) +/// [`ClusteredPartialAggregateStream`](crate::aggregates::clustered_partial_stream::ClusteredPartialAggregateStream) /// for the high-level plan shape. /// /// Example: `AVG(v) FILTER (WHERE v>0) GROUP BY k` @@ -111,15 +109,15 @@ impl OrderedAggregateTableMetrics { /// /// # Marker Type /// -/// `OrderedAggrMode` selects the aggregate semantics. For example, -/// `OrderedAggregateTable::::new(...)` consumes raw rows +/// `AggrMode` selects the aggregate semantics. For example, +/// `ClusteredAggregateTable::::new(...)` consumes raw rows /// and emits partial states, while -/// `OrderedAggregateTable::::new_with_input_order(...)` +/// `ClusteredAggregateTable::::new_with_group_clustering(...)` /// consumes partial states and emits final values. /// /// Shared methods live on `impl`; single/partial/final behavior lives on /// marker-specific impls. -pub(in crate::aggregates) struct OrderedAggregateTable { +pub(in crate::aggregates) struct ClusteredAggregateTable { /// Output schema: group columns followed by aggregate state or final values. pub(super) output_schema: SchemaRef, @@ -139,26 +137,26 @@ pub(in crate::aggregates) struct OrderedAggregateTable { /// Optional internal metrics owned by each aggregate expression. pub(super) aggregate_submetrics: Vec>, - /// Group keys, ordering state, and accumulator states. - pub(super) buffer: OrderedAggregateTableBuffer, + /// Group keys, completion state, and accumulator states. + pub(super) buffer: ClusteredAggregateTableBuffer, - _mode: PhantomData, + _mode: PhantomData, } -/// Buffer for the ordered aggregate table's group keys and accumulator states. +/// Buffer for the clustered aggregate table's group keys and accumulator states. /// -/// It accumulates input during aggregation and emits output rows as soon as the -/// input ordering proves those groups are complete. +/// It accumulates input during aggregation and emits output rows as soon as +/// groups are known to be complete. /// -/// [`GroupOrdering`] tracks when and how to do early emit. +/// [`GroupClustering`] tracks when and how to do early emit. /// [`GroupValues`] stores the physical group-key layout, while /// [`datafusion_expr::GroupsAccumulator`] stores per-group aggregate state. -pub(super) struct OrderedAggregateTableBuffer { +pub(super) struct ClusteredAggregateTableBuffer { /// GROUP BY expressions evaluated against input batches. pub(super) group_by: Arc, - /// Tracks how far ordered input allows this table to drain safely. - pub(super) group_ordering: GroupOrdering, + /// Tracks which groups are complete and can be emitted safely. + pub(super) group_clustering: GroupClustering, /// Interned group keys, in the same group-id order used by accumulators. pub(super) group_values: Box, @@ -174,24 +172,24 @@ pub(super) struct OrderedAggregateTableBuffer { } /// Methods shared by all aggregate modes -impl OrderedAggregateTable { +impl ClusteredAggregateTable { #[expect( clippy::too_many_arguments, - reason = "keeps ordered single, partial and final table construction explicit" + reason = "keeps clustered single, partial and final table construction explicit" )] pub(super) fn new_for_mode( agg: &AggregateExec, input_schema: &SchemaRef, output_schema: SchemaRef, state_schema: SchemaRef, - input_order_mode: &InputOrderMode, + group_clustering_mode: &GroupClusteringMode, aggregate_mode: &AggregateMode, filters: Vec>>, - metrics: OrderedAggregateTableMetrics, + metrics: ClusteredAggregateTableMetrics, ) -> Result { - let group_ordering = GroupOrdering::try_new(input_order_mode)?; + let group_clustering = GroupClustering::try_new(group_clustering_mode)?; let group_schema = agg.group_by().group_schema(input_schema)?; - let group_values = new_group_values(group_schema, &group_ordering)?; + let group_values = new_group_values(group_schema, &group_clustering)?; let aggregate_arguments = aggregate_expressions( agg.aggr_expr(), aggregate_mode, @@ -223,9 +221,9 @@ impl OrderedAggregateTable { aggregate_argument_metrics: metrics.aggregate_arguments, aggregate_accumulator_metrics: metrics.accumulator, aggregate_submetrics: metrics.submetrics, - buffer: OrderedAggregateTableBuffer { + buffer: ClusteredAggregateTableBuffer { group_by: Arc::clone(agg.group_by()), - group_ordering, + group_clustering, group_values, group_indices: vec![], accumulators, @@ -268,15 +266,15 @@ impl OrderedAggregateTable { /// Called after the input stream is exhausted and the last batch has been /// aggregated. /// - /// Updates the internal `GroupOrdering` so it can continue emitting until + /// Updates the internal [`GroupClustering`] so it can continue emitting until /// the buffer is empty. pub(in crate::aggregates) fn input_done(&mut self) { - self.buffer.group_ordering.input_done(); + self.buffer.group_clustering.input_done(); } - /// Returns the ordering state used to decide how memory pressure is handled. - pub(in crate::aggregates) fn group_ordering(&self) -> &GroupOrdering { - &self.buffer.group_ordering + /// Returns the completion state used to decide how memory pressure is handled. + pub(in crate::aggregates) fn group_clustering(&self) -> &GroupClustering { + &self.buffer.group_clustering } /// Number of groups currently buffered. @@ -297,12 +295,12 @@ impl OrderedAggregateTable { .map(|acc| acc.size()) .sum::() + self.buffer.group_values.size() - + self.buffer.group_ordering.size() + + self.buffer.group_clustering.size() + self.buffer.group_indices.allocated_size() } - pub(in crate::aggregates) fn metrics(&self) -> OrderedAggregateTableMetrics { - OrderedAggregateTableMetrics { + pub(in crate::aggregates) fn metrics(&self) -> ClusteredAggregateTableMetrics { + ClusteredAggregateTableMetrics { group_by: self.group_by_metrics.clone(), aggregate_arguments: self.aggregate_argument_metrics.clone(), accumulator: Arc::clone(&self.aggregate_accumulator_metrics), @@ -311,9 +309,9 @@ impl OrderedAggregateTable { } /// Takes every intermediate aggregate state and resets the table so it can - /// continue with a new ordered input segment. + /// continue with a new clustered input segment. /// - /// Unlike normal ordered emission, this operation is allowed to take the + /// Unlike normal group-completion emission, this operation is allowed to take the /// active (incomplete) groups. Partial aggregation can pass those states to /// its final stage, while single and final aggregation sort and spill them /// before replay. @@ -346,7 +344,7 @@ impl OrderedAggregateTable { self.buffer.group_values.clear_shrink(0); self.buffer.group_indices.clear(); self.buffer.group_indices.shrink_to_fit(); - self.buffer.group_ordering.reset(); + self.buffer.group_clustering.reset(); Ok(Some(batch)) } @@ -374,7 +372,7 @@ impl OrderedAggregateTable { .intern(group_values, &mut self.buffer.group_indices)?; let total_num_groups = self.buffer.group_values.len(); if total_num_groups > starting_num_groups { - self.buffer.group_ordering.new_groups( + self.buffer.group_clustering.new_groups( group_values, &self.buffer.group_indices, total_num_groups, @@ -418,10 +416,10 @@ impl OrderedAggregateTable { let accumulator_metrics = Arc::clone(&self.aggregate_accumulator_metrics); let output = self.group_by_metrics.time_emitting(|| { let mut output = self.buffer.group_values.emit(emit_to)?; - // `EmitTo::All` is only used after `input_done`, when the ordering + // `EmitTo::All` is only used after `input_done`, when the completion // state no longer tracks group indexes. if let EmitTo::First(n) = emit_to { - self.buffer.group_ordering.remove_groups(n); + self.buffer.group_clustering.remove_groups(n); } for (idx, acc) in self.buffer.accumulators.iter_mut().enumerate() { diff --git a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/mod.rs b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/mod.rs index 960b7498f1eb2..a356ea5c79d10 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/mod.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/mod.rs @@ -15,12 +15,12 @@ // specific language governing permissions and limitations // under the License. +mod clustered_final_table; +mod clustered_partial_table; +mod clustered_single_table; mod common; -mod common_ordered; +mod common_clustered; mod final_table; -mod ordered_final_table; -mod ordered_partial_table; -mod ordered_single_table; mod partial_reduce_table; mod partial_table; mod single_table; @@ -101,7 +101,9 @@ pub(super) use common::{ AggregateHashTable, FinalMarker, PartialMarker, PartialReduceMarker, PartialSkipMarker, SingleMarker, create_group_accumulator, }; -pub(super) use common_ordered::{OrderedAggregateTable, OrderedAggregateTableMetrics}; +pub(super) use common_clustered::{ + ClusteredAggregateTable, ClusteredAggregateTableMetrics, +}; #[cfg(test)] mod tests { diff --git a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/partial_table.rs b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/partial_table.rs index 397b766f41697..9c95bec83e3c8 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/partial_table.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/partial_table.rs @@ -23,7 +23,7 @@ use arrow::record_batch::RecordBatch; use datafusion_common::{Result, assert_eq_or_internal_err}; use crate::aggregates::group_values::{AccumulatorPhase, new_group_values}; -use crate::aggregates::order::GroupOrdering; +use crate::aggregates::order::GroupClustering; use crate::aggregates::{AggregateExec, evaluate_group_by}; use super::common::{ @@ -78,7 +78,7 @@ impl AggregateHashTable { ) -> Result> { let state = self.state.building(); let group_schema = state.group_by.group_schema(&self.input_schema)?; - let group_values = new_group_values(group_schema, &GroupOrdering::None)?; + let group_values = new_group_values(group_schema, &GroupClustering::None)?; let accumulators = state .accumulators .iter() diff --git a/datafusion/physical-plan/src/aggregates/ordered_final_stream.rs b/datafusion/physical-plan/src/aggregates/clustered_final_stream.rs similarity index 89% rename from datafusion/physical-plan/src/aggregates/ordered_final_stream.rs rename to datafusion/physical-plan/src/aggregates/clustered_final_stream.rs index daa5a85323a85..9c71c5ee13328 100644 --- a/datafusion/physical-plan/src/aggregates/ordered_final_stream.rs +++ b/datafusion/physical-plan/src/aggregates/clustered_final_stream.rs @@ -15,7 +15,8 @@ // specific language governing permissions and limitations // under the License. -//! Final aggregate stream for ordered partial-state input. +//! Final aggregate stream for partial-state input with group-clustering +//! guarantees. use std::sync::Arc; @@ -28,24 +29,25 @@ use futures::stream::StreamExt; use super::AggregateExec; use super::aggregate_hash_table::{ - FinalMarker, OrderedAggregateTable, OrderedAggregateTableMetrics, + ClusteredAggregateTable, ClusteredAggregateTableMetrics, FinalMarker, }; +use super::order::GroupClusteringMode; use super::spill::AggregateSpill; +use crate::SendableRecordBatchStream; use crate::aggregates::AggregateMode; use crate::metrics::{BaselineMetrics, SpillMetrics}; use crate::stream::{ObservedStream, RecordBatchStreamAdapter}; -use crate::{InputOrderMode, SendableRecordBatchStream}; -/// Final aggregate stream for `InputOrderMode::Sorted` and -/// `InputOrderMode::PartiallySorted`. +/// Final aggregate stream for [`GroupClusteringMode::Partial`] and +/// [`GroupClusteringMode::Full`]. /// -/// See comments at [`super::ordered_partial_stream::OrderedPartialAggregateStream`] for details. +/// See comments at [`super::clustered_partial_stream::ClusteredPartialAggregateStream`] for details. /// /// # Spilling /// -/// This section is only for implementation notes, for background, see [`super::ordered_partial_stream::OrderedPartialAggregateStream`] +/// This section is only for implementation notes, for background, see [`super::clustered_partial_stream::ClusteredPartialAggregateStream`] /// -/// For partially sorted input, spilling works as follows: +/// For partial group clustering, spilling works as follows: /// /// - Reserve the table footprint plus one `u32` sort index per buffered group. The /// extra index array is used in later sorting before spilling. @@ -55,13 +57,13 @@ use crate::{InputOrderMode, SendableRecordBatchStream}; /// batch and full index remain live until the run is written. /// - After input ends, merge the sorted runs and replay them through a fully /// ordered final aggregate stream. -pub(crate) struct OrderedFinalAggregateStream { +pub(crate) struct ClusteredFinalAggregateStream { reservation: MemoryReservation, - context: OrderedFinalAggregateContext, + context: ClusteredFinalAggregateContext, stage: ExecutionStage, } -/// Execution stages described in [`OrderedFinalAggregateStream::into_stream`]. +/// Execution stages described in [`ClusteredFinalAggregateStream::into_stream`]. enum ExecutionStage { Aggregating(Aggregating), Outputting(Outputting), @@ -70,8 +72,8 @@ enum ExecutionStage { struct Aggregating { input: SendableRecordBatchStream, - table: OrderedAggregateTable, - /// None when temporary files are disabled or all group keys are ordered. + table: ClusteredAggregateTable, + /// None when temporary files are disabled or group clustering is full. spill_context: Option>, } @@ -83,13 +85,13 @@ struct Outputting { } /// Immutable execution context shared by aggregation and output emission. -struct OrderedFinalAggregateContext { +struct ClusteredFinalAggregateContext { schema: SchemaRef, batch_size: usize, baseline_metrics: BaselineMetrics, } -impl OrderedFinalAggregateStream { +impl ClusteredFinalAggregateStream { pub fn new( agg: &AggregateExec, context: &Arc, @@ -99,10 +101,10 @@ impl OrderedFinalAggregateStream { agg.mode, AggregateMode::Final | AggregateMode::FinalPartitioned )); - debug_assert_ne!(agg.input_order_mode, InputOrderMode::Linear); + debug_assert_ne!(agg.group_clustering_mode, GroupClusteringMode::None); let input = agg.input.execute(partition, Arc::clone(context))?; - Self::new_with_input(agg, context, partition, input, &agg.input_order_mode) + Self::new_with_input(agg, context, partition, input, &agg.group_clustering_mode) } pub(in crate::aggregates) fn new_with_input( @@ -110,15 +112,15 @@ impl OrderedFinalAggregateStream { context: &Arc, partition: usize, input: SendableRecordBatchStream, - input_order_mode: &InputOrderMode, + group_clustering_mode: &GroupClusteringMode, ) -> Result { let baseline_metrics = BaselineMetrics::new(&agg.metrics, partition); - let metrics = OrderedAggregateTableMetrics::new(agg, partition); + let metrics = ClusteredAggregateTableMetrics::new(agg, partition); let spill_metrics = SpillMetrics::new(&agg.metrics, partition); let reservation = - MemoryConsumer::new(format!("OrderedFinalAggregateStream[{partition}]")) - // HACK: Technically, fully ordered aggregate is a non-spillable - // consumer, since it uses bounded memory. There is a known race + MemoryConsumer::new(format!("ClusteredFinalAggregateStream[{partition}]")) + // HACK: Full group clustering uses a non-spilling execution + // path. There is a known race // condition bug, and we set it to spillable to let it have larger // memory budget to suppress the bug. // Bug issue: https://github.com/apache/datafusion/issues/17334 @@ -129,7 +131,7 @@ impl OrderedFinalAggregateStream { context, partition, input, - input_order_mode, + group_clustering_mode, baseline_metrics, metrics, Some(spill_metrics), @@ -149,9 +151,9 @@ impl OrderedFinalAggregateStream { context: &Arc, partition: usize, input: SendableRecordBatchStream, - input_order_mode: &InputOrderMode, + group_clustering_mode: &GroupClusteringMode, baseline_metrics: BaselineMetrics, - metrics: OrderedAggregateTableMetrics, + metrics: ClusteredAggregateTableMetrics, spill_metrics: Option, reservation: MemoryReservation, ) -> Result { @@ -159,25 +161,27 @@ impl OrderedFinalAggregateStream { agg.mode, AggregateMode::Final | AggregateMode::FinalPartitioned )); - debug_assert_ne!(*input_order_mode, InputOrderMode::Linear); + debug_assert_ne!(*group_clustering_mode, GroupClusteringMode::None); let schema = Arc::clone(&agg.schema); let input_schema = input.schema(); let batch_size = context.session_config().batch_size(); - let can_spill = matches!(input_order_mode, InputOrderMode::PartiallySorted(_)) + let can_spill = matches!(group_clustering_mode, GroupClusteringMode::Partial(_)) && context.runtime_env().disk_manager.tmp_files_enabled(); let spill_context = if can_spill { let Some(spill_metrics) = spill_metrics else { - return internal_err!("Spillable ordered final stream requires metrics"); + return internal_err!( + "Spillable clustered final stream requires metrics" + ); }; Some(Box::new(AggregateSpill::try_new( - "OrderedFinalAggregateSpill", + "ClusteredFinalAggregateSpill", agg, context, partition, batch_size, - input_order_mode, + group_clustering_mode, &input_schema, spill_metrics, )?)) @@ -185,11 +189,11 @@ impl OrderedFinalAggregateStream { None }; - let table = OrderedAggregateTable::::new_with_input_order( + let table = ClusteredAggregateTable::::new_with_group_clustering( agg, &input_schema, Arc::clone(&schema), - input_order_mode, + group_clustering_mode, metrics, )?; @@ -198,7 +202,7 @@ impl OrderedFinalAggregateStream { Ok(Self { reservation, - context: OrderedFinalAggregateContext { + context: ClusteredFinalAggregateContext { schema, batch_size, baseline_metrics, @@ -211,9 +215,9 @@ impl OrderedFinalAggregateStream { }) } - /// Entry point for the ordered final aggregate execution stages. + /// Entry point for the clustered final aggregate execution stages. /// - /// See [`OrderedFinalAggregateStream`] for high-level ideas. + /// See [`ClusteredFinalAggregateStream`] for high-level ideas. /// /// # Stage transition graph: /// @@ -248,9 +252,9 @@ impl OrderedFinalAggregateStream { /// /// ### Incremental output /// - /// See the [ordered partial aggregate notes] for details. + /// See the [clustered partial aggregate notes] for details. /// - /// [ordered partial aggregate notes]: super::ordered_partial_stream::OrderedPartialAggregateStream::into_stream + /// [clustered partial aggregate notes]: super::clustered_partial_stream::ClusteredPartialAggregateStream::into_stream /// /// ## Transition Edges /// @@ -259,7 +263,7 @@ impl OrderedFinalAggregateStream { /// - If memory fits and no groups are complete, continue reading input. /// - If OOM, spill. /// 3. Prepare output: - /// - Before any spill, ordering proves a prefix complete: materialize the + /// - Before any spill, a prefix of groups is complete: materialize the /// entire prefix once, retaining the input and active groups to resume /// aggregation. /// - At EOF without spills, materialize all remaining results and prepare @@ -312,7 +316,7 @@ impl Aggregating { fn reservation_size(&self) -> usize { let table_size = self.table.memory_size(); if self.spill_context.is_some() { - // See `OrderedFinalAggregateStream` for the spill memory estimate. + // See `ClusteredFinalAggregateStream` for the spill memory estimate. table_size .saturating_add(self.table.num_groups().saturating_mul(size_of::())) } else { @@ -323,7 +327,7 @@ impl Aggregating { /// Merges partial states until final results are ready or spill replay begins. async fn handle_stage( mut self, - context: &OrderedFinalAggregateContext, + context: &ClusteredFinalAggregateContext, reservation: &MemoryReservation, ) -> Result> { let elapsed_compute = context.baseline_metrics.elapsed_compute(); @@ -406,7 +410,7 @@ impl Outputting { /// Emits slices of one materialized batch without touching the aggregate table. async fn handle_stage( self, - context: &OrderedFinalAggregateContext, + context: &ClusteredFinalAggregateContext, reservation: &MemoryReservation, emitter: &mut TryEmitter, ) -> Result> { @@ -454,6 +458,7 @@ impl Outputting { mod tests { use super::*; use crate::ExecutionPlan; + use crate::aggregates::GroupClusteringMode; use crate::aggregates::PhysicalGroupBy; use crate::common::collect; use crate::test::TestMemoryExec; @@ -556,8 +561,8 @@ mod tests { schema, )?; assert_eq!( - aggregate.input_order_mode(), - &InputOrderMode::PartiallySorted(vec![0]) + aggregate.group_clustering_mode(), + &GroupClusteringMode::Partial(vec![0]) ); let pool: Arc = Arc::new(GreedyMemoryPool::new(limit)); @@ -582,12 +587,12 @@ mod tests { Arc::clone(&partial_schema), receiver, )); - let stream = OrderedFinalAggregateStream::new_with_input( + let stream = ClusteredFinalAggregateStream::new_with_input( &aggregate, &context, partition, input, - aggregate.input_order_mode(), + &aggregate.group_clustering_mode, )?; Ok((sender, stream.into_stream())) }; diff --git a/datafusion/physical-plan/src/aggregates/ordered_partial_stream.rs b/datafusion/physical-plan/src/aggregates/clustered_partial_stream.rs similarity index 81% rename from datafusion/physical-plan/src/aggregates/ordered_partial_stream.rs rename to datafusion/physical-plan/src/aggregates/clustered_partial_stream.rs index 83a25620cd328..f7f7e2d9031dd 100644 --- a/datafusion/physical-plan/src/aggregates/ordered_partial_stream.rs +++ b/datafusion/physical-plan/src/aggregates/clustered_partial_stream.rs @@ -15,7 +15,7 @@ // specific language governing permissions and limitations // under the License. -//! Partial aggregate stream for ordered group input. +//! Partial aggregate stream for input with group-clustering guarantees. use std::sync::Arc; @@ -27,27 +27,27 @@ use datafusion_execution::{TaskContext, TryEmitter, async_try_stream}; use futures::stream::StreamExt; use super::AggregateExec; -use super::aggregate_hash_table::{OrderedAggregateTable, PartialMarker}; +use super::aggregate_hash_table::{ClusteredAggregateTable, PartialMarker}; use crate::aggregates::AggregateMode; -use crate::aggregates::order::GroupOrdering; +use crate::aggregates::order::{GroupClustering, GroupClusteringMode}; use crate::metrics::{BaselineMetrics, MetricBuilder, SpillMetrics}; use crate::stream::{ObservedStream, RecordBatchStreamAdapter}; -use crate::{InputOrderMode, SendableRecordBatchStream, metrics}; +use crate::{SendableRecordBatchStream, metrics}; -/// Partial aggregate stream for `InputOrderMode::Sorted` and -/// `InputOrderMode::PartiallySorted`. +/// Partial aggregate stream for [`GroupClusteringMode::Partial`] and +/// [`GroupClusteringMode::Full`]. /// /// # Example /// /// SELECT k, AVG(v) FROM t GROUP BY k; /// -/// If the input is ordered by `k`, the aggregate can use ordered partial and +/// If the input is ordered by `k`, the aggregate can use clustered partial and /// final stages: /// /// ## Plan -/// AggregateExec(stage=final, ordered) +/// AggregateExec(stage=final, clustered) /// -- RepartitionExec(hash(k), preserves_order=true) -/// ---- AggregateExec(stage=partial, ordered) +/// ---- AggregateExec(stage=partial, clustered) /// /// ## Partial Stage Behavior /// Input: raw rows @@ -59,22 +59,24 @@ use crate::{InputOrderMode, SendableRecordBatchStream, metrics}; /// Output: results for all groups (for example, `AVG(x)` calculated from the /// state) /// -/// # Order-based Optimization +/// # Group Clustering Optimization /// /// For the aggregation work, the hash aggregation implementation is reused. /// -/// After each input batch, check whether any groups can be emitted eagerly to -/// improve memory efficiency. For example, if the last group key seen is -/// `k = 100`, it is safe to emit all groups with keys less than 100 because the -/// input is ordered. Materialize that entire completed prefix once, then emit -/// slices of it before reading more input. This avoids repeatedly removing small -/// batches of groups and shifting the remaining hash table and accumulator state. +/// After each input batch, the group-clustering mode determines whether any +/// groups can be emitted eagerly to improve memory efficiency. For example, if +/// the input is ordered by `k` and the last group key seen is `k = 100`, all +/// groups with keys less than 100 are complete. Materialize that entire completed +/// prefix once, then emit slices of it before reading more input. This avoids +/// repeatedly removing small batches of groups and shifting the remaining hash +/// table and accumulator state. /// /// # Memory Pressure and Spilling /// -/// ## Fully ordered case +/// ## Full group clustering /// -/// If the input is ordered by every group key, for example: +/// Every complete grouping tuple is contiguous. Ordering by every group key is +/// one way to establish this mode, for example: /// /// - Input order: `a, b` /// - `GROUP BY`: `a, b` @@ -86,9 +88,10 @@ use crate::{InputOrderMode, SendableRecordBatchStream, metrics}; /// If a memory reservation nevertheless fails, the stream returns the error /// directly, indicating an unexpected behavior. /// -/// ## Partially ordered case +/// ## Partial group clustering /// -/// If the input is ordered by only a subset of the group keys, for example: +/// Rows are contiguous for a subset of the group keys. Ordering by that subset +/// is one way to establish this mode, for example: /// /// - Input order: `a` /// - `GROUP BY`: `a, b` @@ -96,21 +99,21 @@ use crate::{InputOrderMode, SendableRecordBatchStream, metrics}; /// If one `a` value contains many distinct `b` values, the table may accumulate /// enough groups to exceed the memory limit. /// -/// - `OrderedPartialAggregateStream`: On reservation failure, it emits all current +/// - `ClusteredPartialAggregateStream`: On reservation failure, it emits all current /// intermediate states downstream and resets the table. The final stage can /// merge repeated `(a, b)` state rows, so no disk spill is required. -/// - `OrderedFinalAggregateStream`: It cannot emit incomplete final results. On +/// - `ClusteredFinalAggregateStream`: It cannot emit incomplete final results. On /// reservation failure, it sorts the current intermediate states by the complete /// group key and spills them as one run. After the input ends, it spills any /// remaining states, performs a sort-preserving merge of all runs, and feeds the -/// merged input into a fully ordered final aggregate stream. -pub(crate) struct OrderedPartialAggregateStream { +/// merged input into a fully clustered final aggregate stream. +pub(crate) struct ClusteredPartialAggregateStream { reservation: MemoryReservation, - context: OrderedPartialAggregateContext, + context: ClusteredPartialAggregateContext, stage: ExecutionStage, } -/// Execution stages described in [`OrderedPartialAggregateStream::into_stream`]. +/// Execution stages described in [`ClusteredPartialAggregateStream::into_stream`]. enum ExecutionStage { Aggregating(Aggregating), Outputting(Outputting), @@ -118,7 +121,7 @@ enum ExecutionStage { struct Aggregating { input: SendableRecordBatchStream, - table: OrderedAggregateTable, + table: ClusteredAggregateTable, } struct Outputting { @@ -130,21 +133,21 @@ struct Outputting { } /// Immutable execution context shared by aggregation and output emission. -struct OrderedPartialAggregateContext { +struct ClusteredPartialAggregateContext { schema: SchemaRef, batch_size: usize, baseline_metrics: BaselineMetrics, reduction_factor: metrics::RatioMetrics, } -impl OrderedPartialAggregateStream { +impl ClusteredPartialAggregateStream { pub fn new( agg: &AggregateExec, context: &Arc, partition: usize, ) -> Result { debug_assert_eq!(agg.mode, AggregateMode::Partial); - debug_assert_ne!(agg.input_order_mode, InputOrderMode::Linear); + debug_assert_ne!(agg.group_clustering_mode, GroupClusteringMode::None); let schema = Arc::clone(&agg.schema); let input = agg.input.execute(partition, Arc::clone(context))?; @@ -157,16 +160,16 @@ impl OrderedPartialAggregateStream { .with_type(metrics::MetricType::Summary) .ratio_metrics("reduction_factor", partition); - let table = OrderedAggregateTable::::new( + let table = ClusteredAggregateTable::::new( agg, partition, Arc::clone(&schema), )?; let reservation = - MemoryConsumer::new(format!("OrderedPartialAggregateStream[{partition}]")) + MemoryConsumer::new(format!("ClusteredPartialAggregateStream[{partition}]")) .with_can_spill(matches!( - table.group_ordering(), - GroupOrdering::Partial(_) + table.group_clustering(), + GroupClustering::Partial(_) )) .register(context.memory_pool()); @@ -175,7 +178,7 @@ impl OrderedPartialAggregateStream { Ok(Self { reservation, - context: OrderedPartialAggregateContext { + context: ClusteredPartialAggregateContext { schema, batch_size, baseline_metrics, @@ -185,9 +188,9 @@ impl OrderedPartialAggregateStream { }) } - /// Entry point for the ordered partial aggregate execution stages. + /// Entry point for the clustered partial aggregate execution stages. /// - /// See [`OrderedPartialAggregateStream`] for high-level ideas. + /// See [`ClusteredPartialAggregateStream`] for high-level ideas. /// /// # Stage transition graph: /// @@ -256,10 +259,10 @@ impl OrderedPartialAggregateStream { /// 2. Aggregate one input batch. If memory fits and no groups are complete, /// continue reading input. /// 3. Prepare output: - /// - Ordering proves a prefix complete: materialize the entire prefix once, + /// - A prefix of groups is complete: materialize the entire prefix once, /// retaining the input and active groups to resume aggregation. - /// - On memory pressure with partial ordering, materialize all current - /// states instead, including incomplete groups, and reset the table. + /// - On memory pressure with partial group clustering, materialize all + /// current states, including incomplete groups, and reset the table. /// - At EOF, materialize all remaining states and prepare to output. /// 4. Input was exhausted with no remaining groups, directly end. /// 5. Yield one slice without materializing the table again. Keep the shared @@ -299,10 +302,10 @@ impl OrderedPartialAggregateStream { impl Aggregating { /// Aggregates raw input and materializes one batch of partial states. /// - /// See [`OrderedPartialAggregateStream::into_stream`] for stage transitions. + /// See [`ClusteredPartialAggregateStream::into_stream`] for stage transitions. async fn handle_stage( mut self, - context: &OrderedPartialAggregateContext, + context: &ClusteredPartialAggregateContext, reservation: &MemoryReservation, ) -> Result> { let elapsed_compute = context.baseline_metrics.elapsed_compute(); @@ -315,9 +318,9 @@ impl Aggregating { let output = match reservation.try_resize(self.table.memory_size()) { Ok(()) => self.table.take_completed_state_batch()?, Err(oom @ DataFusionError::ResourcesExhausted(_)) => { - // Partial ordering may have an unbounded active key range. + // Partial group clustering may have an unbounded active key range. // The final stage can merge incomplete states emitted here. - if matches!(self.table.group_ordering(), GroupOrdering::Full(_)) { + if matches!(self.table.group_clustering(), GroupClustering::Full(_)) { return Err(oom); } let Some(batch) = self.table.take_state_batch()? else { @@ -363,11 +366,11 @@ impl Aggregating { impl Outputting { /// Emits slices of one materialized batch without touching the hash table. /// - /// See [`OrderedPartialAggregateStream::into_stream`] for stage transitions + /// See [`ClusteredPartialAggregateStream::into_stream`] for stage transitions /// and output memory accounting. async fn handle_stage( self, - context: &OrderedPartialAggregateContext, + context: &ClusteredPartialAggregateContext, reservation: &MemoryReservation, emitter: &mut TryEmitter, ) -> Result> { diff --git a/datafusion/physical-plan/src/aggregates/ordered_single_stream.rs b/datafusion/physical-plan/src/aggregates/clustered_single_stream.rs similarity index 77% rename from datafusion/physical-plan/src/aggregates/ordered_single_stream.rs rename to datafusion/physical-plan/src/aggregates/clustered_single_stream.rs index 57744dec57f12..7095921ac78fc 100644 --- a/datafusion/physical-plan/src/aggregates/ordered_single_stream.rs +++ b/datafusion/physical-plan/src/aggregates/clustered_single_stream.rs @@ -15,7 +15,7 @@ // specific language governing permissions and limitations // under the License. -//! Single-stage aggregate stream for ordered raw input. +//! Single-stage aggregate stream for raw input with group-clustering guarantees. use std::ops::ControlFlow; use std::sync::Arc; @@ -28,51 +28,52 @@ use datafusion_execution::TaskContext; use datafusion_execution::memory_pool::{MemoryConsumer, MemoryReservation}; use futures::stream::{Stream, StreamExt}; -use super::aggregate_hash_table::{OrderedAggregateTable, SingleMarker}; +use super::aggregate_hash_table::{ClusteredAggregateTable, SingleMarker}; +use super::order::GroupClusteringMode; use super::spill::AggregateSpill; use super::{AggregateExec, create_schema}; use crate::aggregates::AggregateMode; use crate::metrics::{BaselineMetrics, RecordOutput, SpillMetrics}; use crate::stream::EmptyRecordBatchStream; -use crate::{InputOrderMode, RecordBatchStream, SendableRecordBatchStream}; +use crate::{RecordBatchStream, SendableRecordBatchStream}; -/// Single aggregate stream for `InputOrderMode::Sorted` and -/// `InputOrderMode::PartiallySorted`. +/// Single aggregate stream for [`GroupClusteringMode::Partial`] and +/// [`GroupClusteringMode::Full`]. /// /// # Example /// /// SELECT k, AVG(v) FROM t GROUP BY k; /// -/// If the input is ordered by `k`, and there are existing key partitioning on group -/// by keys, the single mode aggregation with ordering optimization can be used: +/// If the input is ordered by `k` and already key-partitioned on the group-by +/// keys, clustered single aggregation can be used: /// /// ## Plan -/// AggregateExec(stage=single, ordered) +/// AggregateExec(stage=single, clustered) /// -- DataSourceExec(t) /// /// ## Single Stage Behavior /// Input: raw rows /// Output: final results for all groups (for example, `AVG(x)`) /// -/// # Order-based Optimization +/// # Group Clustering Optimization /// /// For the aggregation work, the hash aggregation implementation is reused. /// -/// After each input batch, check whether any groups can be emitted eagerly to -/// improve memory efficiency. For example, if the last group key seen is -/// `k = 100`, it is safe to emit all groups with keys less than 100 because the -/// input is ordered. Materialize that entire completed prefix once, then emit -/// slices of it before reading more input. See -/// [`OrderedPartialAggregateStream::into_stream`] for why this avoids +/// After each input batch, the group-clustering mode determines whether any +/// groups can be emitted eagerly to improve memory efficiency. Materialize +/// that entire completed prefix once, then emit slices of it before reading +/// more input. See +/// [`ClusteredPartialAggregateStream::into_stream`] for why this avoids /// repeatedly removing small batches of groups from the table. /// -/// [`OrderedPartialAggregateStream::into_stream`]: super::ordered_partial_stream::OrderedPartialAggregateStream::into_stream +/// [`ClusteredPartialAggregateStream::into_stream`]: super::clustered_partial_stream::ClusteredPartialAggregateStream::into_stream /// /// # Memory Pressure and Spilling /// -/// ## Fully ordered case +/// ## Full group clustering /// -/// If the input is ordered by every group key, for example: +/// Every complete grouping tuple is contiguous. Ordering by every group key is +/// one way to establish this mode, for example: /// /// - Input order: `a, b` /// - `GROUP BY`: `a, b` @@ -84,9 +85,10 @@ use crate::{InputOrderMode, RecordBatchStream, SendableRecordBatchStream}; /// If a memory reservation nevertheless fails, the stream returns the error /// directly, indicating an unexpected behavior. /// -/// ## Partially ordered case +/// ## Partial group clustering /// -/// If the input is ordered by only a subset of the group keys, for example: +/// Rows are contiguous for a subset of the group keys. Ordering by that subset +/// is one way to establish this mode, for example: /// /// - Input order: `a` /// - `GROUP BY`: `a, b` @@ -97,27 +99,27 @@ use crate::{InputOrderMode, RecordBatchStream, SendableRecordBatchStream}; /// On reservation failure, the stream sorts the current intermediate states by /// the complete group key and spills them as one run. After the input ends, it /// spills any remaining states, performs a sort-preserving merge of all runs, -/// and feeds the merged input into a fully ordered final aggregate stream. -pub(crate) struct OrderedSingleAggregateStream { +/// and feeds the merged input into a fully clustered final aggregate stream. +pub(crate) struct ClusteredSingleAggregateStream { schema: SchemaRef, input: SendableRecordBatchStream, reservation: MemoryReservation, baseline_metrics: BaselineMetrics, batch_size: usize, - state: Option, + state: Option, } /// See comments at `poll_next()` for details. -enum OrderedSingleAggregateState { +enum ClusteredSingleAggregateState { ReadingInput { - table: OrderedAggregateTable, + table: ClusteredAggregateTable, /// None if either /// - Disk Manager doesn't enable temporary file creation - /// - The group keys are fully ordered, it's expected to use bounded memory + /// - Full group clustering is used, so completed groups can be released spill_context: Option>, }, Spilling { - table: OrderedAggregateTable, + table: ClusteredAggregateTable, spill_context: Box, }, /// Emits one materialized batch in `batch_size` slices, then continues @@ -126,10 +128,10 @@ enum OrderedSingleAggregateState { batch: RecordBatch, /// Reserved memory of `batch`, released when handing off the last slice. batch_memory: usize, - next_state: Box, + next_state: Box, }, PreparingMergeInput { - table: OrderedAggregateTable, + table: ClusteredAggregateTable, spill_context: Box, }, MergingSpills { @@ -142,13 +144,13 @@ enum OrderedSingleAggregateState { Error, } -type OrderedSingleAggregatePoll = Poll>>; -type OrderedSingleAggregateStateTransition = ControlFlow< - (OrderedSingleAggregatePoll, OrderedSingleAggregateState), - OrderedSingleAggregateState, +type ClusteredSingleAggregatePoll = Poll>>; +type ClusteredSingleAggregateStateTransition = ControlFlow< + (ClusteredSingleAggregatePoll, ClusteredSingleAggregateState), + ClusteredSingleAggregateState, >; -impl OrderedSingleAggregateStream { +impl ClusteredSingleAggregateStream { pub fn new( agg: &AggregateExec, context: &Arc, @@ -158,7 +160,7 @@ impl OrderedSingleAggregateStream { agg.mode, AggregateMode::Single | AggregateMode::SinglePartitioned )); - debug_assert_ne!(agg.input_order_mode, InputOrderMode::Linear); + debug_assert_ne!(agg.group_clustering_mode, GroupClusteringMode::None); let schema = Arc::clone(&agg.schema); let input = agg.input.execute(partition, Arc::clone(context))?; @@ -173,7 +175,7 @@ impl OrderedSingleAggregateStream { AggregateMode::Partial, )?); - let table = OrderedAggregateTable::::new( + let table = ClusteredAggregateTable::::new( agg, partition, Arc::clone(&schema), @@ -181,16 +183,16 @@ impl OrderedSingleAggregateStream { )?; let can_spill = - matches!(agg.input_order_mode, InputOrderMode::PartiallySorted(_)) + matches!(agg.group_clustering_mode, GroupClusteringMode::Partial(_)) && context.runtime_env().disk_manager.tmp_files_enabled(); let spill_context = if can_spill { Some(Box::new(AggregateSpill::try_new( - "OrderedSingleAggregateSpill", + "ClusteredSingleAggregateSpill", agg, context, partition, batch_size, - &agg.input_order_mode, + &agg.group_clustering_mode, &state_schema, spill_metrics, )?)) @@ -199,7 +201,7 @@ impl OrderedSingleAggregateStream { }; let reservation = - MemoryConsumer::new(format!("OrderedSingleAggregateStream[{partition}]")) + MemoryConsumer::new(format!("ClusteredSingleAggregateStream[{partition}]")) .with_can_spill(can_spill) .register(context.memory_pool()); @@ -212,7 +214,7 @@ impl OrderedSingleAggregateStream { reservation, baseline_metrics, batch_size, - state: Some(OrderedSingleAggregateState::ReadingInput { + state: Some(ClusteredSingleAggregateState::ReadingInput { table, spill_context, }), @@ -224,25 +226,25 @@ impl OrderedSingleAggregateStream { self.input = Box::pin(EmptyRecordBatchStream::new(input_schema)); } - fn break_with_err(error: DataFusionError) -> OrderedSingleAggregateStateTransition { + fn break_with_err(error: DataFusionError) -> ClusteredSingleAggregateStateTransition { ControlFlow::Break(( Poll::Ready(Some(Err(error))), - OrderedSingleAggregateState::Error, + ClusteredSingleAggregateState::Error, )) } - fn break_with_internal_err(message: &str) -> OrderedSingleAggregateStateTransition { + fn break_with_internal_err(message: &str) -> ClusteredSingleAggregateStateTransition { Self::break_with_err(internal_datafusion_err!("{message}")) } /// Reserve memory for the current aggregate table. fn reservation_size_for_table( - table: &OrderedAggregateTable, + table: &ClusteredAggregateTable, spill_context: Option<&AggregateSpill>, ) -> usize { let table_size = table.memory_size(); if spill_context.is_some() { - // See `OrderedSingleAggregateStream` comments for how is it estimated + // See `ClusteredSingleAggregateStream` comments for how is it estimated table_size.saturating_add(table.num_groups().saturating_mul(size_of::())) } else { table_size @@ -256,11 +258,11 @@ impl OrderedSingleAggregateStream { &mut self, batch: RecordBatch, table_memory: usize, - next_state: OrderedSingleAggregateState, - ) -> OrderedSingleAggregateStateTransition { + next_state: ClusteredSingleAggregateState, + ) -> ClusteredSingleAggregateStateTransition { let batch_memory = batch.get_array_memory_size(); match self.reservation.try_resize(table_memory + batch_memory) { - Ok(()) => ControlFlow::Continue(OrderedSingleAggregateState::Outputting { + Ok(()) => ControlFlow::Continue(ClusteredSingleAggregateState::Outputting { batch, batch_memory, next_state: Box::new(next_state), @@ -279,8 +281,8 @@ impl OrderedSingleAggregateStream { } } - /// Consumes one ordered raw input batch, then materializes all finalized - /// groups if the ordering proves any group is ready. + /// Consumes one clustered raw input batch, then materializes all finalized + /// groups if any are complete. /// /// See comments at `poll_next()` for details. /// @@ -288,22 +290,22 @@ impl OrderedSingleAggregateStream { fn handle_reading_input( &mut self, cx: &mut Context<'_>, - original_state: OrderedSingleAggregateState, - ) -> OrderedSingleAggregateStateTransition { - let OrderedSingleAggregateState::ReadingInput { + original_state: ClusteredSingleAggregateState, + ) -> ClusteredSingleAggregateStateTransition { + let ClusteredSingleAggregateState::ReadingInput { mut table, spill_context, } = original_state else { return Self::break_with_internal_err( - "Ordered single aggregate stream expected ReadingInput state", + "Clustered single aggregate stream expected ReadingInput state", ); }; match self.input.poll_next_unpin(cx) { Poll::Pending => ControlFlow::Break(( Poll::Pending, - OrderedSingleAggregateState::ReadingInput { + ClusteredSingleAggregateState::ReadingInput { table, spill_context, }, @@ -332,16 +334,16 @@ impl OrderedSingleAggregateStream { Err(e @ DataFusionError::ResourcesExhausted(_)) => { let Some(spill_context) = spill_context else { // `None` means spilling is not supported, see comments - // at `OrderedSingleAggregateState` for details. + // at `ClusteredSingleAggregateState` for details. return Self::break_with_err(e); }; if table.is_empty() { return Self::break_with_internal_err( - "Ordered single aggregate ran out of memory with no aggregated groups", + "Clustered single aggregate ran out of memory with no aggregated groups", ); } return ControlFlow::Continue( - OrderedSingleAggregateState::Spilling { + ClusteredSingleAggregateState::Spilling { table, spill_context, }, @@ -377,19 +379,19 @@ impl OrderedSingleAggregateStream { self.start_outputting( batch, table_memory, - OrderedSingleAggregateState::ReadingInput { + ClusteredSingleAggregateState::ReadingInput { table, spill_context, }, ) } // Can't do early emit, continue aggregating. - Ok(None) => { - ControlFlow::Continue(OrderedSingleAggregateState::ReadingInput { + Ok(None) => ControlFlow::Continue( + ClusteredSingleAggregateState::ReadingInput { table, spill_context, - }) - } + }, + ), Err(e) => Self::break_with_err(e), } } @@ -399,7 +401,7 @@ impl OrderedSingleAggregateStream { match spill_context { Some(spill_context) if spill_context.has_spills() => { ControlFlow::Continue( - OrderedSingleAggregateState::PreparingMergeInput { + ClusteredSingleAggregateState::PreparingMergeInput { table, spill_context, }, @@ -417,10 +419,10 @@ impl OrderedSingleAggregateStream { Ok(Some(batch)) => self.start_outputting( batch, 0, - OrderedSingleAggregateState::Done, + ClusteredSingleAggregateState::Done, ), Ok(None) => { - ControlFlow::Continue(OrderedSingleAggregateState::Done) + ControlFlow::Continue(ClusteredSingleAggregateState::Done) } Err(e) => Self::break_with_err(e), } @@ -437,22 +439,22 @@ impl OrderedSingleAggregateStream { /// Returns the next operator state with control flow decision. fn handle_spilling( &mut self, - original_state: OrderedSingleAggregateState, - ) -> OrderedSingleAggregateStateTransition { - let OrderedSingleAggregateState::Spilling { + original_state: ClusteredSingleAggregateState, + ) -> ClusteredSingleAggregateStateTransition { + let ClusteredSingleAggregateState::Spilling { mut table, mut spill_context, } = original_state else { return Self::break_with_internal_err( - "Ordered single aggregate stream expected Spilling state", + "Clustered single aggregate stream expected Spilling state", ); }; // Sanity check: it's impossible to OOM when the table is empty if table.is_empty() { return Self::break_with_internal_err( - "Ordered single aggregation entered Spilling with an empty table", + "Clustered single aggregation entered Spilling with an empty table", ); } @@ -472,10 +474,12 @@ impl OrderedSingleAggregateStream { match result { // Finished spilling the aggregate table, continue aggregating from input - Ok(()) => ControlFlow::Continue(OrderedSingleAggregateState::ReadingInput { - table, - spill_context: Some(spill_context), - }), + Ok(()) => { + ControlFlow::Continue(ClusteredSingleAggregateState::ReadingInput { + table, + spill_context: Some(spill_context), + }) + } Err(e) => Self::break_with_err(e), } } @@ -483,7 +487,7 @@ impl OrderedSingleAggregateStream { /// 1. Spills the last in-memory run. /// 2. Constructs a globally ordered input stream by applying a sort-preserving /// merge to all spills. - /// 3. Constructs a replay stream: an ordered aggregate stream over the fully + /// 3. Constructs a replay stream: a clustered aggregate stream over the fully /// ordered input constructed from the spills. /// /// See comments at `poll_next()` for details. @@ -491,15 +495,15 @@ impl OrderedSingleAggregateStream { /// Returns the next operator state with control flow decision. fn handle_preparing_merge_input( &mut self, - original_state: OrderedSingleAggregateState, - ) -> OrderedSingleAggregateStateTransition { - let OrderedSingleAggregateState::PreparingMergeInput { + original_state: ClusteredSingleAggregateState, + ) -> ClusteredSingleAggregateStateTransition { + let ClusteredSingleAggregateState::PreparingMergeInput { mut table, mut spill_context, } = original_state else { return Self::break_with_internal_err( - "Ordered single aggregate stream expected PreparingMergeInput state", + "Clustered single aggregate stream expected PreparingMergeInput state", ); }; @@ -527,7 +531,7 @@ impl OrderedSingleAggregateStream { match replay { Ok(stream) => { - ControlFlow::Continue(OrderedSingleAggregateState::MergingSpills { + ControlFlow::Continue(ClusteredSingleAggregateState::MergingSpills { stream, }) } @@ -544,26 +548,28 @@ impl OrderedSingleAggregateStream { fn handle_merging_spills( &mut self, cx: &mut Context<'_>, - original_state: OrderedSingleAggregateState, - ) -> OrderedSingleAggregateStateTransition { - let OrderedSingleAggregateState::MergingSpills { mut stream } = original_state + original_state: ClusteredSingleAggregateState, + ) -> ClusteredSingleAggregateStateTransition { + let ClusteredSingleAggregateState::MergingSpills { mut stream } = original_state else { return Self::break_with_internal_err( - "Ordered single aggregate stream expected MergingSpills state", + "Clustered single aggregate stream expected MergingSpills state", ); }; match stream.poll_next_unpin(cx) { Poll::Pending => ControlFlow::Break(( Poll::Pending, - OrderedSingleAggregateState::MergingSpills { stream }, + ClusteredSingleAggregateState::MergingSpills { stream }, )), Poll::Ready(Some(Ok(batch))) => ControlFlow::Break(( Poll::Ready(Some(Ok(batch))), - OrderedSingleAggregateState::MergingSpills { stream }, + ClusteredSingleAggregateState::MergingSpills { stream }, )), Poll::Ready(Some(Err(e))) => Self::break_with_err(e), - Poll::Ready(None) => ControlFlow::Continue(OrderedSingleAggregateState::Done), + Poll::Ready(None) => { + ControlFlow::Continue(ClusteredSingleAggregateState::Done) + } } } @@ -574,16 +580,16 @@ impl OrderedSingleAggregateStream { /// Returns the next operator state with control flow decision. fn handle_outputting( &mut self, - original_state: OrderedSingleAggregateState, - ) -> OrderedSingleAggregateStateTransition { - let OrderedSingleAggregateState::Outputting { + original_state: ClusteredSingleAggregateState, + ) -> ClusteredSingleAggregateStateTransition { + let ClusteredSingleAggregateState::Outputting { batch, batch_memory, next_state, } = original_state else { return Self::break_with_internal_err( - "Ordered single aggregate stream expected Outputting state", + "Clustered single aggregate stream expected Outputting state", ); }; @@ -592,7 +598,7 @@ impl OrderedSingleAggregateStream { let batch = batch.slice(self.batch_size, batch.num_rows() - self.batch_size); return ControlFlow::Break(( Poll::Ready(Some(Ok(output.record_output(&self.baseline_metrics)))), - OrderedSingleAggregateState::Outputting { + ClusteredSingleAggregateState::Outputting { batch, batch_memory, next_state, @@ -611,20 +617,20 @@ impl OrderedSingleAggregateStream { } } -impl Stream for OrderedSingleAggregateStream { +impl Stream for ClusteredSingleAggregateStream { type Item = Result; - /// Entry point for the ordered single aggregate state machine. + /// Entry point for the clustered single aggregate state machine. /// - /// See comments in [`OrderedSingleAggregateStream`] for high-level ideas. + /// See comments in [`ClusteredSingleAggregateStream`] for high-level ideas. /// /// State transition graph: /// /// ```text /// (start) /// -> ReadingInput - /// The stream starts by polling ordered raw input and updating the - /// ordered single aggregate table. + /// The stream starts by polling clustered raw input and updating the + /// clustered single aggregate table. /// /// ReadingInput /// -> ReadingInput @@ -634,7 +640,7 @@ impl Stream for OrderedSingleAggregateStream { /// The table cannot reserve enough memory. Move all current states into /// one fully group-key-sorted spill run. /// -> Outputting - /// Either the input ordering proves some groups complete, or input was + /// Either some groups are complete, or input was /// exhausted without spilling and every remaining group is complete. /// Materialize all of them once into one batch. If the batch cannot be /// reserved, yield it whole and go to the state after `Outputting`. @@ -689,31 +695,31 @@ impl Stream for OrderedSingleAggregateStream { let cur_state = self .state .take() - .expect("OrderedSingleAggregateStream state should not be None"); + .expect("ClusteredSingleAggregateStream state should not be None"); let next_state = match cur_state { - state @ OrderedSingleAggregateState::ReadingInput { .. } => { + state @ ClusteredSingleAggregateState::ReadingInput { .. } => { self.handle_reading_input(cx, state) } - state @ OrderedSingleAggregateState::Spilling { .. } => { + state @ ClusteredSingleAggregateState::Spilling { .. } => { self.handle_spilling(state) } - state @ OrderedSingleAggregateState::PreparingMergeInput { .. } => { + state @ ClusteredSingleAggregateState::PreparingMergeInput { .. } => { self.handle_preparing_merge_input(state) } - state @ OrderedSingleAggregateState::MergingSpills { .. } => { + state @ ClusteredSingleAggregateState::MergingSpills { .. } => { self.handle_merging_spills(cx, state) } - state @ OrderedSingleAggregateState::Outputting { .. } => { + state @ ClusteredSingleAggregateState::Outputting { .. } => { self.handle_outputting(state) } - state @ OrderedSingleAggregateState::Error => { + state @ ClusteredSingleAggregateState::Error => { self.close_input(); self.reservation.free(); self.state = Some(state); return Poll::Ready(None); } - state @ OrderedSingleAggregateState::Done => { + state @ ClusteredSingleAggregateState::Done => { let _ = self.reservation.try_resize(0); self.state = Some(state); return Poll::Ready(None); @@ -727,14 +733,14 @@ impl Stream for OrderedSingleAggregateStream { ControlFlow::Break((Poll::Ready(Some(Err(e))), next_state)) => { debug_assert!(matches!( next_state, - OrderedSingleAggregateState::Error + ClusteredSingleAggregateState::Error )); // The handler has already discarded its state-owned resources. // Release the remaining stream-owned resources before returning. self.close_input(); self.reservation.free(); - self.state = Some(OrderedSingleAggregateState::Error); + self.state = Some(ClusteredSingleAggregateState::Error); return Poll::Ready(Some(Err(e))); } ControlFlow::Break((poll, next_state)) => { @@ -746,7 +752,7 @@ impl Stream for OrderedSingleAggregateStream { } } -impl RecordBatchStream for OrderedSingleAggregateStream { +impl RecordBatchStream for ClusteredSingleAggregateStream { fn schema(&self) -> SchemaRef { Arc::clone(&self.schema) } diff --git a/datafusion/physical-plan/src/aggregates/group_values/mod.rs b/datafusion/physical-plan/src/aggregates/group_values/mod.rs index 32eabfe8dc26c..93e2efbfbd648 100644 --- a/datafusion/physical-plan/src/aggregates/group_values/mod.rs +++ b/datafusion/physical-plan/src/aggregates/group_values/mod.rs @@ -29,7 +29,7 @@ mod row; pub use row::GroupValuesRows; mod single_group_by; use datafusion_physical_expr::binary_map::OutputType; -use multi_group_by::{GroupValuesColumn, GroupValuesOrdered}; +use multi_group_by::{GroupValuesClustered, GroupValuesColumn}; pub(crate) use single_group_by::primitive::HashValue; @@ -38,7 +38,7 @@ use crate::aggregates::{ boolean::GroupValuesBoolean, bytes::GroupValuesBytes, bytes_view::GroupValuesBytesView, primitive::GroupValuesPrimitive, }, - order::GroupOrdering, + order::GroupClustering, }; mod metrics; @@ -139,7 +139,7 @@ pub trait GroupValues: Send { /// /// [`GroupValues`] implementations choosing logic: /// -/// - Fully ordered multi-column keys with supported scalar types use adjacent +/// - Fully clustered multi-column keys with supported scalar types use adjacent /// comparisons and column builders, without a hash table. /// /// - If group by single column, and type of this column has @@ -156,12 +156,12 @@ pub trait GroupValues: Send { /// `GroupValuesRows`: crate::aggregates::group_values::GroupValuesRows pub fn new_group_values( schema: SchemaRef, - group_ordering: &GroupOrdering, + group_clustering: &GroupClustering, ) -> Result> { - if matches!(group_ordering, GroupOrdering::Full(_)) - && GroupValuesOrdered::supports_schema(&schema) + if matches!(group_clustering, GroupClustering::Full(_)) + && GroupValuesClustered::supports_schema(&schema) { - return Ok(Box::new(GroupValuesOrdered::try_new(schema)?)); + return Ok(Box::new(GroupValuesClustered::try_new(schema)?)); } if schema.fields.len() == 1 { let d = schema.fields[0].data_type(); @@ -200,7 +200,7 @@ pub fn new_group_values( } if multi_group_by::supported_schema(schema.as_ref()) { - if matches!(group_ordering, GroupOrdering::None) { + if matches!(group_clustering, GroupClustering::None) { Ok(Box::new(GroupValuesColumn::::try_new(schema)?)) } else { Ok(Box::new(GroupValuesColumn::::try_new(schema)?)) @@ -221,7 +221,7 @@ mod tests { use datafusion_expr::{EmitTo, GroupSelection}; use super::new_group_values; - use crate::aggregates::order::GroupOrdering; + use crate::aggregates::order::GroupClustering; #[test] fn preserving_values_keep_group_indices_valid() { @@ -230,7 +230,7 @@ mod tests { DataType::Int32, true, )])); - let mut group_values = new_group_values(schema, &GroupOrdering::None).unwrap(); + let mut group_values = new_group_values(schema, &GroupClustering::None).unwrap(); assert!(group_values.supports_values_preserving()); let input = Arc::new(Int32Array::from(vec![ @@ -287,7 +287,7 @@ mod tests { Field::new("primitive", DataType::Int32, false), Field::new("boolean", DataType::Boolean, false), ])); - let mut group_values = new_group_values(schema, &GroupOrdering::None).unwrap(); + let mut group_values = new_group_values(schema, &GroupClustering::None).unwrap(); let input = vec![ Arc::new(Int32Array::from(vec![10, 20, 10])) as ArrayRef, Arc::new(BooleanArray::from(vec![true, false, true])) as ArrayRef, @@ -318,7 +318,7 @@ mod tests { true, )])); let mut group_values = - new_group_values(schema, &GroupOrdering::None).unwrap(); + new_group_values(schema, &GroupClustering::None).unwrap(); let input: ArrayRef = match data_type { DataType::Utf8 => Arc::new(StringArray::from(vec![ Some("a"), diff --git a/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/ordered.rs b/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/clustered.rs similarity index 87% rename from datafusion/physical-plan/src/aggregates/group_values/multi_group_by/ordered.rs rename to datafusion/physical-plan/src/aggregates/group_values/multi_group_by/clustered.rs index 9e34ac7aeac17..1fcd1fa0bff03 100644 --- a/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/ordered.rs +++ b/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/clustered.rs @@ -27,7 +27,7 @@ use datafusion_expr::{EmitTo, GroupSelection}; use super::{GroupColumn, GroupValuesColumn}; use crate::aggregates::group_values::GroupValues; -/// Columnar group keys for input fully ordered by all grouping expressions. +/// Columnar group keys for input clustered by the complete grouping tuple. /// /// Equal keys must be contiguous across input batches. Within a batch Arrow's /// partition kernel finds those runs. Only the first run can continue the last @@ -36,14 +36,15 @@ use crate::aggregates::group_values::GroupValues; /// /// This removes hashing and hash-table storage, but does not change the dense /// group-id or emission contracts. In particular, `First(n)` shifts the remaining -/// keys and their ids together. Partially ordered inputs must use a hash table. -pub(crate) struct GroupValuesOrdered { +/// keys and their ids together. Inputs clustered by only a subset of the keys +/// must use a hash table. +pub(crate) struct GroupValuesClustered { schema: SchemaRef, columns: Vec>, new_groups: Vec, } -impl GroupValuesOrdered { +impl GroupValuesClustered { /// Types for which adjacent Arrow equality agrees with GROUP BY equality. /// Keep single-column specializations and floats/nested/encoded keys on the /// established path until their semantics and performance are validated. @@ -74,7 +75,7 @@ impl GroupValuesOrdered { pub(crate) fn try_new(schema: SchemaRef) -> Result { if !Self::supports_schema(&schema) { return not_impl_err!( - "Unsupported schema for fully ordered group values: {schema}" + "Unsupported schema for fully clustered group values: {schema}" ); } let columns = GroupValuesColumn::::build_group_columns(&schema)?; @@ -86,7 +87,7 @@ impl GroupValuesOrdered { } } -impl GroupValues for GroupValuesOrdered { +impl GroupValues for GroupValuesClustered { fn intern(&mut self, cols: &[ArrayRef], groups: &mut Vec) -> Result<()> { groups.clear(); let ranges = partition(cols)?.ranges(); @@ -170,7 +171,7 @@ impl GroupValues for GroupValuesOrdered { fn clear_shrink(&mut self, num_rows: usize) { self.columns = GroupValuesColumn::::build_group_columns(&self.schema) - .expect("schema validated by GroupValuesOrdered::try_new"); + .expect("schema validated by GroupValuesClustered::try_new"); self.new_groups.clear(); self.new_groups.shrink_to(num_rows); } @@ -191,8 +192,12 @@ mod tests { // Compare with the existing hash implementation, changing actual input // boundaries and removing completed groups between batches. This checks the // semantic contract rather than duplicating the adjacent-run algorithm. - #[test] - fn ordered_keys_match_hash_grouping_across_batches_and_emits() -> Result<()> { + #[rstest::rstest] + #[case::sorted(false)] + #[case::unsorted(true)] + fn clustered_keys_match_hash_grouping_across_batches_and_emits( + #[case] unsorted: bool, + ) -> Result<()> { for key_type in [ DataType::Boolean, DataType::Int8, @@ -213,7 +218,9 @@ mod tests { let rows = 1025; let first: ArrayRef = Arc::new(Int32Array::from_iter((0..rows).map(|row| { let group = row / 7; - (group >= 4).then_some(group / 4) + let key = group / 4; + // Swap neighboring key values while keeping each run contiguous. + (group >= 4).then_some(if unsorted { key ^ 1 } else { key }) }))); let second: ArrayRef = match key_type { DataType::Boolean => { @@ -262,8 +269,8 @@ mod tests { ])); for batch_size in [1, 2, 7, 8, 63, 1024] { for emit_limit in [0, 1, 17, usize::MAX] { - let mut ordered = - GroupValuesOrdered::try_new(Arc::clone(&schema))?; + let mut clustered = + GroupValuesClustered::try_new(Arc::clone(&schema))?; let mut hashed = GroupValuesColumn::::try_new(Arc::clone(&schema))?; let mut actual = Vec::new(); @@ -274,16 +281,16 @@ mod tests { .iter() .map(|array| array.slice(offset, length)) .collect::>(); - ordered.intern(&batch, &mut actual)?; + clustered.intern(&batch, &mut actual)?; hashed.intern(&batch, &mut expected)?; assert_eq!( actual, expected, "{key_type:?}, batch={batch_size}, offset={offset}, descending={descending}" ); - assert_eq!(ordered.len(), hashed.len()); - let selection = GroupSelection::all(ordered.len()); + assert_eq!(clustered.len(), hashed.len()); + let selection = GroupSelection::all(clustered.len()); assert_eq!( - ordered.values_preserving(selection)?, + clustered.values_preserving(selection)?, hashed.values_preserving(selection)? ); // A zero-row batch must neither forget the boundary @@ -292,23 +299,29 @@ mod tests { .iter() .map(|array| array.slice(0, 0)) .collect::>(); - ordered.intern(&empty, &mut actual)?; + clustered.intern(&empty, &mut actual)?; assert!(actual.is_empty()); - let emit = emit_limit.min(ordered.len().saturating_sub(1)); + let emit = emit_limit.min(clustered.len().saturating_sub(1)); assert_eq!( - ordered.emit(EmitTo::First(emit))?, + clustered.emit(EmitTo::First(emit))?, hashed.emit(EmitTo::First(emit))? ); } - assert_eq!(ordered.emit(EmitTo::All)?, hashed.emit(EmitTo::All)?); - assert!(ordered.is_empty()); + assert_eq!( + clustered.emit(EmitTo::All)?, + hashed.emit(EmitTo::All)? + ); + assert!(clustered.is_empty()); // All and clear_shrink reset the cross-batch state. - ordered.clear_shrink(0); + clustered.clear_shrink(0); hashed.clear_shrink(0); - ordered.intern(&input, &mut actual)?; + clustered.intern(&input, &mut actual)?; hashed.intern(&input, &mut expected)?; assert_eq!(actual, expected); - assert_eq!(ordered.emit(EmitTo::All)?, hashed.emit(EmitTo::All)?); + assert_eq!( + clustered.emit(EmitTo::All)?, + hashed.emit(EmitTo::All)? + ); } } } @@ -317,7 +330,7 @@ mod tests { } #[test] - fn ordered_schema_gate_keeps_unvalidated_types_on_existing_paths() { + fn clustered_schema_gate_keeps_unvalidated_types_on_existing_paths() { for data_type in [ DataType::Float32, DataType::Float64, @@ -329,18 +342,20 @@ mod tests { Field::new("a", DataType::Int32, false), Field::new("b", data_type, true), ]); - assert!(!GroupValuesOrdered::supports_schema(&schema)); + assert!(!GroupValuesClustered::supports_schema(&schema)); } - assert!(!GroupValuesOrdered::supports_schema(&Schema::new(vec![ + assert!(!GroupValuesClustered::supports_schema(&Schema::new(vec![ Field::new("a", DataType::Int32, false) ]))); } #[tokio::test] async fn ordered_single_and_partial_final_match_unordered_execution() -> Result<()> { - use crate::aggregates::{AggregateExec, AggregateMode, PhysicalGroupBy}; + use crate::aggregates::{ + AggregateExec, AggregateMode, GroupClusteringMode, PhysicalGroupBy, + }; use crate::test::TestMemoryExec; - use crate::{ExecutionPlan, InputOrderMode, collect}; + use crate::{ExecutionPlan, collect}; use arrow::compute::{SortOptions, take_record_batch}; use arrow::record_batch::RecordBatch; use arrow::row::{RowConverter, SortField}; @@ -437,7 +452,10 @@ mod tests { Arc::clone(&schema), )?; if sorted { - assert_eq!(plan.input_order_mode(), &InputOrderMode::Sorted); + assert_eq!( + plan.group_clustering_mode(), + &GroupClusteringMode::Full + ); } let plan: Arc = if two_stage { let plan = AggregateExec::try_new( @@ -450,8 +468,8 @@ mod tests { )?; if sorted { assert_eq!( - plan.input_order_mode(), - &InputOrderMode::Sorted + plan.group_clustering_mode(), + &GroupClusteringMode::Full ); } Arc::new(plan) diff --git a/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/list.rs b/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/list.rs index b5dbdb38763f4..a59be76298a32 100644 --- a/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/list.rs +++ b/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/list.rs @@ -445,7 +445,7 @@ mod tests { // List/Struct together correctly and that intern/emit round-trips // through `GroupValuesColumn` rather than `GroupValuesRows`. use crate::aggregates::group_values::new_group_values; - use crate::aggregates::order::GroupOrdering; + use crate::aggregates::order::GroupClustering; use arrow::array::{ Int32Array, LargeListArray, StringArray, StructArray, builder::Int32Builder, builder::LargeListBuilder, builder::StringBuilder, builder::StructBuilder, @@ -507,7 +507,7 @@ mod tests { Arc::new(list_builder.finish()) }; - let mut gv = new_group_values(schema, &GroupOrdering::None).unwrap(); + let mut gv = new_group_values(schema, &GroupClustering::None).unwrap(); // Batch 1: a mix of duplicate / distinct / null lists. let batch1 = notes_v(&[ @@ -584,7 +584,7 @@ mod tests { #[test] fn list_dispatcher_round_trip_through_new_group_values() { use crate::aggregates::group_values::new_group_values; - use crate::aggregates::order::GroupOrdering; + use crate::aggregates::order::GroupClustering; use arrow::datatypes::Schema; use datafusion_expr::EmitTo; @@ -593,7 +593,7 @@ mod tests { DataType::List(child_field()), true, )])); - let mut gv = new_group_values(schema, &GroupOrdering::None).unwrap(); + let mut gv = new_group_values(schema, &GroupClustering::None).unwrap(); // Batch 1. let batch1: ArrayRef = list_array(&[ @@ -869,7 +869,7 @@ mod tests { // ints. Exercises a recursive child GroupColumn built via the // dispatcher. use crate::aggregates::group_values::new_group_values; - use crate::aggregates::order::GroupOrdering; + use crate::aggregates::order::GroupClustering; use arrow::array::{ListArray, builder::Int32Builder, builder::ListBuilder}; use arrow::datatypes::Schema; use datafusion_expr::EmitTo; @@ -917,7 +917,7 @@ mod tests { Arc::new(outer.finish()) }; - let mut gv = new_group_values(schema, &GroupOrdering::None).unwrap(); + let mut gv = new_group_values(schema, &GroupClustering::None).unwrap(); // Three groups: [[1,2],[3]], its duplicate, a distinct value, and a null. let batch = mk(&[ diff --git a/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/mod.rs b/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/mod.rs index 9497d521e6321..136a40d263c8f 100644 --- a/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/mod.rs +++ b/datafusion/physical-plan/src/aggregates/group_values/multi_group_by/mod.rs @@ -20,11 +20,11 @@ mod boolean; mod bytes; pub mod bytes_view; +mod clustered; mod dictionary; mod fixed_size_binary; mod list; -mod ordered; -pub(super) use ordered::GroupValuesOrdered; +pub(super) use clustered::GroupValuesClustered; pub mod primitive; pub mod row_backed; diff --git a/datafusion/physical-plan/src/aggregates/grouped_hash_stream.rs b/datafusion/physical-plan/src/aggregates/grouped_hash_stream.rs index b9a6d2833ab7b..15166e4278d60 100644 --- a/datafusion/physical-plan/src/aggregates/grouped_hash_stream.rs +++ b/datafusion/physical-plan/src/aggregates/grouped_hash_stream.rs @@ -35,14 +35,14 @@ use std::task::{Context, Poll}; use std::vec; use super::aggregate_hash_table::{accumulator_phases, create_group_accumulator}; -use super::order::GroupOrdering; +use super::order::GroupClustering; use super::skip_partial::SkipAggregationProbe; use super::{AggregateExec, format_human_display}; use crate::aggregates::group_values::{ AccumulatorPhase, AggregateAccumulatorMetrics, AggregateArgumentMetrics, GroupByMetrics, GroupValues, aggregate_sub_metrics, new_group_values, }; -use crate::aggregates::order::GroupOrderingFull; +use crate::aggregates::order::GroupClusteringFull; use crate::aggregates::{ AggregateInputMode, AggregateMode, AggregateOutputMode, PhysicalGroupBy, aggregate_metric_label, create_schema, evaluate_group_by, evaluate_optional, @@ -353,10 +353,8 @@ pub(crate) struct GroupedHashAggregateStream { // TASK-SPECIFIC STATES: // Inner states groups together properties, states for a specific task. // ======================================================================== - /// Optional ordering information, that might allow groups to be - /// emitted from the hash table prior to seeing the end of the - /// input - group_ordering: GroupOrdering, + /// Tracks groups that can be emitted from the hash table before the input ends. + group_clustering: GroupClustering, /// The spill state object spill_state: SpillState, @@ -528,20 +526,20 @@ impl GroupedHashAggregateStream { .collect::>() .join(", "); let name = format!("GroupedHashAggregateStream[{partition}] ({agg_fn_names})"); - let group_ordering = GroupOrdering::try_new(&agg.input_order_mode)?; - let oom_mode = match (agg.mode, &group_ordering) { + let group_clustering = GroupClustering::try_new(&agg.group_clustering_mode)?; + let oom_mode = match (agg.mode, &group_clustering) { // In partial aggregation mode, always prefer to emit incomplete results early. (AggregateMode::Partial, _) => OutOfMemoryMode::EmitEarly, // For non-partial aggregation modes, emitting incomplete results is not an option. // Instead, use disk spilling to store sorted, incomplete results, and merge them // afterwards. - (_, GroupOrdering::None | GroupOrdering::Partial(_)) + (_, GroupClustering::None | GroupClustering::Partial(_)) if context.runtime_env().disk_manager.tmp_files_enabled() => { OutOfMemoryMode::Spill } - // For `GroupOrdering::Full`, the incoming stream is already sorted. This ensures the - // number of incomplete groups can be kept small at all times. If we still hit + // For `GroupClustering::Full`, each group is contiguous in the input. This keeps + // the number of incomplete groups small at all times. If we still hit // an out-of-memory condition, spilling to disk would not be beneficial since the same // situation is likely to reoccur when reading back the spilled data. // Therefore, we fall back to simply reporting the error immediately. @@ -550,7 +548,7 @@ impl GroupedHashAggregateStream { _ => OutOfMemoryMode::ReportError, }; - let group_values = new_group_values(group_schema, &group_ordering)?; + let group_values = new_group_values(group_schema, &group_clustering)?; let reservation = MemoryConsumer::new(name) // We interpret 'can spill' as 'can handle memory back pressure'. // This value needs to be set to true for the default memory pool implementations @@ -586,7 +584,7 @@ impl GroupedHashAggregateStream { // since Final mode expects unique group values as its input // - there is only one GROUP BY expressions set let skip_aggregation_probe = if agg.mode == AggregateMode::Partial - && matches!(group_ordering, GroupOrdering::None) + && matches!(group_clustering, GroupClustering::None) && agg_group_by.is_single() { let options = &context.session_config().options().execution; @@ -641,7 +639,7 @@ impl GroupedHashAggregateStream { aggregate_argument_metrics, aggregate_accumulator_metrics, batch_size, - group_ordering, + group_clustering, input_done: false, spill_state, group_values_soft_limit: agg.limit_options().map(|config| config.limit()), @@ -698,7 +696,7 @@ impl Stream for GroupedHashAggregateStream { // this might lead to incorrect output ordering if (self.spill_state.spills.is_empty() || self.spill_state.is_stream_merging) - && let Some(to_emit) = self.group_ordering.emit_to() + && let Some(to_emit) = self.group_clustering.emit_to() { timer.done(); if let Some(batch) = self.emit(to_emit, false)? { @@ -899,7 +897,7 @@ impl GroupedHashAggregateStream { })?; for group_values in &group_by_values { - // Calculate group indices and update ordering information. + // Calculate group indices and update the completion tracker. let total_num_groups = self.group_by_metrics.time_group_key_preparation(|| { let starting_num_groups = self.group_values.len(); @@ -908,7 +906,7 @@ impl GroupedHashAggregateStream { let group_indices = &self.current_group_indices; let total_num_groups = self.group_values.len(); if total_num_groups > starting_num_groups { - self.group_ordering.new_groups( + self.group_clustering.new_groups( group_values, group_indices, total_num_groups, @@ -997,7 +995,7 @@ impl GroupedHashAggregateStream { self.group_values.len() }; - if let Some(emit_to) = self.group_ordering.oom_emit_to(n) + if let Some(emit_to) = self.group_clustering.oom_emit_to(n) && let Some(batch) = self.emit(emit_to, false)? { return Ok(Some(ExecutionState::ProducingOutput(batch))); @@ -1014,7 +1012,7 @@ impl GroupedHashAggregateStream { let acc = self.accumulators.iter().map(|x| x.size()).sum::(); let groups_and_acc_size = acc + self.group_values.size() - + self.group_ordering.size() + + self.group_clustering.size() + self.current_group_indices.allocated_size(); // Reserve extra headroom for sorting during potential spill. @@ -1060,7 +1058,7 @@ impl GroupedHashAggregateStream { let output = group_by_metrics.time_emitting(|| { let mut output = self.group_values.emit(emit_to)?; if let EmitTo::First(n) = emit_to { - self.group_ordering.remove_groups(n); + self.group_clustering.remove_groups(n); } // Next output each aggregate value. @@ -1140,7 +1138,7 @@ impl GroupedHashAggregateStream { .intern(&cols, &mut self.current_group_indices)?; let total_groups = self.group_values.len(); if total_groups > starting_groups { - self.group_ordering.new_groups( + self.group_clustering.new_groups( &cols, &self.current_group_indices, total_groups, @@ -1314,7 +1312,7 @@ impl GroupedHashAggregateStream { /// in case of disk spilling, the SPM stream have been drained. fn set_input_done_and_produce_output(&mut self) -> Result<()> { self.input_done = true; - self.group_ordering.input_done(); + self.group_clustering.input_done(); // Release the original input pipeline's resources now that we're done // reading from it. In the spill branch below, `self.input` is replaced // again with a stream that merges spill files. @@ -1359,12 +1357,12 @@ impl GroupedHashAggregateStream { // Reset the group values collectors. self.clear_all(); - // We can now use `GroupOrdering::Full` since the spill files are sorted + // We can now use `GroupClustering::Full` since the spill files are sorted // on the grouping columns. - self.group_ordering = GroupOrdering::Full(GroupOrderingFull::new()); + self.group_clustering = GroupClustering::Full(GroupClusteringFull::new()); // Recreate `group_values` for streaming merge so group ids are assigned - // in first-seen order, as required by `GroupOrderingFull`. + // in first-seen order, as required by `GroupClusteringFull`. // The pre-spill collector may use `vectorized_intern`, which can assign // new group ids out of input order under hash collisions. That is the // multi-column collector, which also serves a single group column @@ -1374,7 +1372,7 @@ impl GroupedHashAggregateStream { .spill_state .merging_group_by .group_schema(&self.spill_state.spill_schema)?; - self.group_values = new_group_values(group_schema, &self.group_ordering)?; + self.group_values = new_group_values(group_schema, &self.group_clustering)?; // Use `OutOfMemoryMode::ReportError` from this point on // to ensure we don't spill the spilled data to disk again. @@ -1483,7 +1481,7 @@ impl GroupedHashAggregateStream { mod tests { use super::*; use crate::ExecutionPlan; - use crate::InputOrderMode; + use crate::aggregates::GroupClusteringMode; use crate::test::TestMemoryExec; use arrow::array::{Int32Array, Int64Array, UInt32Array}; use arrow::datatypes::{DataType, Field, Schema}; @@ -1770,7 +1768,7 @@ mod tests { Ok(()) } - // Migrated to OrderedPartialAggregateStream coverage in aggregates/mod.rs; + // Migrated to ClusteredPartialAggregateStream coverage in aggregates/mod.rs; // kept here for the legacy GroupedHashAggregateStream implementation. #[tokio::test] async fn test_emit_early_with_partially_sorted() -> Result<()> { @@ -1838,11 +1836,11 @@ mod tests { Arc::clone(&schema), )?; assert!(matches!( - aggregate_exec.input_order_mode(), - InputOrderMode::PartiallySorted(_) + aggregate_exec.group_clustering_mode(), + GroupClusteringMode::Partial(_) )); - // Must not panic with "assertion failed: *current_sort >= n" + // Must not panic with "assertion failed: *current_run_start >= n" let mut stream = GroupedHashAggregateStream::new(&aggregate_exec, &task_ctx, 0)?; while let Some(result) = stream.next().await { if let Err(e) = result { diff --git a/datafusion/physical-plan/src/aggregates/hash_stream.rs b/datafusion/physical-plan/src/aggregates/hash_stream.rs index 2f1bb9b638a4c..0b7612d14556f 100644 --- a/datafusion/physical-plan/src/aggregates/hash_stream.rs +++ b/datafusion/physical-plan/src/aggregates/hash_stream.rs @@ -34,16 +34,17 @@ use futures::stream::{Stream, StreamExt}; use super::AggregateExec; use super::aggregate_hash_table::{ - AggregateHashTable, FinalMarker, OrderedAggregateTableMetrics, PartialMarker, + AggregateHashTable, ClusteredAggregateTableMetrics, FinalMarker, PartialMarker, PartialSkipMarker, }; +use super::order::GroupClusteringMode; use super::skip_partial::SkipAggregationProbe; use super::spill::AggregateSpill; use crate::metrics::{ BaselineMetrics, MetricBuilder, MetricCategory, RecordOutput, SpillMetrics, }; use crate::stream::{EmptyRecordBatchStream, RecordBatchStreamAdapter}; -use crate::{InputOrderMode, SendableRecordBatchStream, metrics}; +use crate::{SendableRecordBatchStream, metrics}; /// Hash aggregation is implemented in two stages: partial and final. This /// stream implements the partial stage. @@ -136,7 +137,7 @@ use crate::{InputOrderMode, SendableRecordBatchStream, metrics}; /// 3. Perform a sort-preserving merge of all spill files and feed the merged output /// into an ordered streaming aggregation, which ensures bounded memory usage and /// evaluates the final result. -/// - [`OrderedFinalAggregateStream`](super::ordered_final_stream::OrderedFinalAggregateStream) is reused for the streaming aggregation. +/// - [`ClusteredFinalAggregateStream`](super::clustered_final_stream::ClusteredFinalAggregateStream) is reused for the streaming aggregation. pub(crate) struct PartialHashAggregateStream { /// Output schema: group columns followed by partial aggregate state columns. schema: SchemaRef, @@ -217,7 +218,7 @@ impl PartialHashAggregateStream { partition: usize, ) -> Result { debug_assert_eq!(agg.mode, super::AggregateMode::Partial); - debug_assert_eq!(agg.input_order_mode, InputOrderMode::Linear); + debug_assert_eq!(agg.group_clustering_mode, GroupClusteringMode::None); let schema = Arc::clone(&agg.schema); let input = agg.input.execute(partition, Arc::clone(context))?; @@ -573,7 +574,7 @@ impl FinalHashAggregateStream { agg.mode, super::AggregateMode::Final | super::AggregateMode::FinalPartitioned )); - debug_assert_eq!(agg.input_order_mode, InputOrderMode::Linear); + debug_assert_eq!(agg.group_clustering_mode, GroupClusteringMode::None); let input = agg.input.execute(partition, Arc::clone(context))?; Self::new_with_input(agg, context, partition, input) @@ -607,7 +608,7 @@ impl FinalHashAggregateStream { context, partition, batch_size, - &InputOrderMode::Linear, + &GroupClusteringMode::None, &input_schema, spill_metrics, )?)) @@ -783,7 +784,7 @@ impl FinalHashAggregateStream { /// Produce output from spills /// 1. Spill in progress in-memory hash table - /// 2. Switch to ordered final stream + /// 2. Switch to clustered final stream /// 3. passthrough stream output async fn produce_output_from_spills( &mut self, @@ -801,7 +802,7 @@ impl FinalHashAggregateStream { // Construct the ordered input used to merge all spill files. let mut output_stream = - self.switch_to_ordered_final_stream(hash_table, spill_context)?; + self.switch_to_clustered_final_stream(hash_table, spill_context)?; timer.done(); @@ -819,16 +820,16 @@ impl FinalHashAggregateStream { /// 1. Constructs a globally ordered input stream by applying a sort-preserving /// merge to all spills. - /// 2. Constructs a replay stream: an ordered final aggregate stream over the + /// 2. Constructs a replay stream: a clustered final aggregate stream over the /// fully ordered input constructed from the spills. /// /// Returns the replay stream - fn switch_to_ordered_final_stream( + fn switch_to_clustered_final_stream( &mut self, hash_table: AggregateHashTable, spill_context: Box, ) -> Result { - let metrics = OrderedAggregateTableMetrics::from_hash_table(&hash_table); + let metrics = ClusteredAggregateTableMetrics::from_hash_table(&hash_table); drop(hash_table); self.reservation.try_resize(0)?; spill_context.into_replay_stream( @@ -879,6 +880,7 @@ mod tests { use std::time::Duration; use super::*; + use crate::aggregates::GroupClusteringMode; use crate::aggregates::{AggregateMode, PhysicalGroupBy}; use crate::common::collect; use crate::execution_plan::ExecutionPlan; @@ -1567,7 +1569,10 @@ mod tests { input, schema, )?; - assert_eq!(aggregate.input_order_mode(), &InputOrderMode::Linear); + assert_eq!( + aggregate.group_clustering_mode(), + &GroupClusteringMode::None + ); let pool: Arc = Arc::new(GreedyMemoryPool::new(limit)); let context = Arc::new( diff --git a/datafusion/physical-plan/src/aggregates/mod.rs b/datafusion/physical-plan/src/aggregates/mod.rs index 6552bd95e06a5..1333f6029b507 100644 --- a/datafusion/physical-plan/src/aggregates/mod.rs +++ b/datafusion/physical-plan/src/aggregates/mod.rs @@ -48,19 +48,19 @@ //! //! See [`PartialHashAggregateStream`] and [`FinalHashAggregateStream`] for details. //! -//! ### Ordering optimization +//! ### Group clustering optimization //! -//! When the input is ordered by the group key, an ordered fast path is used. It -//! uses a similar two-stage hash aggregation with an early-emission optimization. +//! When the input is ordered by group keys, rows are clustered by those keys. +//! The clustered paths use this guarantee to emit completed groups early. //! //! ```text -//! AggregateExec (final, ordered) +//! AggregateExec (final, clustered) //! RepartitionExec (hash by group keys, order-preserving) -//! AggregateExec (partial, ordered) +//! AggregateExec (partial, clustered) //! ``` //! -//! See [`OrderedPartialAggregateStream`], [`OrderedFinalAggregateStream`], and -//! [`OrderedSingleAggregateStream`] for details. +//! See [`ClusteredPartialAggregateStream`], [`ClusteredFinalAggregateStream`], and +//! [`ClusteredSingleAggregateStream`] for details. //! //! Related configuration: //! @@ -81,7 +81,7 @@ //! input //! ``` //! -//! See [`SingleHashAggregateStream`] and [`OrderedSingleAggregateStream`] for +//! See [`SingleHashAggregateStream`] and [`ClusteredSingleAggregateStream`] for //! details. //! //! Related configuration: @@ -153,12 +153,12 @@ use std::sync::Arc; use super::{DisplayAs, ExecutionPlanProperties, PlanProperties}; use crate::aggregates::{ aggregate_stream::AggregateStream, + clustered_final_stream::ClusteredFinalAggregateStream, + clustered_partial_stream::ClusteredPartialAggregateStream, + clustered_single_stream::ClusteredSingleAggregateStream, grouped_hash_stream::GroupedHashAggregateStream, grouped_topk_stream::GroupedTopKAggregateStream, hash_stream::{FinalHashAggregateStream, PartialHashAggregateStream}, - ordered_final_stream::OrderedFinalAggregateStream, - ordered_partial_stream::OrderedPartialAggregateStream, - ordered_single_stream::OrderedSingleAggregateStream, partial_reduce_stream::PartialReduceHashAggregateStream, single_stream::SingleHashAggregateStream, }; @@ -174,7 +174,7 @@ use crate::statistics::{ChildStats, StatisticsArgs}; use crate::{ChildrenPropertiesMode, ReplaceChildrenOptions, validate_child_count}; use crate::{ DisplayFormatType, Distribution, ExecutionPlan, InputDistributionRequirements, - InputOrderMode, Partitioning, SendableRecordBatchStream, Statistics, + Partitioning, SendableRecordBatchStream, Statistics, }; use datafusion_common::config::ConfigOptions; use parking_lot::Mutex; @@ -207,19 +207,20 @@ use datafusion_physical_expr_common::sort_expr::{ use datafusion_expr::utils::AggregateOrderSensitivity; use datafusion_physical_expr_common::utils::evaluate_expressions_to_arrays; use itertools::Itertools; +pub use order::GroupClusteringMode; use topk::hash_table::is_supported_hash_key_type; use topk::heap::is_supported_heap_type; mod aggregate_hash_table; mod aggregate_stream; +mod clustered_final_stream; +mod clustered_partial_stream; +mod clustered_single_stream; pub mod group_values; mod grouped_hash_stream; mod grouped_topk_stream; mod hash_stream; pub mod order; -mod ordered_final_stream; -mod ordered_partial_stream; -mod ordered_single_stream; mod partial_reduce_stream; mod single_stream; mod skip_partial; @@ -691,12 +692,12 @@ enum StreamType { /// Single stage of the hash aggregation /// Input output scheme: initial input -> final result SingleHash(SingleHashAggregateStream), - /// Partial stage of aggregation for ordered input. - OrderedPartialAggregate(OrderedPartialAggregateStream), - /// Final stage of aggregation for ordered input. - OrderedFinalAggregate(OrderedFinalAggregateStream), - /// Single stage of aggregation for ordered input. - OrderedSingleAggregate(OrderedSingleAggregateStream), + /// Partial stage of aggregation for clustered input. + ClusteredPartialAggregate(ClusteredPartialAggregateStream), + /// Final stage of aggregation for clustered input. + ClusteredFinalAggregate(ClusteredFinalAggregateStream), + /// Single stage of aggregation for clustered input. + ClusteredSingleAggregate(ClusteredSingleAggregateStream), /// Legacy hash aggregation reused for multiple stages /// /// Every path it handles now has a dedicated stream, so this variant is only @@ -719,9 +720,9 @@ impl From for SendableRecordBatchStream { StreamType::PartialReduceHash(stream) => Box::pin(stream), StreamType::FinalHash(stream) => stream.into_stream(), StreamType::SingleHash(stream) => stream.into_stream(), - StreamType::OrderedPartialAggregate(stream) => stream.into_stream(), - StreamType::OrderedFinalAggregate(stream) => stream.into_stream(), - StreamType::OrderedSingleAggregate(stream) => Box::pin(stream), + StreamType::ClusteredPartialAggregate(stream) => stream.into_stream(), + StreamType::ClusteredFinalAggregate(stream) => stream.into_stream(), + StreamType::ClusteredSingleAggregate(stream) => Box::pin(stream), StreamType::GroupedHash(stream) => Box::pin(stream), StreamType::GroupedPriorityQueue(stream) => Box::pin(stream), } @@ -910,12 +911,12 @@ pub struct AggregateExec { /// Execution metrics metrics: ExecutionPlanMetricsSet, required_input_ordering: Option, - /// Describes how the input is ordered relative to the group by columns + /// Describes how input rows are clustered by grouping expressions. /// - /// This field is also overloaded to mean "the output MUST preserve this - /// input order". When that is not possible, the constructor overwrites it - /// with the unordered variant [`InputOrderMode::Linear`]. - input_order_mode: InputOrderMode, + /// Input ordering describes a subset of the cases in which groups can be + /// safely emitted before the input ends. Full group clustering requires only + /// that rows for each complete grouping tuple are contiguous. + group_clustering_mode: GroupClusteringMode, cache: Arc, /// During initialization, if the plan supports dynamic filtering (see [`AggrDynFilter`]), /// it is set to `Some(..)` regardless of whether it can be pushed down to a child node. @@ -1106,7 +1107,7 @@ impl AggregateExec { // Commit the kind and properties together: heap output is unordered and final. self.kind = kind; - self.input_order_mode = InputOrderMode::Linear; + self.group_clustering_mode = GroupClusteringMode::None; self.required_input_ordering = None; // Keep unchanged properties so parent aggregates do not need rebuilding. if !self.cache.eq_properties.oeq_class().is_empty() @@ -1331,21 +1332,25 @@ impl AggregateExec { .iter() .filter(|expr| input_eq_properties.is_expr_constant(expr).is_none()) .count(); - let mut input_order_mode = if indices.len() == num_non_constant_groupby_exprs + let mut group_clustering_mode = if indices.len() == num_non_constant_groupby_exprs && !indices.is_empty() && group_by.groups.len() == 1 { - InputOrderMode::Sorted + GroupClusteringMode::Full } else if !indices.is_empty() { - InputOrderMode::PartiallySorted(indices) + GroupClusteringMode::Partial(indices) } else { - InputOrderMode::Linear + GroupClusteringMode::None }; - // Input order mode is also used to advertise plan output ordering, grouping - // sets handling, and partial reduce aggregation can't promise that. + // Grouping sets can change group keys. PartialReduce combines intermediate + // states without using group boundaries to recognize completed groups. if group_by.has_grouping_set() || mode == AggregateMode::PartialReduce { - input_order_mode = InputOrderMode::Linear; + group_clustering_mode = GroupClusteringMode::None; + } else if num_non_constant_groupby_exprs > 0 + && input_eq_properties.grouping_satisfy(groupby_exprs.iter().cloned())? + { + group_clustering_mode = GroupClusteringMode::Full; } // construct a map from the input expression to the output expression of the Aggregation group by @@ -1361,7 +1366,7 @@ impl AggregateExec { &group_expr_mapping, group_by.is_true_no_grouping(), &mode, - &input_order_mode, + &group_clustering_mode, aggr_expr.as_ref(), )? }; @@ -1378,7 +1383,7 @@ impl AggregateExec { input_schema, metrics: ExecutionPlanMetricsSet::new(), required_input_ordering, - input_order_mode, + group_clustering_mode, cache: Arc::new(cache), dynamic_filter: None, }; @@ -1683,47 +1688,51 @@ impl AggregateExec { ); } - // Choose the execution path based on (aggregation mode, ordering). - // - // Note that `self.input_order_mode` represents both input ordering and output - // order promise. See its comment for details. + // Choose the execution path based on aggregation mode and when groups + // are known to be complete. use AggregateMode::*; - use InputOrderMode::*; - let stream = match (self.mode, &self.input_order_mode) { - (Partial, Linear) => StreamType::PartialHash( + let stream = match (self.mode, &self.group_clustering_mode) { + (Partial, GroupClusteringMode::None) => StreamType::PartialHash( PartialHashAggregateStream::new(self, context, partition)?, ), - (Partial, Sorted | PartiallySorted(_)) => { - StreamType::OrderedPartialAggregate(OrderedPartialAggregateStream::new( - self, context, partition, - )?) + (Partial, GroupClusteringMode::Partial(_) | GroupClusteringMode::Full) => { + StreamType::ClusteredPartialAggregate( + ClusteredPartialAggregateStream::new(self, context, partition)?, + ) } - (PartialReduce, Linear) => StreamType::PartialReduceHash( + (PartialReduce, GroupClusteringMode::None) => StreamType::PartialReduceHash( PartialReduceHashAggregateStream::new(self, context, partition)?, ), - (PartialReduce, Sorted | PartiallySorted(_)) => { - // See the comment above: the builder enforces `Linear` order for - // `PartialReduce` mode. + ( + PartialReduce, + GroupClusteringMode::Partial(_) | GroupClusteringMode::Full, + ) => { return internal_err!( - "PartialReduce aggregation must use InputOrderMode::Linear" + "PartialReduce aggregation must use GroupClusteringMode::None" ); } - (Final | FinalPartitioned, Linear) => StreamType::FinalHash( - FinalHashAggregateStream::new(self, context, partition)?, - ), - (Final | FinalPartitioned, Sorted | PartiallySorted(_)) => { - StreamType::OrderedFinalAggregate(OrderedFinalAggregateStream::new( + (Final | FinalPartitioned, GroupClusteringMode::None) => { + StreamType::FinalHash(FinalHashAggregateStream::new( self, context, partition, )?) } - (Single | SinglePartitioned, Linear) => StreamType::SingleHash( - SingleHashAggregateStream::new(self, context, partition)?, - ), - (Single | SinglePartitioned, Sorted | PartiallySorted(_)) => { - StreamType::OrderedSingleAggregate(OrderedSingleAggregateStream::new( + ( + Final | FinalPartitioned, + GroupClusteringMode::Partial(_) | GroupClusteringMode::Full, + ) => StreamType::ClusteredFinalAggregate(ClusteredFinalAggregateStream::new( + self, context, partition, + )?), + (Single | SinglePartitioned, GroupClusteringMode::None) => { + StreamType::SingleHash(SingleHashAggregateStream::new( self, context, partition, )?) } + ( + Single | SinglePartitioned, + GroupClusteringMode::Partial(_) | GroupClusteringMode::Full, + ) => StreamType::ClusteredSingleAggregate( + ClusteredSingleAggregateStream::new(self, context, partition)?, + ), }; Ok(stream) } @@ -1781,17 +1790,23 @@ impl AggregateExec { group_expr_mapping: &ProjectionMapping, is_true_no_grouping: bool, mode: &AggregateMode, - input_order_mode: &InputOrderMode, + group_clustering_mode: &GroupClusteringMode, aggr_exprs: &[Arc], ) -> Result { // Construct equivalence properties: let mut eq_properties = input .equivalence_properties() .project(group_expr_mapping, schema); - - // An aggregation that does not maintain its input order must not - // propegrate the input's ordering either, match `maintains_input_order` value - if *input_order_mode == InputOrderMode::Linear { + // Grouping information is consumed by this aggregate. The aggregate + // output may have a different row layout, so do not pass explicit + // input grouping assertions to another aggregate. + eq_properties.clear_groupings(); + + // Only the clustered paths preserve existing ordering on group keys. + // Project the input's actual sort expressions; clustering alone does + // not establish an output ordering. Keep this consistent with + // `maintains_input_order`. + if *group_clustering_mode == GroupClusteringMode::None { eq_properties.clear_orderings(); } @@ -1839,7 +1854,7 @@ impl AggregateExec { }; // TODO: Emission type and boundedness information can be enhanced here - let emission_type = if *input_order_mode == InputOrderMode::Linear { + let emission_type = if *group_clustering_mode == GroupClusteringMode::None { EmissionType::Final } else { input.pipeline_behavior() @@ -1869,8 +1884,12 @@ impl AggregateExec { ) } - pub fn input_order_mode(&self) -> &InputOrderMode { - &self.input_order_mode + /// Describes how input rows are clustered by grouping expressions. + /// + /// This does not imply a sort order. See [`ExecutionPlanProperties::output_ordering`] + /// for the ordering of the aggregate's output. + pub fn group_clustering_mode(&self) -> &GroupClusteringMode { + &self.group_clustering_mode } /// Estimates output statistics for this aggregate node. @@ -2354,8 +2373,12 @@ impl DisplayAs for AggregateExec { write!(f, ", lim=[{}]", config.limit)?; } - if self.input_order_mode != InputOrderMode::Linear { - write!(f, ", ordering_mode={:?}", self.input_order_mode)?; + if self.group_clustering_mode != GroupClusteringMode::None { + write!( + f, + ", group_clustering_mode={:?}", + self.group_clustering_mode + )?; } } DisplayFormatType::TreeRender => { @@ -2481,17 +2504,10 @@ impl ExecutionPlan for AggregateExec { vec![self.required_input_ordering.clone()] } - /// The output ordering of [`AggregateExec`] is determined by its `group_by` - /// columns. Although this method is not explicitly used by any optimizer - /// rules yet, overriding the default implementation ensures that it - /// accurately reflects the actual behavior. - /// - /// If the [`InputOrderMode`] is `Linear`, the `group_by` columns don't have - /// an ordering, which means the results do not either. However, in the - /// `Ordered` and `PartiallyOrdered` cases, the `group_by` columns do have - /// an ordering, which is preserved in the output. + /// Clustered aggregation preserves existing ordering on the group-by + /// columns. Aggregate result columns do not inherit input ordering. fn maintains_input_order(&self) -> Vec { - vec![self.input_order_mode != InputOrderMode::Linear] + vec![self.group_clustering_mode != GroupClusteringMode::None] } fn children(&self) -> Vec<&Arc> { @@ -2752,8 +2768,8 @@ impl ExecutionPlan for AggregateExec { metrics: _, // Derived at construction from the input ordering and `group_by`. required_input_ordering: _, - // Derived at construction from the input ordering and `group_by`. - input_order_mode: _, + // Derived at construction from the input properties and `group_by`. + group_clustering_mode: _, // Derived at construction by `Self::compute_properties`. cache: _, dynamic_filter, @@ -3659,7 +3675,8 @@ mod tests { use crate::test::TestMemoryExec; use crate::test::assert_is_pending; use crate::test::exec::{ - BlockingExec, PanicExec, StatisticsExec, assert_strong_count_converges_to_zero, + BarrierExec, BlockingExec, PanicExec, StatisticsExec, + assert_strong_count_converges_to_zero, }; use arrow::array::AsArray; @@ -3668,7 +3685,7 @@ mod tests { Int64Array, NullArray, StringArray, StructArray, UInt32Array, UInt64Array, }; use arrow::compute::{SortOptions, concat_batches}; - use arrow::datatypes::{Float64Type, Int32Type, Int64Type, UInt32Type}; + use arrow::datatypes::{Float64Type, Int32Type, Int64Type, TimeUnit, UInt32Type}; use datafusion_common::test_util::{batches_to_sort_string, batches_to_string}; use datafusion_common::{DataFusionError, assert_contains, internal_err}; use datafusion_execution::config::SessionConfig; @@ -3681,6 +3698,7 @@ mod tests { Accumulator, AggregateUDF, AggregateUDFImpl, EmitTo, GroupsAccumulator, Operator, Signature, Volatility, }; + use datafusion_functions::datetime::date_bin; use datafusion_functions_aggregate::approx_percentile_cont::approx_percentile_cont_udaf; use datafusion_functions_aggregate::array_agg::array_agg_udaf; use datafusion_functions_aggregate::average::avg_udaf; @@ -3689,10 +3707,9 @@ mod tests { use datafusion_functions_aggregate::median::median_udaf; use datafusion_functions_aggregate::min_max::min_udaf; use datafusion_functions_aggregate::sum::sum_udaf; - use datafusion_physical_expr::Partitioning; - use datafusion_physical_expr::PhysicalSortExpr; use datafusion_physical_expr::aggregate::AggregateExprBuilder; use datafusion_physical_expr::expressions::{Literal, NotExpr, binary}; + use datafusion_physical_expr::{Partitioning, PhysicalSortExpr, ScalarFunctionExpr}; use crate::projection::ProjectionExec; use crate::repartition::RepartitionExec; @@ -4163,11 +4180,11 @@ mod tests { match (mode, ordered, &stream) { (AggregateMode::Final, false, StreamType::FinalHash(_)) | (AggregateMode::Single, false, StreamType::SingleHash(_)) => {} - (AggregateMode::Final, true, StreamType::OrderedFinalAggregate(_)) - | (AggregateMode::Single, true, StreamType::OrderedSingleAggregate(_)) => { + (AggregateMode::Final, true, StreamType::ClusteredFinalAggregate(_)) + | (AggregateMode::Single, true, StreamType::ClusteredSingleAggregate(_)) => { assert_eq!( - aggregate.input_order_mode(), - &InputOrderMode::PartiallySorted(vec![0]) + aggregate.group_clustering_mode(), + &GroupClusteringMode::Partial(vec![0]) ); } _ => panic!("unexpected stream for {mode:?}, ordered={ordered}"), @@ -5109,12 +5126,12 @@ mod tests { let stream = aggregate.execute_typed(0, &fits)?; match (mode, ordered, &stream) { (AggregateMode::Partial, false, StreamType::PartialHash(_)) - | (AggregateMode::Partial, true, StreamType::OrderedPartialAggregate(_)) + | (AggregateMode::Partial, true, StreamType::ClusteredPartialAggregate(_)) | (AggregateMode::PartialReduce, false, StreamType::PartialReduceHash(_)) | (AggregateMode::Final, false, StreamType::FinalHash(_)) - | (AggregateMode::Final, true, StreamType::OrderedFinalAggregate(_)) + | (AggregateMode::Final, true, StreamType::ClusteredFinalAggregate(_)) | (AggregateMode::Single, false, StreamType::SingleHash(_)) - | (AggregateMode::Single, true, StreamType::OrderedSingleAggregate(_)) => {} + | (AggregateMode::Single, true, StreamType::ClusteredSingleAggregate(_)) => {} _ => panic!("unexpected stream for {mode:?}, ordered={ordered}"), } let reserved = fits.memory_pool().reserved(); @@ -5602,7 +5619,116 @@ mod tests { Ok(()) } - /// Ensures `OrderedSingleAggregateStream` is used for ordered raw input. + #[rstest::rstest] + #[case::full(true)] + #[case::partial(false)] + fn group_clustering_preserves_sort_options( + #[case] full: bool, + #[values(AggregateMode::Partial, AggregateMode::Single)] mode: AggregateMode, + #[values(false, true)] descending: bool, + #[values(false, true)] nulls_first: bool, + ) -> Result<()> { + let schema = Arc::new(Schema::new(vec![ + Field::new("a", DataType::Int32, true), + Field::new("b", DataType::Int32, true), + Field::new("value", DataType::Int64, true), + ])); + let options = SortOptions::new(descending, nulls_first); + let mut input_ordering = vec![PhysicalSortExpr::new(col("a", &schema)?, options)]; + let mut output_ordering = vec![PhysicalSortExpr::new( + Arc::new(Column::new("group_a", 1)), + options, + )]; + if full { + input_ordering.push(PhysicalSortExpr::new(col("b", &schema)?, options)); + output_ordering.push(PhysicalSortExpr::new( + Arc::new(Column::new("group_b", 0)), + options, + )); + } + let input_ordering = LexOrdering::new(input_ordering).unwrap(); + let input = TestMemoryExec::try_new(&[vec![]], Arc::clone(&schema), None)? + .try_with_sort_information(vec![input_ordering.clone()])?; + let aggregate = AggregateExec::try_new( + mode, + PhysicalGroupBy::new_single(vec![ + (col("b", &schema)?, "group_b".to_string()), + (col("a", &schema)?, "group_a".to_string()), + ]), + vec![Arc::new( + AggregateExprBuilder::new(sum_udaf(), vec![col("value", &schema)?]) + .schema(Arc::clone(&schema)) + .alias("SUM(value)") + .build()?, + )], + vec![None], + Arc::new(TestMemoryExec::update_cache(&Arc::new(input))), + schema, + )?; + + let expected_mode = if full { + GroupClusteringMode::Full + } else { + GroupClusteringMode::Partial(vec![1]) + }; + assert_eq!(aggregate.group_clustering_mode(), &expected_mode); + assert_eq!(aggregate.maintains_input_order(), vec![true]); + assert_eq!( + aggregate.properties().emission_type, + EmissionType::Incremental + ); + assert_eq!( + aggregate.properties().output_ordering(), + LexOrdering::new(output_ordering).as_ref() + ); + assert_eq!( + aggregate.required_input_ordering(), + vec![Some(OrderingRequirements::new_soft(input_ordering.into()))] + ); + Ok(()) + } + + #[test] + fn group_clustering_is_recomputed_with_new_children() -> Result<()> { + let aggregate = Arc::new(single_test_aggregate()?); + let original_input = Arc::clone(aggregate.input()); + let schema = original_input.schema(); + let ordering = + LexOrdering::new([PhysicalSortExpr::new_default(col("a", &schema)?)]) + .unwrap(); + let sorted_input = TestMemoryExec::try_new(&[vec![]], schema, None)? + .try_with_sort_information(vec![ordering.clone()])?; + let sorted_input = + Arc::new(TestMemoryExec::update_cache(&Arc::new(sorted_input))); + + let sorted = aggregate.replace_children( + vec![sorted_input], + ReplaceChildrenOptions::new(ChildrenPropertiesMode::Recompute), + )?; + let sorted_aggregate = sorted.downcast_ref::().unwrap(); + assert_eq!( + sorted_aggregate.group_clustering_mode(), + &GroupClusteringMode::Full + ); + assert_eq!(sorted.output_ordering(), Some(&ordering)); + assert_eq!(sorted.pipeline_behavior(), EmissionType::Incremental); + + let unordered = sorted.replace_children( + vec![original_input], + ReplaceChildrenOptions::new(ChildrenPropertiesMode::Recompute), + )?; + let unordered_aggregate = unordered.downcast_ref::().unwrap(); + assert_eq!( + unordered_aggregate.group_clustering_mode(), + &GroupClusteringMode::None + ); + assert_eq!(unordered.maintains_input_order(), vec![false]); + assert!(unordered.output_ordering().is_none()); + assert_eq!(unordered.pipeline_behavior(), EmissionType::Final); + Ok(()) + } + + /// Ensures `ClusteredSingleAggregateStream` is used for ordered raw input. #[tokio::test] async fn ordered_single_aggregate_planning() -> Result<()> { let schema = Arc::new(Schema::new(vec![ @@ -5651,14 +5777,14 @@ mod tests { input, Arc::clone(&schema), )?; - assert!(matches!( - aggregate.input_order_mode(), - InputOrderMode::PartiallySorted(_) - )); + assert_eq!( + aggregate.group_clustering_mode(), + &GroupClusteringMode::Partial(vec![0]) + ); let task_ctx = new_migrated_hash_ctx(2); let stream = aggregate.execute_typed(0, &task_ctx)?; - assert!(matches!(stream, StreamType::OrderedSingleAggregate(_))); + assert!(matches!(stream, StreamType::ClusteredSingleAggregate(_))); let stream: SendableRecordBatchStream = stream.into(); let output = collect(stream).await?; assert_snapshot!(batches_to_sort_string(&output), @r" @@ -5674,13 +5800,13 @@ mod tests { let finite_memory_task_ctx = new_finite_memory_migrated_hash_ctx(2, 1024 * 1024)?; let stream = aggregate.execute_typed(0, &finite_memory_task_ctx)?; - assert!(matches!(stream, StreamType::OrderedSingleAggregate(_))); + assert!(matches!(stream, StreamType::ClusteredSingleAggregate(_))); Ok(()) } #[test] - fn partial_reduce_does_not_advertise_input_ordering() -> Result<()> { + fn partial_reduce_does_not_use_group_clustering() -> Result<()> { let schema = Arc::new(Schema::new(vec![ Field::new("a", DataType::UInt32, false), Field::new("b", DataType::Float64, false), @@ -5689,36 +5815,113 @@ mod tests { Column::new("a", 0), ))]) .unwrap(); - let input = TestMemoryExec::try_new(&[vec![]], Arc::clone(&schema), None)? - .try_with_sort_information(vec![ordering])?; - let input = Arc::new(TestMemoryExec::update_cache(&Arc::new(input))); - assert!( - input.properties().output_ordering().is_some(), - "test setup: the input is ordered by the group key" - ); + let ordered_input = + TestMemoryExec::try_new(&[vec![]], Arc::clone(&schema), None)? + .try_with_sort_information(vec![ordering])?; + let grouped_input = + TestMemoryExec::try_new(&[vec![]], Arc::clone(&schema), None)? + .try_with_grouping_information(vec![vec![col("a", &schema)?]])?; - let partial_reduce = AggregateExec::try_new( - AggregateMode::PartialReduce, - PhysicalGroupBy::new_single(vec![(col("a", &schema)?, "a".to_string())]), - vec![Arc::new( - AggregateExprBuilder::new(sum_udaf(), vec![col("b", &schema)?]) - .schema(Arc::clone(&schema)) - .alias("SUM(b)") - .build()?, - )], - vec![None], - input, + for input in [ordered_input, grouped_input] { + assert!( + input + .properties() + .equivalence_properties() + .grouping_satisfy([col("a", &schema)?])? + ); + let partial_reduce = AggregateExec::try_new( + AggregateMode::PartialReduce, + PhysicalGroupBy::new_single(vec![(col("a", &schema)?, "a".to_string())]), + vec![Arc::new( + AggregateExprBuilder::new(sum_udaf(), vec![col("b", &schema)?]) + .schema(Arc::clone(&schema)) + .alias("SUM(b)") + .build()?, + )], + vec![None], + Arc::new(input), + Arc::clone(&schema), + )?; + + assert_eq!( + partial_reduce.group_clustering_mode(), + &GroupClusteringMode::None + ); + assert_eq!(partial_reduce.maintains_input_order(), vec![false]); + assert!( + partial_reduce.properties().output_ordering().is_none(), + "partial reduce advertised an ordering it does not maintain: {:?}", + partial_reduce.properties().output_ordering() + ); + + let stream = partial_reduce.execute_typed(0, &new_migrated_hash_ctx(1024))?; + assert!(matches!(stream, StreamType::PartialReduceHash(_))); + } + + Ok(()) + } + + #[tokio::test] + async fn constant_only_grouping_uses_final_emission() -> Result<()> { + let schema = Arc::new(Schema::new(vec![ + Field::new("key", DataType::Int32, false), + Field::new("value", DataType::Int64, false), + ])); + let batch = RecordBatch::try_new( + Arc::clone(&schema), + vec![ + Arc::new(Int32Array::from(vec![1, 1])), + Arc::new(Int64Array::from(vec![10, 20])), + ], + )?; + let input = TestMemoryExec::try_new_exec( + &[vec![batch.clone(), batch]], Arc::clone(&schema), + None, )?; + let predicate = binary(col("key", &schema)?, Operator::Eq, lit(1i32), &schema)?; + let input: Arc = + Arc::new(FilterExecBuilder::new(predicate, input).build()?); - assert_eq!(partial_reduce.input_order_mode(), &InputOrderMode::Linear); - assert_eq!(partial_reduce.maintains_input_order(), vec![false]); - assert!( - partial_reduce.properties().output_ordering().is_none(), - "partial reduce advertised an ordering it does not maintain: {:?}", - partial_reduce.properties().output_ordering() - ); + // Cover both a literal and a column known to be constant after filtering. + for key in [lit(1i32), col("key", &schema)?] { + assert!( + input + .equivalence_properties() + .grouping_satisfy([Arc::clone(&key)])? + ); + let aggregate = AggregateExec::try_new( + AggregateMode::Single, + PhysicalGroupBy::new_single(vec![(key, "key".to_string())]), + vec![Arc::new( + AggregateExprBuilder::new(sum_udaf(), vec![col("value", &schema)?]) + .schema(Arc::clone(&schema)) + .alias("SUM(value)") + .build()?, + )], + vec![None], + Arc::clone(&input), + Arc::clone(&schema), + )?; + assert_eq!( + aggregate.group_clustering_mode(), + &GroupClusteringMode::None + ); + assert_eq!(aggregate.cache().emission_type, EmissionType::Final); + let stream = aggregate.execute_typed(0, &new_migrated_hash_ctx(1024))?; + assert!(matches!(stream, StreamType::SingleHash(_))); + let output = collect(stream.into()).await?; + allow_duplicates! { + assert_snapshot!(batches_to_string(&output), @r" + +-----+------------+ + | key | SUM(value) | + +-----+------------+ + | 1 | 60 | + +-----+------------+ + "); + } + } Ok(()) } @@ -5766,7 +5969,10 @@ mod tests { None, )?; let aggregate = build_aggregate(unordered_input)?; - assert_eq!(aggregate.input_order_mode(), &InputOrderMode::Linear); + assert_eq!( + aggregate.group_clustering_mode(), + &GroupClusteringMode::None + ); assert_eq!( aggregate.schema().as_ref(), &Schema::new(vec![ @@ -5798,7 +6004,10 @@ mod tests { let ordered_input = Arc::new(TestMemoryExec::update_cache(&Arc::new(ordered_input))); let aggregate = build_aggregate(ordered_input)?; - assert_eq!(aggregate.input_order_mode(), &InputOrderMode::Sorted); + assert_eq!( + aggregate.group_clustering_mode(), + &GroupClusteringMode::Full + ); Ok(()) } @@ -6065,37 +6274,38 @@ mod tests { } #[tokio::test] - async fn unsorted_contiguous_groups_use_final_emission() -> Result<()> { + async fn unsorted_contiguous_groups_use_incremental_emission() -> Result<()> { let schema = Arc::new(Schema::new(vec![ Field::new("key", DataType::Int32, false), Field::new("time_bin", DataType::Int64, false), Field::new("value", DataType::Int64, false), ])); - // Two sorted logical runs are emitted as batches in one DataFusion - // partition. Every distinct grouping tuple occupies one contiguous range, - // but tuple order resets at the batch boundary, so (key, time_bin) is not - // globally sorted. + // Every grouping tuple occupies one contiguous range in a single + // partition, with (2, 20) spanning both batches. Tuple order resets + // within the second batch, so (key, time_bin) is not globally sorted. let input_batches = vec![ RecordBatch::try_new( Arc::clone(&schema), vec![ - Arc::new(Int32Array::from(vec![1, 1, 2, 2])), - Arc::new(Int64Array::from(vec![20, 20, 20, 20])), - Arc::new(Int64Array::from(vec![10, 20, 30, 40])), + Arc::new(Int32Array::from(vec![1, 1, 2])), + Arc::new(Int64Array::from(vec![20, 20, 20])), + Arc::new(Int64Array::from(vec![10, 20, 30])), ], )?, RecordBatch::try_new( Arc::clone(&schema), vec![ - Arc::new(Int32Array::from(vec![1, 1, 2, 2])), - Arc::new(Int64Array::from(vec![0, 0, 0, 0])), - Arc::new(Int64Array::from(vec![50, 60, 70, 80])), + Arc::new(Int32Array::from(vec![2, 1, 1, 2, 2])), + Arc::new(Int64Array::from(vec![20, 0, 0, 0, 0])), + Arc::new(Int64Array::from(vec![40, 50, 60, 70, 80])), ], )?, ]; + let key = col("key", &schema)?; + let time_bin = col("time_bin", &schema)?; let group_by = PhysicalGroupBy::new_single(vec![ - (col("key", &schema)?, "key".to_string()), - (col("time_bin", &schema)?, "time_bin".to_string()), + (Arc::clone(&key), "key".to_string()), + (Arc::clone(&time_bin), "time_bin".to_string()), ]); let aggr_expr = Arc::new( AggregateExprBuilder::new(sum_udaf(), vec![col("value", &schema)?]) @@ -6103,30 +6313,65 @@ mod tests { .alias("SUM(value)") .build()?, ); - let input: Arc = - TestMemoryExec::try_new_exec(&[input_batches], Arc::clone(&schema), None)?; - assert_eq!(input.output_partitioning().partition_count(), 1); + let mut eq_properties = EquivalenceProperties::new(Arc::clone(&schema)); + eq_properties.add_groupings([vec![key, time_bin]]); + let input = Arc::new( + BarrierExec::new(vec![input_batches], Arc::clone(&schema)) + .without_start_barrier() + .with_batch_barrier() + .with_equivalence_properties(eq_properties) + .with_log(false), + ); + assert_eq!( + input.properties().output_partitioning().partition_count(), + 1 + ); let aggregate = AggregateExec::try_new( AggregateMode::Single, group_by, vec![aggr_expr], vec![None], - input, + Arc::clone(&input) as Arc, schema, )?; - assert_eq!(aggregate.input_order_mode(), &InputOrderMode::Linear); - // This captures the behavior before #24438. When the source can declare - // `(key, time_bin)` group-contiguous, the corresponding case can use - // `EmissionType::Incremental`. - assert_eq!(aggregate.cache().emission_type, EmissionType::Final); + assert_eq!( + aggregate.group_clustering_mode(), + &GroupClusteringMode::Full + ); + assert_eq!(aggregate.cache().emission_type, EmissionType::Incremental); + assert!( + aggregate + .cache() + .equivalence_properties() + .geq_class() + .is_empty() + ); + assert!(aggregate.cache().output_ordering().is_none()); let task_ctx = new_migrated_hash_ctx(1024); let stream = aggregate.execute_typed(0, &task_ctx)?; - assert!(matches!(stream, StreamType::SingleHash(_))); - let stream: SendableRecordBatchStream = stream.into(); - let output = collect(stream).await?; + assert!(matches!(stream, StreamType::ClusteredSingleAggregate(_))); + let mut stream: SendableRecordBatchStream = stream.into(); + input.wait_batch().await; + // The second batch and EOF remain blocked until the first completed + // group has been emitted. + let first = + tokio::time::timeout(std::time::Duration::from_secs(5), stream.next()) + .await + .expect("completed group was not emitted before the next input batch") + .expect("stream ended before the next input batch")?; + assert_snapshot!(batches_to_string(std::slice::from_ref(&first)), @r" ++-----+----------+------------+ +| key | time_bin | SUM(value) | ++-----+----------+------------+ +| 1 | 20 | 30 | ++-----+----------+------------+ +"); + input.wait_batch().await; + let mut output = vec![first]; + output.extend(collect(stream).await?); assert_snapshot!(batches_to_sort_string(&output), @r" +-----+----------+------------+ | key | time_bin | SUM(value) | @@ -6141,7 +6386,69 @@ mod tests { Ok(()) } - /// Ensures for ordered input, `OrderedPartialAggregateStream` is used. + #[test] + fn grouped_date_bin_projects_to_aggregate() -> Result<()> { + let schema = Arc::new(Schema::new(vec![ + Field::new("key", DataType::Int32, false), + Field::new("time", DataType::Timestamp(TimeUnit::Second, None), false), + ])); + let time_bin_expr = || -> Result> { + Ok(Arc::new(ScalarFunctionExpr::try_new( + date_bin(), + vec![ + lit(ScalarValue::new_interval_dt(0, 10_000)), + col("time", &schema)?, + ], + &schema, + Arc::new(ConfigOptions::default()), + )?)) + }; + let input = TestMemoryExec::try_new(&[vec![]], Arc::clone(&schema), None)? + .try_with_grouping_information(vec![vec![ + col("key", &schema)?, + time_bin_expr()?, + ]])?; + let projection = ProjectionExec::try_new( + [ + ProjectionExpr::new(col("key", &schema)?, "key"), + // Build this expression independently from the source + // assertion so the test exercises semantic expression matching. + ProjectionExpr::new(time_bin_expr()?, "time_bin"), + ], + Arc::new(input), + )?; + + let projected_schema = projection.schema(); + let key = col("key", &projected_schema)?; + let time_bin = col("time_bin", &projected_schema)?; + assert!( + projection + .properties() + .equivalence_properties() + .grouping_satisfy([Arc::clone(&key), Arc::clone(&time_bin)])? + ); + let aggregate = AggregateExec::try_new( + AggregateMode::Single, + PhysicalGroupBy::new_single(vec![ + (key, "key".to_string()), + (time_bin, "time_bin".to_string()), + ]), + vec![], + vec![], + Arc::new(projection), + projected_schema, + )?; + + assert!(aggregate.input().output_ordering().is_none()); + assert_eq!( + aggregate.group_clustering_mode(), + &GroupClusteringMode::Full + ); + assert_eq!(aggregate.cache().emission_type, EmissionType::Incremental); + Ok(()) + } + + /// Ensures for ordered input, `ClusteredPartialAggregateStream` is used. #[tokio::test] async fn ordered_partial_aggregate_planning() -> Result<()> { let schema = Arc::new(Schema::new(vec![ @@ -6195,13 +6502,13 @@ mod tests { Arc::clone(&schema), )?; assert!(matches!( - aggregate.input_order_mode(), - InputOrderMode::PartiallySorted(_) + aggregate.group_clustering_mode(), + GroupClusteringMode::Partial(_) )); let task_ctx = new_migrated_hash_ctx(2); let stream = aggregate.execute_typed(0, &task_ctx)?; - assert!(matches!(stream, StreamType::OrderedPartialAggregate(_))); + assert!(matches!(stream, StreamType::ClusteredPartialAggregate(_))); let stream: SendableRecordBatchStream = stream.into(); let output = collect(stream).await?; @@ -6217,15 +6524,15 @@ mod tests { +----------+-----------+-------------------------+ "); - // Ordered partial aggregation supports finite memory. + // Clustered partial aggregation supports finite memory. let finite_memory_task_ctx = new_finite_memory_migrated_hash_ctx(2, 1024 * 1024)?; let stream = aggregate.execute_typed(0, &finite_memory_task_ctx)?; - assert!(matches!(stream, StreamType::OrderedPartialAggregate(_))); + assert!(matches!(stream, StreamType::ClusteredPartialAggregate(_))); Ok(()) } - /// Ensures for ordered input, `OrderedFinalAggregateStream` is used. + /// Ensures for ordered input, `ClusteredFinalAggregateStream` is used. #[tokio::test] async fn ordered_final_aggregate_planning() -> Result<()> { let schema = Arc::new(Schema::new(vec![ @@ -6276,11 +6583,14 @@ mod tests { final_input, Arc::clone(&schema), )?; - assert_eq!(final_aggregate.input_order_mode(), &InputOrderMode::Sorted); + assert_eq!( + final_aggregate.group_clustering_mode(), + &GroupClusteringMode::Full + ); let task_ctx = new_migrated_hash_ctx(2); let stream = final_aggregate.execute_typed(0, &task_ctx)?; - assert!(matches!(stream, StreamType::OrderedFinalAggregate(_))); + assert!(matches!(stream, StreamType::ClusteredFinalAggregate(_))); let stream: SendableRecordBatchStream = stream.into(); let output = collect(stream).await?; @@ -6295,10 +6605,10 @@ mod tests { +-----+--------------+ "); - // Ordered final aggregation supports finite memory. + // Clustered final aggregation supports finite memory. let finite_memory_task_ctx = new_finite_memory_migrated_hash_ctx(2, 1024 * 1024)?; let stream = final_aggregate.execute_typed(0, &finite_memory_task_ctx)?; - assert!(matches!(stream, StreamType::OrderedFinalAggregate(_))); + assert!(matches!(stream, StreamType::ClusteredFinalAggregate(_))); Ok(()) } @@ -6350,8 +6660,8 @@ mod tests { Arc::clone(&schema), )?; assert!(matches!( - aggregate.input_order_mode(), - InputOrderMode::PartiallySorted(_) + aggregate.group_clustering_mode(), + GroupClusteringMode::Partial(_) )); let runtime = RuntimeEnvBuilder::default() @@ -6368,7 +6678,7 @@ mod tests { ); let mut stream: SendableRecordBatchStream = - OrderedPartialAggregateStream::new(&aggregate, &task_ctx, 0)?.into_stream(); + ClusteredPartialAggregateStream::new(&aggregate, &task_ctx, 0)?.into_stream(); while let Some(result) = stream.next().await { if let Err(e) = result { @@ -6649,7 +6959,7 @@ mod tests { // // "AggregateExec: mode=Final, gby=[a@0 as a], aggr=[FIRST_VALUE(b)]", // " CoalescePartitionsExec", - // " AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[FIRST_VALUE(b)], ordering_mode=None", + // " AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[FIRST_VALUE(b)], group_clustering_mode=None", // " DataSourceExec: partitions=4, partition_sizes=[1, 1, 1, 1]", // // and checks whether the function `merge_batch` works correctly for diff --git a/datafusion/physical-plan/src/aggregates/order/full.rs b/datafusion/physical-plan/src/aggregates/order/full.rs index ca818d6a2d598..11bed6e533f3f 100644 --- a/datafusion/physical-plan/src/aggregates/order/full.rs +++ b/datafusion/physical-plan/src/aggregates/order/full.rs @@ -18,10 +18,10 @@ use datafusion_expr::EmitTo; use std::mem::size_of; -/// Tracks grouping state when the data is ordered entirely by its -/// group keys +/// Tracks group completion when rows are contiguous for the complete +/// grouping tuple. /// -/// When the group values are sorted, as soon as we see group `n+1` we +/// When groups are contiguous, as soon as we see group `n+1` we /// know we will never see any rows for group `n` again and thus they /// can be emitted. /// @@ -55,7 +55,7 @@ use std::mem::size_of; /// `0..12` can be emitted. Note that `13` can not yet be emitted as /// there may be more values in the next batch with the same group_id. #[derive(Debug)] -pub struct GroupOrderingFull { +pub struct GroupClusteringFull { state: State, } @@ -72,7 +72,7 @@ enum State { Complete, } -impl GroupOrderingFull { +impl GroupClusteringFull { pub fn new() -> Self { Self { state: State::Start, @@ -115,13 +115,13 @@ impl GroupOrderingFull { self.state = State::Complete; } - /// Starts tracking a new fully ordered input segment. + /// Starts tracking a new input segment with contiguous groups. pub fn reset(&mut self) { self.state = State::Start; } /// Called when new groups are added in a batch. See documentation - /// on [`super::GroupOrdering::new_groups`] + /// on [`super::GroupClustering::new_groups`] pub fn new_groups(&mut self, total_num_groups: usize) { assert_ne!(total_num_groups, 0); @@ -149,7 +149,7 @@ impl GroupOrderingFull { } } -impl Default for GroupOrderingFull { +impl Default for GroupClusteringFull { fn default() -> Self { Self::new() } diff --git a/datafusion/physical-plan/src/aggregates/order/mod.rs b/datafusion/physical-plan/src/aggregates/order/mod.rs index 147e4e0d6f18d..8707a4f4820e7 100644 --- a/datafusion/physical-plan/src/aggregates/order/mod.rs +++ b/datafusion/physical-plan/src/aggregates/order/mod.rs @@ -24,48 +24,86 @@ use datafusion_expr::EmitTo; mod full; mod partial; -use crate::InputOrderMode; -pub use full::GroupOrderingFull; -pub use partial::GroupOrderingPartial; +pub use full::GroupClusteringFull; +pub use partial::GroupClusteringPartial; -/// Ordering information for each group in the hash table +/// Describes how rows are clustered by grouping expressions within each input +/// partition. +/// +/// Input ordering is one way to establish clustering. Aggregate execution uses +/// these guarantees to determine when groups are complete and can be emitted. +/// This mode does not describe the sort order of the input or output. +/// +/// For example, when grouping by `key`, both inputs have fully contiguous +/// groups within the input partition: +/// +/// ```text +/// sorted: A A B B C C +/// not sorted: C C A A B B +/// ``` +/// +/// In both cases, once the key changes, the previous key will not appear again, +/// so its group is complete and can be emitted. +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum GroupClusteringMode { + /// No clustering guarantee is available to complete groups before the input ends. + None, + /// Rows with the same values at these grouping-expression indices form one + /// contiguous range. When those values change, every group in the previous + /// range is complete and can be emitted. + /// + /// For example, with `GROUP BY (a, b)`, `Partial(vec![0])` means all rows + /// for each value of `a` are contiguous, while an `(a, b)` tuple may recur + /// within that range. + Partial(Vec), + /// Rows with the same complete grouping tuple form one contiguous range. + /// When the tuple changes, the previous group can be emitted. + Full, +} + +/// Tracks when groups in the hash table are complete and can be emitted. #[derive(Debug)] -pub enum GroupOrdering { - /// Groups are not ordered +pub enum GroupClustering { + /// No group can be known complete before the input ends. None, - /// Groups are ordered by some pre-set of the group keys - Partial(GroupOrderingPartial), - /// Groups are entirely contiguous, - Full(GroupOrderingFull), + /// Rows are contiguous for a subset of the grouping keys. + /// When those key values change, all groups in the previous run + /// are complete and can be emitted. + Partial(GroupClusteringPartial), + /// Rows are contiguous for the complete grouping tuple. + /// When the tuple changes, the previous group can be emitted. + Full(GroupClusteringFull), } -impl GroupOrdering { - /// Create a `GroupOrdering` for the specified ordering - pub fn try_new(mode: &InputOrderMode) -> Result { +impl GroupClustering { + /// Create a `GroupClustering` for the specified group-clustering mode. + pub fn try_new(mode: &GroupClusteringMode) -> Result { match mode { - InputOrderMode::Linear => Ok(GroupOrdering::None), - InputOrderMode::PartiallySorted(order_indices) => { - GroupOrderingPartial::try_new(order_indices.clone()) - .map(GroupOrdering::Partial) + GroupClusteringMode::None => Ok(GroupClustering::None), + GroupClusteringMode::Partial(grouping_indices) => { + GroupClusteringPartial::try_new(grouping_indices.clone()) + .map(GroupClustering::Partial) + } + GroupClusteringMode::Full => { + Ok(GroupClustering::Full(GroupClusteringFull::new())) } - InputOrderMode::Sorted => Ok(GroupOrdering::Full(GroupOrderingFull::new())), } } - /// Returns how many groups can be emitted while respecting the current - /// ordering guarantees, or `None` if no data can be emitted. + /// Returns how many completed groups can be emitted, or `None` if no data + /// can be emitted. pub fn emit_to(&self) -> Option { match self { - GroupOrdering::None => None, - GroupOrdering::Partial(partial) => partial.emit_to(), - GroupOrdering::Full(full) => full.emit_to(), + GroupClustering::None => None, + GroupClustering::Partial(partial) => partial.emit_to(), + GroupClustering::Full(full) => full.emit_to(), } } /// Returns the emit strategy to use under memory pressure (OOM). /// /// Returns the strategy that must be used when emitting up to `n` groups - /// while respecting the current ordering guarantees. + /// while respecting the configured group-clustering mode. /// /// Returns `None` if no data can be emitted. pub fn oom_emit_to(&self, n: usize) -> Option { @@ -74,8 +112,8 @@ impl GroupOrdering { } match self { - GroupOrdering::None => Some(EmitTo::First(n)), - GroupOrdering::Partial(_) | GroupOrdering::Full(_) => { + GroupClustering::None => Some(EmitTo::First(n)), + GroupClustering::Partial(_) | GroupClustering::Full(_) => { self.emit_to().map(|emit_to| match emit_to { EmitTo::First(max) => EmitTo::First(n.min(max)), EmitTo::All => EmitTo::First(n), @@ -87,23 +125,23 @@ impl GroupOrdering { /// Updates the state to indicate that the input is complete. pub fn input_done(&mut self) { match self { - GroupOrdering::None => {} - GroupOrdering::Partial(partial) => partial.input_done(), - GroupOrdering::Full(full) => full.input_done(), + GroupClustering::None => {} + GroupClustering::Partial(partial) => partial.input_done(), + GroupClustering::Full(full) => full.input_done(), } } - /// Resets the ordering state while preserving the configured ordering mode. + /// Resets the completion state while preserving the configured mode. /// - /// Ordered partial aggregation uses this after passing intermediate states - /// downstream, and ordered final aggregation uses it after spilling a run. + /// Clustered partial aggregation uses this after passing intermediate states + /// downstream, and clustered final aggregation uses it after spilling a run. /// In both cases the hash table is empty and can start tracking the next - /// input batch from a fresh ordering state. + /// input batch from a fresh completion state. pub fn reset(&mut self) { match self { - GroupOrdering::None => {} - GroupOrdering::Partial(partial) => partial.reset(), - GroupOrdering::Full(full) => full.reset(), + GroupClustering::None => {} + GroupClustering::Partial(partial) => partial.reset(), + GroupClustering::Full(full) => full.reset(), } } @@ -111,9 +149,9 @@ impl GroupOrdering { /// existing indexes down by `n`. pub fn remove_groups(&mut self, n: usize) { match self { - GroupOrdering::None => {} - GroupOrdering::Partial(partial) => partial.remove_groups(n), - GroupOrdering::Full(full) => full.remove_groups(n), + GroupClustering::None => {} + GroupClustering::Partial(partial) => partial.remove_groups(n), + GroupClustering::Full(full) => full.remove_groups(n), } } @@ -132,28 +170,28 @@ impl GroupOrdering { total_num_groups: usize, ) -> Result<()> { match self { - GroupOrdering::None => {} - GroupOrdering::Partial(partial) => { + GroupClustering::None => {} + GroupClustering::Partial(partial) => { partial.new_groups( batch_group_values, group_indices, total_num_groups, )?; } - GroupOrdering::Full(full) => { + GroupClustering::Full(full) => { full.new_groups(total_num_groups); } } Ok(()) } - /// Returns the size of memory used by the ordering state, in bytes. + /// Returns the size of memory used by the completion state, in bytes. pub fn size(&self) -> usize { size_of::() + match self { - GroupOrdering::None => 0, - GroupOrdering::Partial(partial) => partial.size(), - GroupOrdering::Full(full) => full.size(), + GroupClustering::None => 0, + GroupClustering::Partial(partial) => partial.size(), + GroupClustering::Full(full) => full.size(), } } } @@ -167,52 +205,52 @@ mod tests { use arrow::array::Int32Array; #[test] - fn test_oom_emit_to_none_ordering() { - let group_ordering = GroupOrdering::None; + fn test_oom_emit_to_none_clustering() { + let group_clustering = GroupClustering::None; - assert_eq!(group_ordering.oom_emit_to(0), None); - assert_eq!(group_ordering.oom_emit_to(5), Some(EmitTo::First(5))); + assert_eq!(group_clustering.oom_emit_to(0), None); + assert_eq!(group_clustering.oom_emit_to(5), Some(EmitTo::First(5))); } - /// Creates a partially ordered grouping state with three groups. + /// Creates a partial group-clustering tracker with three groups. /// - /// `sort_key_values` controls whether a sort boundary exists in the batch: + /// `group_key_values` controls whether a run boundary exists in the batch: /// distinct values such as `[1, 2, 3]` create boundaries, while repeated /// values such as `[1, 1, 1]` do not. - fn partial_ordering(sort_key_values: Vec) -> Result { - let mut group_ordering = - GroupOrdering::Partial(GroupOrderingPartial::try_new(vec![0])?); + fn partial_clustering(group_key_values: Vec) -> Result { + let mut group_clustering = + GroupClustering::Partial(GroupClusteringPartial::try_new(vec![0])?); let batch_group_values: Vec = vec![ - Arc::new(Int32Array::from(sort_key_values)), + Arc::new(Int32Array::from(group_key_values)), Arc::new(Int32Array::from(vec![10, 20, 30])), ]; let group_indices = vec![0, 1, 2]; - group_ordering.new_groups(&batch_group_values, &group_indices, 3)?; + group_clustering.new_groups(&batch_group_values, &group_indices, 3)?; - Ok(group_ordering) + Ok(group_clustering) } #[test] fn test_oom_emit_to_partial_clamps_to_boundary() -> Result<()> { - let group_ordering = partial_ordering(vec![1, 2, 3])?; + let group_clustering = partial_clustering(vec![1, 2, 3])?; // Can emit both `1` and `2` groups because we have seen `3` - assert_eq!(group_ordering.emit_to(), Some(EmitTo::First(2))); - assert_eq!(group_ordering.oom_emit_to(1), Some(EmitTo::First(1))); - assert_eq!(group_ordering.oom_emit_to(3), Some(EmitTo::First(2))); + assert_eq!(group_clustering.emit_to(), Some(EmitTo::First(2))); + assert_eq!(group_clustering.oom_emit_to(1), Some(EmitTo::First(1))); + assert_eq!(group_clustering.oom_emit_to(3), Some(EmitTo::First(2))); Ok(()) } #[test] fn test_oom_emit_to_partial_without_boundary() -> Result<()> { - let group_ordering = partial_ordering(vec![1, 1, 1])?; + let group_clustering = partial_clustering(vec![1, 1, 1])?; // Can't emit the last `1` group as it may have more values - assert_eq!(group_ordering.emit_to(), None); - assert_eq!(group_ordering.oom_emit_to(3), None); + assert_eq!(group_clustering.emit_to(), None); + assert_eq!(group_clustering.oom_emit_to(3), None); Ok(()) } diff --git a/datafusion/physical-plan/src/aggregates/order/partial.rs b/datafusion/physical-plan/src/aggregates/order/partial.rs index 1603bb6d079be..c02d19ef4e472 100644 --- a/datafusion/physical-plan/src/aggregates/order/partial.rs +++ b/datafusion/physical-plan/src/aggregates/order/partial.rs @@ -27,12 +27,11 @@ use datafusion_common::{Result, ScalarValue}; use datafusion_execution::memory_pool::proxy::VecAllocExt; use datafusion_expr::EmitTo; -/// Tracks grouping state when the data is ordered by some subset of +/// Tracks group completion when rows are contiguous for a subset of /// the group keys. /// -/// Once the next *sort key* value is seen, never see groups with that -/// sort key again, so we can emit all groups with the previous sort -/// key and earlier. +/// Once those key values change, they will not appear again, so all groups +/// in the previous run are complete and can be emitted. /// /// For example, given `SUM(amt) GROUP BY id, state` if the input is /// sorted by `state`, when a new value of `state` is seen, all groups @@ -44,11 +43,11 @@ use datafusion_expr::EmitTo; /// ┏━━━━━━━━━━━━━━━━━┓ ┏━━━━━━━┓ /// ┌─────┐ ┌───────────────────┐ ┌─────┃ 9 ┃ ┃ "MD" ┃ /// │┌───┐│ │ ┌──────────────┐ │ │ ┗━━━━━━━━━━━━━━━━━┛ ┗━━━━━━━┛ -/// ││ 0 ││ │ │ 123, "MA" │ │ │ current_sort sort_key +/// ││ 0 ││ │ │ 123, "MA" │ │ │ current_run_start group_key /// │└───┘│ │ └──────────────┘ │ │ -/// │ ... │ │ ... │ │ current_sort tracks the +/// │ ... │ │ ... │ │ current_run_start tracks the /// │┌───┐│ │ ┌──────────────┐ │ │ smallest group index that had -/// ││ 8 ││ │ │ 765, "MA" │ │ │ the same sort_key as current +/// ││ 8 ││ │ │ 765, "MA" │ │ │ the same group_key as current /// │├───┤│ │ ├──────────────┤ │ │ /// ││ 9 ││ │ │ 923, "MD" │◀─┼─┘ /// │├───┤│ │ ├──────────────┤ │ ┏━━━━━━━━━━━━━━┓ @@ -63,22 +62,22 @@ use datafusion_expr::EmitTo; /// order) recent group index /// ``` #[derive(Debug)] -pub struct GroupOrderingPartial { +pub struct GroupClusteringPartial { /// State machine state: State, - /// The indexes of the group by columns that form the sort key. - /// For example if grouping by `id, state` and ordered by `state` + /// The indexes of the group by columns whose values form contiguous runs. + /// For example if grouping by `id, state` and contiguous on `state` /// this would be `[1]`. - order_indices: Vec, + grouping_indices: Vec, } #[derive(Debug, Default, PartialEq)] enum State { - /// The ordering was temporarily taken. `Self::Taken` is left + /// The state was temporarily taken. `Self::Taken` is left /// when state must be temporarily taken to satisfy the borrow /// checker. If an error happens before the state can be restored, - /// the ordering information is lost and execution can not + /// the completion information is lost and execution can not /// proceed, but there is no undefined behavior. #[default] Taken, @@ -88,10 +87,10 @@ enum State { /// Data is in progress. InProgress { - /// Smallest group index with the sort_key - current_sort: usize, - /// The sort key of group_index `current_sort` - sort_key: Vec, + /// Smallest group index in the current run. + current_run_start: usize, + /// The key values of the current run. + group_key: Vec, /// index of the current group for which values are being /// generated current: usize, @@ -106,7 +105,7 @@ impl State { match self { State::Taken => 0, State::Start => 0, - State::InProgress { sort_key, .. } => sort_key + State::InProgress { group_key, .. } => group_key .iter() .map(|scalar_value| scalar_value.size()) .sum(), @@ -115,24 +114,23 @@ impl State { } } -impl GroupOrderingPartial { - /// TODO: Remove unnecessary `input_schema` parameter. - pub fn try_new(order_indices: Vec) -> Result { - debug_assert!(!order_indices.is_empty()); +impl GroupClusteringPartial { + /// Creates a tracker for runs defined by the specified grouping columns. + pub fn try_new(grouping_indices: Vec) -> Result { + debug_assert!(!grouping_indices.is_empty()); Ok(Self { state: State::Start, - order_indices, + grouping_indices, }) } - /// Select sort keys from the group values + /// Select the keys that define contiguous runs from the group values. /// - /// For example, if group_values had `A, B, C` but the input was - /// only sorted on `B` and `C` this should return rows for (`B`, - /// `C`) - fn compute_sort_keys(&mut self, group_values: &[ArrayRef]) -> Vec { - // Take only the columns that are in the sort key - self.order_indices + /// For example, if `group_values` contains `A, B, C` but the input is + /// contiguous on `(B, C)`, this returns the arrays for `B` and `C`. + fn compute_group_keys(&mut self, group_values: &[ArrayRef]) -> Vec { + // Take only the columns that define contiguous runs. + self.grouping_indices .iter() .map(|&idx| Arc::clone(&group_values[idx])) .collect() @@ -143,14 +141,14 @@ impl GroupOrderingPartial { match &self.state { State::Taken => unreachable!("State previously taken"), State::Start => None, - State::InProgress { current_sort, .. } => { - // Can not emit if we are still on the first row sort - // row otherwise we can emit all groups that had earlier sort keys - // - if *current_sort == 0 { + State::InProgress { + current_run_start, .. + } => { + // The current run is incomplete; only groups from earlier runs can be emitted. + if *current_run_start == 0 { None } else { - Some(EmitTo::First(*current_sort)) + Some(EmitTo::First(*current_run_start)) } } State::Complete => Some(EmitTo::All), @@ -164,15 +162,15 @@ impl GroupOrderingPartial { State::Taken => unreachable!("State previously taken"), State::Start => panic!("invalid state: start"), State::InProgress { - current_sort, + current_run_start, current, - sort_key: _, + group_key: _, } => { // shift indexes down by n assert!(*current >= n); *current -= n; - assert!(*current_sort >= n); - *current_sort -= n; + assert!(*current_run_start >= n); + *current_run_start -= n; } State::Complete => panic!("invalid state: complete"), } @@ -186,31 +184,31 @@ impl GroupOrderingPartial { }; } - /// Starts tracking a new ordered input segment with the same sort-key + /// Starts tracking a new input segment with the same contiguous-key /// columns. pub fn reset(&mut self) { self.state = State::Start; } - fn updated_sort_key( - current_sort: usize, - sort_key: Option>, - range_current_sort: usize, - range_sort_key: Vec, + fn updated_group_key( + current_run_start: usize, + group_key: Option>, + range_current_run_start: usize, + range_group_key: Vec, ) -> Result<(usize, Vec)> { - if let Some(sort_key) = sort_key { - let sort_options = vec![SortOptions::new(false, false); sort_key.len()]; - let ordering = compare_rows(&sort_key, &range_sort_key, &sort_options)?; + if let Some(group_key) = group_key { + let sort_options = vec![SortOptions::new(false, false); group_key.len()]; + let ordering = compare_rows(&group_key, &range_group_key, &sort_options)?; if ordering == Ordering::Equal { - return Ok((current_sort, sort_key)); + return Ok((current_run_start, group_key)); } } - Ok((range_current_sort, range_sort_key)) + Ok((range_current_run_start, range_group_key)) } /// Called when new groups are added in a batch. See documentation - /// on [`super::GroupOrdering::new_groups`] + /// on [`super::GroupClustering::new_groups`] pub fn new_groups( &mut self, batch_group_values: &[ArrayRef], @@ -222,46 +220,46 @@ impl GroupOrderingPartial { let max_group_index = total_num_groups - 1; - let (current_sort, sort_key) = match std::mem::take(&mut self.state) { + let (current_run_start, group_key) = match std::mem::take(&mut self.state) { State::Taken => unreachable!("State previously taken"), State::Start => (0, None), State::InProgress { - current_sort, - sort_key, + current_run_start, + group_key, .. - } => (current_sort, Some(sort_key)), + } => (current_run_start, Some(group_key)), State::Complete => { panic!("Saw new group after the end of input"); } }; - // Select the sort key columns - let sort_keys = self.compute_sort_keys(batch_group_values); + // Select the columns that define contiguous runs. + let group_keys = self.compute_group_keys(batch_group_values); - // Check if the sort keys indicate a boundary inside the batch - let ranges = partition(&sort_keys)?.ranges(); + // Check if the key values indicate a boundary inside the batch. + let ranges = partition(&group_keys)?.ranges(); let last_range = ranges.last().unwrap(); - let range_current_sort = group_indices[last_range.start]; - let range_sort_key = get_row_at_idx(&sort_keys, last_range.start)?; + let range_current_run_start = group_indices[last_range.start]; + let range_group_key = get_row_at_idx(&group_keys, last_range.start)?; - let (current_sort, sort_key) = if last_range.start == 0 { - // There was no boundary in the batch. Compare with the previous sort_key (if present) + let (current_run_start, group_key) = if last_range.start == 0 { + // There was no boundary in the batch. Compare with the previous group_key (if present) // to check if there was a boundary between the current batch and the previous one. - Self::updated_sort_key( - current_sort, - sort_key, - range_current_sort, - range_sort_key, + Self::updated_group_key( + current_run_start, + group_key, + range_current_run_start, + range_group_key, )? } else { - (range_current_sort, range_sort_key) + (range_current_run_start, range_group_key) }; self.state = State::InProgress { - current_sort, + current_run_start, current: max_group_index, - sort_key, + group_key, }; Ok(()) @@ -269,7 +267,7 @@ impl GroupOrderingPartial { /// Return the size of memory allocated by this structure pub(crate) fn size(&self) -> usize { - size_of::() + self.order_indices.allocated_size() + self.state.size() + size_of::() + self.grouping_indices.allocated_size() + self.state.size() } } @@ -279,79 +277,85 @@ mod tests { use arrow::array::Int32Array; - #[test] - fn test_group_ordering_partial() -> Result<()> { - // Ordered on column a - let order_indices = vec![0]; - let mut group_ordering = GroupOrderingPartial::try_new(order_indices)?; + #[rstest::rstest] + #[case::sorted([1, 2, 3, 4])] + #[case::clustered([3, 1, 4, 2])] + fn test_group_clustering_partial(#[case] keys: [i32; 4]) -> Result<()> { + let [first, second, third, fourth] = keys; + // Contiguous on column a. + let grouping_indices = vec![0]; + let mut group_clustering = GroupClusteringPartial::try_new(grouping_indices)?; let batch_group_values: Vec = vec![ - Arc::new(Int32Array::from(vec![1, 2, 3])), + Arc::new(Int32Array::from(vec![first, second, third])), Arc::new(Int32Array::from(vec![2, 1, 3])), ]; let group_indices = vec![0, 1, 2]; let total_num_groups = 3; - group_ordering.new_groups( + group_clustering.new_groups( &batch_group_values, &group_indices, total_num_groups, )?; assert_eq!( - group_ordering.state, + group_clustering.state, State::InProgress { - current_sort: 2, - sort_key: vec![ScalarValue::Int32(Some(3))], + current_run_start: 2, + group_key: vec![ScalarValue::Int32(Some(third))], current: 2 } ); + assert_eq!(group_clustering.emit_to(), Some(EmitTo::First(2))); // push without a boundary let batch_group_values: Vec = vec![ - Arc::new(Int32Array::from(vec![3, 3, 3])), + Arc::new(Int32Array::from(vec![third, third, third])), Arc::new(Int32Array::from(vec![2, 1, 7])), ]; let group_indices = vec![3, 4, 5]; let total_num_groups = 6; - group_ordering.new_groups( + group_clustering.new_groups( &batch_group_values, &group_indices, total_num_groups, )?; assert_eq!( - group_ordering.state, + group_clustering.state, State::InProgress { - current_sort: 2, - sort_key: vec![ScalarValue::Int32(Some(3))], + current_run_start: 2, + group_key: vec![ScalarValue::Int32(Some(third))], current: 5 } ); + assert_eq!(group_clustering.emit_to(), Some(EmitTo::First(2))); // push with only a boundary to previous batch let batch_group_values: Vec = vec![ - Arc::new(Int32Array::from(vec![4, 4, 4])), - Arc::new(Int32Array::from(vec![1, 1, 1])), + Arc::new(Int32Array::from(vec![fourth, fourth, fourth])), + Arc::new(Int32Array::from(vec![1, 2, 3])), ]; let group_indices = vec![6, 7, 8]; let total_num_groups = 9; - group_ordering.new_groups( + group_clustering.new_groups( &batch_group_values, &group_indices, total_num_groups, )?; assert_eq!( - group_ordering.state, + group_clustering.state, State::InProgress { - current_sort: 6, - sort_key: vec![ScalarValue::Int32(Some(4))], + current_run_start: 6, + group_key: vec![ScalarValue::Int32(Some(fourth))], current: 8 } ); + assert_eq!(group_clustering.emit_to(), Some(EmitTo::First(6))); Ok(()) } diff --git a/datafusion/physical-plan/src/aggregates/partial_reduce_stream.rs b/datafusion/physical-plan/src/aggregates/partial_reduce_stream.rs index d5057d53cf19d..941edca2dd9fd 100644 --- a/datafusion/physical-plan/src/aggregates/partial_reduce_stream.rs +++ b/datafusion/physical-plan/src/aggregates/partial_reduce_stream.rs @@ -29,10 +29,11 @@ use datafusion_execution::memory_pool::{MemoryConsumer, MemoryReservation}; use futures::stream::{Stream, StreamExt}; use super::AggregateExec; +use super::GroupClusteringMode; use super::aggregate_hash_table::{AggregateHashTable, PartialReduceMarker}; use crate::metrics::{BaselineMetrics, Count, MetricBuilder, RecordOutput, SpillMetrics}; use crate::stream::EmptyRecordBatchStream; -use crate::{InputOrderMode, RecordBatchStream, SendableRecordBatchStream}; +use crate::{RecordBatchStream, SendableRecordBatchStream}; /// Hash aggregation can combine multiple partial stages before final /// evaluation. This stream implements the partial-reduce stage. @@ -182,7 +183,7 @@ impl PartialReduceHashAggregateStream { partition: usize, ) -> Result { debug_assert_eq!(agg.mode, super::AggregateMode::PartialReduce); - debug_assert_eq!(agg.input_order_mode, InputOrderMode::Linear); + debug_assert_eq!(agg.group_clustering_mode, GroupClusteringMode::None); let schema = Arc::clone(&agg.schema); let input = agg.input.execute(partition, Arc::clone(context))?; diff --git a/datafusion/physical-plan/src/aggregates/single_stream.rs b/datafusion/physical-plan/src/aggregates/single_stream.rs index f27017784bfd3..8d2edb7407244 100644 --- a/datafusion/physical-plan/src/aggregates/single_stream.rs +++ b/datafusion/physical-plan/src/aggregates/single_stream.rs @@ -30,14 +30,15 @@ use datafusion_execution::{TaskContext, TryEmitter, async_try_stream}; use futures::stream::StreamExt; use super::aggregate_hash_table::{ - AggregateHashTable, OrderedAggregateTableMetrics, SingleMarker, + AggregateHashTable, ClusteredAggregateTableMetrics, SingleMarker, }; +use super::order::GroupClusteringMode; use super::spill::AggregateSpill; use super::{AggregateExec, create_schema}; +use crate::SendableRecordBatchStream; use crate::aggregates::AggregateMode; use crate::metrics::{BaselineMetrics, SpillMetrics}; use crate::stream::{ObservedStream, RecordBatchStreamAdapter}; -use crate::{InputOrderMode, SendableRecordBatchStream}; /// Hash aggregation can run the full logical aggregation in one operator. This /// stream implements the single stage for grouped hash aggregation. @@ -82,7 +83,7 @@ use crate::{InputOrderMode, SendableRecordBatchStream}; /// 3. Perform a sort-preserving merge of all spill files and feed the merged output /// into an ordered streaming aggregation, which ensures bounded memory usage and /// evaluates the final result. -/// - [`OrderedFinalAggregateStream`](super::ordered_final_stream::OrderedFinalAggregateStream) is reused for the streaming aggregation. +/// - [`ClusteredFinalAggregateStream`](super::clustered_final_stream::ClusteredFinalAggregateStream) is reused for the streaming aggregation. /// /// # Optimization: DISTINCT LIMIT Soft Limit /// @@ -162,7 +163,7 @@ impl SingleHashAggregateStream { agg.mode, AggregateMode::Single | AggregateMode::SinglePartitioned )); - debug_assert_eq!(agg.input_order_mode, InputOrderMode::Linear); + debug_assert_eq!(agg.group_clustering_mode, GroupClusteringMode::None); let schema = Arc::clone(&agg.schema); let input = agg.input.execute(partition, Arc::clone(context))?; @@ -193,7 +194,7 @@ impl SingleHashAggregateStream { context, partition, batch_size, - &InputOrderMode::Linear, + &GroupClusteringMode::None, &state_schema, spill_metrics, )?)) @@ -398,7 +399,7 @@ impl Aggregating { // Construct the replay stream: an ordered final aggregate stream // over the sort-preserving merge of all spill runs. - let metrics = OrderedAggregateTableMetrics::from_hash_table(&hash_table); + let metrics = ClusteredAggregateTableMetrics::from_hash_table(&hash_table); drop(hash_table); reservation.try_resize(0)?; // The outer ObservedStream counts output; replay only shares compute time. diff --git a/datafusion/physical-plan/src/aggregates/spill.rs b/datafusion/physical-plan/src/aggregates/spill.rs index 887be41855996..d2f763cfbd868 100644 --- a/datafusion/physical-plan/src/aggregates/spill.rs +++ b/datafusion/physical-plan/src/aggregates/spill.rs @@ -28,14 +28,15 @@ use datafusion_physical_expr::PhysicalSortExpr; use datafusion_physical_expr::expressions::Column; use datafusion_physical_expr_common::sort_expr::LexOrdering; -use super::aggregate_hash_table::OrderedAggregateTableMetrics; -use super::ordered_final_stream::OrderedFinalAggregateStream; +use super::aggregate_hash_table::ClusteredAggregateTableMetrics; +use super::clustered_final_stream::ClusteredFinalAggregateStream; +use super::order::GroupClusteringMode; use super::{AggregateExec, AggregateMode}; +use crate::SendableRecordBatchStream; use crate::metrics::{BaselineMetrics, SpillMetrics}; use crate::sorts::IncrementalSortIterator; use crate::sorts::streaming_merge::{SortedSpillFile, StreamingMergeBuilder}; use crate::spill::spill_manager::SpillManager; -use crate::{InputOrderMode, SendableRecordBatchStream}; /// Spill configuration and accumulated runs of one grouped aggregation stream. /// @@ -43,7 +44,7 @@ use crate::{InputOrderMode, SendableRecordBatchStream}; /// drains all currently buffered groups as intermediate state (see /// `take_state_batch` on the aggregate tables), sorts them by the full group /// key, and writes them to one spill file. After the original input ends, all -/// files are merged and replayed through an [`OrderedFinalAggregateStream`], +/// files are merged and replayed through a [`ClusteredFinalAggregateStream`], /// which merges the states and evaluates the final aggregate values. pub(super) struct AggregateSpill { /// Aggregate configuration used to construct the replay stream. @@ -92,7 +93,7 @@ pub(super) struct AggregateSpill { /// using the two previously sorted spill files. /// 2. Build a final aggregation stream: /// - The input is the SPM stream. - /// - It reuses `OrderedFinalAggregateStream` for processing. + /// - It reuses `ClusteredFinalAggregateStream` for processing. /// - It returns the final aggregation result directly. /// /// SPM output Final aggregate output @@ -126,10 +127,11 @@ impl AggregateSpill { /// Creates the spill context of a stream, whose spill requests are described /// as `label`. /// - /// `input_order_mode` is the order of the stream's input: spill files are - /// sorted by the already ordered group columns first, followed by the - /// remaining ones, so that replay keeps the ordering the stream promised. - /// Fully sorted input aggregates in bounded memory and never spills. + /// `group_clustering_mode` determines which group columns are already + /// contiguous. Spill files are sorted by those columns first, followed by + /// the remaining ones. Existing output sort options are retained so replay + /// preserves any advertised ordering as well as group clustering. + /// Full group clustering aggregates in bounded memory and never spills. /// /// `spill_schema` is the schema of the intermediate state batches. #[expect(clippy::too_many_arguments)] @@ -139,12 +141,12 @@ impl AggregateSpill { context: &Arc, partition: usize, batch_size: usize, - input_order_mode: &InputOrderMode, + group_clustering_mode: &GroupClusteringMode, spill_schema: &SchemaRef, spill_metrics: SpillMetrics, ) -> Result { let mut replay_agg = agg.clone(); - replay_agg.input_order_mode = InputOrderMode::Sorted; + replay_agg.group_clustering_mode = GroupClusteringMode::Full; let group_schema = match agg.mode { AggregateMode::Final | AggregateMode::FinalPartitioned => { agg.group_by().group_schema(spill_schema)? @@ -164,17 +166,16 @@ impl AggregateSpill { }; let num_group_columns = group_schema.fields().len(); - let ordered_indices: &[usize] = match input_order_mode { - InputOrderMode::Linear => &[], - InputOrderMode::PartiallySorted(ordered_indices) => ordered_indices, - InputOrderMode::Sorted => { - return internal_err!("{label}: fully ordered input does not spill"); + let contiguous_indices: &[usize] = match group_clustering_mode { + GroupClusteringMode::None => &[], + GroupClusteringMode::Partial(contiguous_indices) => contiguous_indices, + GroupClusteringMode::Full => { + return internal_err!("{label}: fully contiguous groups do not spill"); } }; - let spill_indices = ordered_indices - .iter() - .copied() - .chain((0..num_group_columns).filter(|idx| !ordered_indices.contains(idx))); + let spill_indices = contiguous_indices.iter().copied().chain( + (0..num_group_columns).filter(|idx| !contiguous_indices.contains(idx)), + ); let output_ordering = agg.cache.output_ordering(); let spill_sort_exprs = spill_indices.map(|idx| { let output_expr = Column::new(group_schema.field(idx).name(), idx); @@ -249,11 +250,11 @@ impl AggregateSpill { } /// Merges every sorted run, and does the aggregate evaluation with - /// [`OrderedFinalAggregateStream`]. + /// [`ClusteredFinalAggregateStream`]. pub(super) fn into_replay_stream( self, baseline_metrics: &BaselineMetrics, - metrics: OrderedAggregateTableMetrics, + metrics: ClusteredAggregateTableMetrics, reservation: MemoryReservation, ) -> Result { let Self { @@ -284,12 +285,12 @@ impl AggregateSpill { .with_replay_headroom() .with_intermediate_merge_sizing(Some(min_spill_batch_rows)) .build()?; - let replay = OrderedFinalAggregateStream::new_with_input_and_metrics( + let replay = ClusteredFinalAggregateStream::new_with_input_and_metrics( &replay_agg, &context, partition, merged, - &InputOrderMode::Sorted, + &GroupClusteringMode::Full, baseline_metrics.clone(), metrics, None, diff --git a/datafusion/physical-plan/src/coalesce_partitions.rs b/datafusion/physical-plan/src/coalesce_partitions.rs index 9e3811e0ada76..4ce40ce43c964 100644 --- a/datafusion/physical-plan/src/coalesce_partitions.rs +++ b/datafusion/physical-plan/src/coalesce_partitions.rs @@ -98,6 +98,9 @@ impl CoalescePartitionsExec { // Coalescing partitions loses existing orderings: let mut eq_properties = input.equivalence_properties().clone(); eq_properties.clear_orderings(); + if input_partitions > 1 { + eq_properties.clear_groupings(); + } eq_properties.clear_per_partition_constants(); PlanProperties::new( eq_properties, // Equivalence Properties @@ -486,9 +489,39 @@ mod tests { use arrow::array::RecordBatch; use arrow::datatypes::{DataType, Field, Schema}; + use datafusion_physical_expr::expressions::col; use futures::FutureExt; + #[test] + fn grouping_is_cleared_only_when_partitions_are_merged() -> Result<()> { + let input = test::mem_exec(2); + let grouping = col("i", &input.schema())?; + let input = Arc::new(input.try_with_grouping_information(vec![vec![grouping]])?); + let merge = CoalescePartitionsExec::new(input); + assert!( + merge + .properties() + .equivalence_properties() + .geq_class() + .is_empty() + ); + + let input = test::mem_exec(1); + let grouping = col("i", &input.schema())?; + let input = Arc::new(input.try_with_grouping_information(vec![vec![grouping]])?); + let merge = CoalescePartitionsExec::new(input); + assert_eq!( + merge + .properties() + .equivalence_properties() + .geq_class() + .len(), + 1 + ); + Ok(()) + } + #[tokio::test] async fn merge() -> Result<()> { let task_ctx = Arc::new(TaskContext::default()); diff --git a/datafusion/physical-plan/src/limit.rs b/datafusion/physical-plan/src/limit.rs index ef504a40ea832..a7655c4febe9f 100644 --- a/datafusion/physical-plan/src/limit.rs +++ b/datafusion/physical-plan/src/limit.rs @@ -107,9 +107,13 @@ impl GlobalLimitExec { /// This function creates the cache object that stores the plan properties such as schema, equivalence properties, ordering, partitioning, etc. fn compute_properties(input: &Arc) -> PlanProperties { + let mut eq_properties = input.equivalence_properties().clone(); + if input.output_partitioning().partition_count() > 1 { + eq_properties.clear_groupings(); + } PlanProperties::new( - input.equivalence_properties().clone(), // Equivalence Properties - Partitioning::UnknownPartitioning(1), // Output Partitioning + eq_properties, // Equivalence Properties + Partitioning::UnknownPartitioning(1), // Output Partitioning input.pipeline_behavior(), // Limit operations are always bounded since they output a finite number of rows Boundedness::Bounded, @@ -787,6 +791,34 @@ mod tests { use datafusion_physical_expr::expressions::col; use datafusion_physical_expr::{PhysicalExpr, PhysicalSortExpr}; + #[test] + fn limits_preserve_grouping_without_merging_partitions() -> Result<()> { + let input = test::mem_exec(2); + let grouping = col("i", &input.schema())?; + let input: Arc = + Arc::new(input.try_with_grouping_information(vec![vec![grouping]])?); + + let local = LocalLimitExec::new(Arc::clone(&input), 10); + assert_eq!( + local + .properties() + .equivalence_properties() + .geq_class() + .len(), + 1 + ); + + let global = GlobalLimitExec::new(input, 0, Some(10)); + assert!( + global + .properties() + .equivalence_properties() + .geq_class() + .is_empty() + ); + Ok(()) + } + #[tokio::test] async fn limit() -> Result<()> { let task_ctx = Arc::new(TaskContext::default()); diff --git a/datafusion/physical-plan/src/recursive_query.rs b/datafusion/physical-plan/src/recursive_query.rs index 0a56488de84dd..494887f9068ef 100644 --- a/datafusion/physical-plan/src/recursive_query.rs +++ b/datafusion/physical-plan/src/recursive_query.rs @@ -23,7 +23,7 @@ use std::task::{Context, Poll}; use super::work_table::{ReservedBatches, WorkTable}; use crate::aggregates::group_values::{GroupValues, new_group_values}; -use crate::aggregates::order::GroupOrdering; +use crate::aggregates::order::GroupClustering; use crate::common::project_plan_to_schema; use crate::execution_plan::{Boundedness, EmissionType, reset_plan_states}; use crate::metrics::{ @@ -464,7 +464,7 @@ struct DistinctDeduplicator { impl DistinctDeduplicator { fn new(schema: SchemaRef, task_context: &TaskContext) -> Result { - let group_values = new_group_values(schema, &GroupOrdering::None)?; + let group_values = new_group_values(schema, &GroupClustering::None)?; let reservation = MemoryConsumer::new("RecursiveQueryHashTable") .register(task_context.memory_pool()); Ok(Self { diff --git a/datafusion/physical-plan/src/repartition/mod.rs b/datafusion/physical-plan/src/repartition/mod.rs index 3df29b577cd1a..1d36e4e2efd26 100644 --- a/datafusion/physical-plan/src/repartition/mod.rs +++ b/datafusion/physical-plan/src/repartition/mod.rs @@ -2199,8 +2199,10 @@ impl RepartitionExec { eq_properties.clear_orderings(); } // When there are more than one input partitions, they will be fused at the output. - // Therefore, remove per partition constants. + // Therefore, remove per-partition properties that may not hold after + // rows from different inputs are combined. if input.output_partitioning().partition_count() > 1 { + eq_properties.clear_groupings(); eq_properties.clear_per_partition_constants(); } eq_properties @@ -2666,6 +2668,39 @@ mod tests { }; use insta::assert_snapshot; + #[test] + fn grouping_is_cleared_only_when_input_partitions_are_merged() -> Result<()> { + let input = crate::test::mem_exec(2); + let grouping = col("i", &input.schema())?; + let input: Arc = + Arc::new(input.try_with_grouping_information(vec![vec![grouping]])?); + let repartition = + RepartitionExec::try_new(input, Partitioning::RoundRobinBatch(2))?; + assert!( + repartition + .properties() + .equivalence_properties() + .geq_class() + .is_empty() + ); + + let input = crate::test::mem_exec(1); + let grouping = col("i", &input.schema())?; + let input: Arc = + Arc::new(input.try_with_grouping_information(vec![vec![grouping]])?); + let repartition = + RepartitionExec::try_new(input, Partitioning::RoundRobinBatch(2))?; + assert_eq!( + repartition + .properties() + .equivalence_properties() + .geq_class() + .len(), + 1 + ); + Ok(()) + } + #[derive(Debug)] struct UnboundedTestPartition { schema: SchemaRef, diff --git a/datafusion/physical-plan/src/sorts/sort_preserving_merge.rs b/datafusion/physical-plan/src/sorts/sort_preserving_merge.rs index 7c924ff32bb04..f6901d59c594a 100644 --- a/datafusion/physical-plan/src/sorts/sort_preserving_merge.rs +++ b/datafusion/physical-plan/src/sorts/sort_preserving_merge.rs @@ -174,6 +174,9 @@ impl SortPreservingMergeExec { }; let mut eq_properties = input.equivalence_properties().clone(); + if input_partitions > 1 { + eq_properties.clear_groupings(); + } eq_properties.clear_per_partition_constants(); eq_properties.add_ordering(ordering); PlanProperties::new( @@ -901,6 +904,27 @@ mod tests { use insta::assert_snapshot; use tokio::time::timeout; + #[test] + fn merging_sorted_partitions_clears_explicit_grouping() -> Result<()> { + let input = test::mem_exec(2); + let grouping = col("i", &input.schema())?; + let ordering: LexOrdering = + [PhysicalSortExpr::new_default(Arc::clone(&grouping))].into(); + let input = input + .try_with_sort_information(vec![ordering.clone()])? + .try_with_grouping_information(vec![vec![grouping]])?; + let merge = SortPreservingMergeExec::new(ordering, Arc::new(input)); + + assert!( + merge + .properties() + .equivalence_properties() + .geq_class() + .is_empty() + ); + Ok(()) + } + // The number in the function is highly related to the memory limit we are testing // any change of the constant should be aware of fn generate_task_ctx_for_round_robin_tie_breaker( diff --git a/datafusion/physical-plan/src/test.rs b/datafusion/physical-plan/src/test.rs index b38a46d160755..3103196675f86 100644 --- a/datafusion/physical-plan/src/test.rs +++ b/datafusion/physical-plan/src/test.rs @@ -73,6 +73,8 @@ pub struct TestMemoryExec { projection: Option>, /// Sort information: one or more equivalent orderings sort_information: Vec, + /// Complete expression tuples whose values are contiguous. + grouping_information: Vec>>, /// if partition sizes should be displayed show_sizes: bool, /// The maximum number of records to read from this plan. If `None`, @@ -233,10 +235,12 @@ impl TestMemoryExec { } fn eq_properties(&self) -> EquivalenceProperties { - EquivalenceProperties::new_with_orderings( + let mut properties = EquivalenceProperties::new_with_orderings( Arc::clone(&self.projected_schema), self.sort_information.clone(), - ) + ); + properties.add_groupings(self.grouping_information.clone()); + properties } fn statistics_inner(&self) -> Result { @@ -268,6 +272,7 @@ impl TestMemoryExec { projected_schema, projection, sort_information: vec![], + grouping_information: vec![], show_sizes: true, fetch: None, }) @@ -363,6 +368,52 @@ impl TestMemoryExec { Ok(self) } + /// Adds grouping information to this source. + pub fn try_with_grouping_information( + mut self, + mut grouping_information: Vec>>, + ) -> Result { + // All grouping expressions must refer to the original schema. + let fields = self.schema.fields(); + let ambiguous_column = grouping_information + .iter() + .flatten() + .flat_map(collect_columns) + .find(|col| { + fields + .get(col.index()) + .map(|field| field.name() != col.name()) + .unwrap_or(true) + }); + assert_or_internal_err!( + ambiguous_column.is_none(), + "Column {:?} is not found in the original schema of the TestMemoryExec", + ambiguous_column.as_ref().unwrap() + ); + + if let Some(projection) = &self.projection { + let base_schema = self.original_schema(); + let proj_exprs = projection.iter().map(|idx| { + let name = base_schema.field(*idx).name(); + (Arc::new(Column::new(name, *idx)) as _, name.to_string()) + }); + let projection_mapping = + ProjectionMapping::try_new(proj_exprs, &base_schema)?; + let mut base_eqp = EquivalenceProperties::new(base_schema); + base_eqp.add_groupings(grouping_information); + grouping_information = base_eqp + .project(&projection_mapping, Arc::clone(&self.projected_schema)) + .geq_class() + .iter() + .cloned() + .collect(); + } + + self.grouping_information = grouping_information; + self.cache = Arc::new(self.compute_properties()); + Ok(self) + } + /// Arc clone of ref to original schema pub fn original_schema(&self) -> SchemaRef { Arc::clone(&self.schema) diff --git a/datafusion/physical-plan/src/test/exec.rs b/datafusion/physical-plan/src/test/exec.rs index dee87011b9812..362eb6976bc31 100644 --- a/datafusion/physical-plan/src/test/exec.rs +++ b/datafusion/physical-plan/src/test/exec.rs @@ -331,6 +331,9 @@ pub struct BarrierExec { /// all streams wait on this barrier to produce start_data_barrier: Option>, + /// each stream waits on this barrier before sending each batch + batch_barrier: Option>, + /// the stream wait for this to return Poll::Ready(None) finish_barrier: Option>, @@ -349,6 +352,7 @@ impl BarrierExec { data, schema, start_data_barrier: barrier, + batch_barrier: None, cache: Arc::new(cache), finish_barrier: None, log: true, @@ -365,6 +369,30 @@ impl BarrierExec { self } + /// Wait for [`Self::wait_batch`] before sending each batch. + pub fn with_batch_barrier(mut self) -> Self { + self.batch_barrier = Some(Arc::new(Barrier::new(self.data.len() + 1))); + self + } + + /// Release the next batch in each partition. + pub async fn wait_batch(&self) { + self.batch_barrier + .as_ref() + .expect("Must only be called when having a batch barrier") + .wait() + .await; + } + + /// Declare the equivalence properties of the test data. + pub fn with_equivalence_properties( + mut self, + properties: EquivalenceProperties, + ) -> Self { + self.cache = Arc::new(self.cache.as_ref().clone().with_eq_properties(properties)); + self + } + pub fn with_finish_barrier(mut self) -> Self { let barrier = Arc::new(( // wait for all streams and the input @@ -499,6 +527,7 @@ impl ExecutionPlan for BarrierExec { // task simply sends data in order after barrier is reached let data = self.data[partition].clone(); let start_barrier = self.start_data_barrier.as_ref().map(Arc::clone); + let batch_barrier = self.batch_barrier.as_ref().map(Arc::clone); let finish_barrier = self.finish_barrier.as_ref().map(Arc::clone); let log = self.log; let tx = builder.tx(); @@ -510,6 +539,9 @@ impl ExecutionPlan for BarrierExec { barrier.wait().await; } for batch in data { + if let Some(barrier) = &batch_barrier { + barrier.wait().await; + } if log { println!("Partition {partition} sending batch"); } diff --git a/datafusion/sqllogictest/test_files/agg_func_substitute.slt b/datafusion/sqllogictest/test_files/agg_func_substitute.slt index be9749bc9543e..f926b8d72df0e 100644 --- a/datafusion/sqllogictest/test_files/agg_func_substitute.slt +++ b/datafusion/sqllogictest/test_files/agg_func_substitute.slt @@ -44,9 +44,9 @@ logical_plan 03)----TableScan: multiple_ordered_table projection=[a, c] physical_plan 01)ProjectionExec: expr=[a@0 as a, nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]@1 as result] -02)--AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], ordering_mode=Sorted +02)--AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], group_clustering_mode=Full 03)----RepartitionExec: partitioning=Hash([a@0], 4), input_partitions=4, preserve_order=true, sort_exprs=a@0 ASC NULLS LAST -04)------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], ordering_mode=Sorted +04)------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], group_clustering_mode=Full 05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, maintains_sort_order=true 06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, c], output_orderings=[[a@0 ASC NULLS LAST], [c@1 ASC NULLS LAST]], file_type=csv, has_header=true @@ -62,9 +62,9 @@ logical_plan 03)----TableScan: multiple_ordered_table projection=[a, c] physical_plan 01)ProjectionExec: expr=[a@0 as a, nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]@1 as result] -02)--AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], ordering_mode=Sorted +02)--AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], group_clustering_mode=Full 03)----RepartitionExec: partitioning=Hash([a@0], 4), input_partitions=4, preserve_order=true, sort_exprs=a@0 ASC NULLS LAST -04)------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], ordering_mode=Sorted +04)------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c,Int64(1)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], group_clustering_mode=Full 05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, maintains_sort_order=true 06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, c], output_orderings=[[a@0 ASC NULLS LAST], [c@1 ASC NULLS LAST]], file_type=csv, has_header=true @@ -79,9 +79,9 @@ logical_plan 03)----TableScan: multiple_ordered_table projection=[a, c] physical_plan 01)ProjectionExec: expr=[a@0 as a, nth_value(multiple_ordered_table.c,Int64(1) + Int64(100)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]@1 as result] -02)--AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c, 101) ORDER BY [multiple_ordered_table.c ASC NULLS LAST] as nth_value(multiple_ordered_table.c,Int64(1) + Int64(100)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], ordering_mode=Sorted +02)--AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c, 101) ORDER BY [multiple_ordered_table.c ASC NULLS LAST] as nth_value(multiple_ordered_table.c,Int64(1) + Int64(100)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], group_clustering_mode=Full 03)----RepartitionExec: partitioning=Hash([a@0], 4), input_partitions=4, preserve_order=true, sort_exprs=a@0 ASC NULLS LAST -04)------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c, 101) ORDER BY [multiple_ordered_table.c ASC NULLS LAST] as nth_value(multiple_ordered_table.c,Int64(1) + Int64(100)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], ordering_mode=Sorted +04)------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[nth_value(multiple_ordered_table.c, 101) ORDER BY [multiple_ordered_table.c ASC NULLS LAST] as nth_value(multiple_ordered_table.c,Int64(1) + Int64(100)) ORDER BY [multiple_ordered_table.c ASC NULLS LAST]], group_clustering_mode=Full 05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1, maintains_sort_order=true 06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, c], output_orderings=[[a@0 ASC NULLS LAST], [c@1 ASC NULLS LAST]], file_type=csv, has_header=true diff --git a/datafusion/sqllogictest/test_files/aggregate.slt b/datafusion/sqllogictest/test_files/aggregate.slt index d72372dee47d3..812d4c0d25202 100644 --- a/datafusion/sqllogictest/test_files/aggregate.slt +++ b/datafusion/sqllogictest/test_files/aggregate.slt @@ -9540,7 +9540,7 @@ CREATE TABLE stream_test ( (3, 1.0, 1.0, 7, false, 'e'), (3, 2.0, 2.0, 8, false, 'f'); # Test comprehensive aggregates with streaming -# This verifies that CORR and other aggregates work together in a streaming plan (ordering_mode=Sorted) +# This verifies that CORR and other aggregates work together in a streaming plan (group_clustering_mode=Full) # Basic Aggregates query TT @@ -9579,7 +9579,7 @@ logical_plan 05)--------TableScan: stream_test projection=[g, x, y, i, b] physical_plan 01)ProjectionExec: expr=[g@0 as g, count(Int64(1))@1 as count(*), sum(stream_test.x)@2 as sum(stream_test.x), avg(stream_test.x)@3 as avg(stream_test.x), avg(stream_test.x)@3 as mean(stream_test.x), min(stream_test.x)@4 as min(stream_test.x), max(stream_test.y)@5 as max(stream_test.y), bit_and(stream_test.i)@6 as bit_and(stream_test.i), bit_or(stream_test.i)@7 as bit_or(stream_test.i), bit_xor(stream_test.i)@8 as bit_xor(stream_test.i), bool_and(stream_test.b)@9 as bool_and(stream_test.b), bool_or(stream_test.b)@10 as bool_or(stream_test.b), median(stream_test.x)@11 as median(stream_test.x), 0 as grouping(stream_test.g), var(stream_test.x)@12 as var(stream_test.x), var(stream_test.x)@12 as var_samp(stream_test.x), var_pop(stream_test.x)@13 as var_pop(stream_test.x), var(stream_test.x)@12 as var_sample(stream_test.x), var_pop(stream_test.x)@13 as var_population(stream_test.x), stddev(stream_test.x)@14 as stddev(stream_test.x), stddev(stream_test.x)@14 as stddev_samp(stream_test.x), stddev_pop(stream_test.x)@15 as stddev_pop(stream_test.x)] -02)--AggregateExec: mode=Single, gby=[g@0 as g], aggr=[count(Int64(1)), sum(stream_test.x), avg(stream_test.x), min(stream_test.x), max(stream_test.y), bit_and(stream_test.i), bit_or(stream_test.i), bit_xor(stream_test.i), bool_and(stream_test.b), bool_or(stream_test.b), median(stream_test.x), var(stream_test.x), var_pop(stream_test.x), stddev(stream_test.x), stddev_pop(stream_test.x)], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[g@0 as g], aggr=[count(Int64(1)), sum(stream_test.x), avg(stream_test.x), min(stream_test.x), max(stream_test.y), bit_and(stream_test.i), bit_or(stream_test.i), bit_xor(stream_test.i), bool_and(stream_test.b), bool_or(stream_test.b), median(stream_test.x), var(stream_test.x), var_pop(stream_test.x), stddev(stream_test.x), stddev_pop(stream_test.x)], group_clustering_mode=Full 03)----SortExec: TopK(fetch=10000), expr=[g@0 ASC NULLS LAST], preserve_partitioning=[false] 04)------DataSourceExec: partitions=1, partition_sizes=[1] @@ -9634,7 +9634,7 @@ logical_plan 03)----Sort: stream_test.g ASC NULLS LAST, fetch=10000 04)------TableScan: stream_test projection=[g, x] physical_plan -01)AggregateExec: mode=Single, gby=[g@0 as g], aggr=[array_agg(stream_test.x) ORDER BY [stream_test.x ASC NULLS LAST], array_agg(DISTINCT stream_test.x) ORDER BY [stream_test.x ASC NULLS LAST], first_value(stream_test.x) ORDER BY [stream_test.x ASC NULLS LAST], last_value(stream_test.x) ORDER BY [stream_test.x ASC NULLS LAST], nth_value(stream_test.x,Int64(1)) ORDER BY [stream_test.x ASC NULLS LAST]], ordering_mode=Sorted +01)AggregateExec: mode=Single, gby=[g@0 as g], aggr=[array_agg(stream_test.x) ORDER BY [stream_test.x ASC NULLS LAST], array_agg(DISTINCT stream_test.x) ORDER BY [stream_test.x ASC NULLS LAST], first_value(stream_test.x) ORDER BY [stream_test.x ASC NULLS LAST], last_value(stream_test.x) ORDER BY [stream_test.x ASC NULLS LAST], nth_value(stream_test.x,Int64(1)) ORDER BY [stream_test.x ASC NULLS LAST]], group_clustering_mode=Full 02)--SortExec: TopK(fetch=10000), expr=[g@0 ASC NULLS LAST, x@1 ASC NULLS LAST], preserve_partitioning=[false] 03)----DataSourceExec: partitions=1, partition_sizes=[1] @@ -9671,7 +9671,7 @@ logical_plan 03)----Sort: stream_test.g ASC NULLS LAST, fetch=10000 04)------TableScan: stream_test projection=[g, s] physical_plan -01)AggregateExec: mode=Single, gby=[g@0 as g], aggr=[array_agg(stream_test.s) ORDER BY [stream_test.s ASC NULLS LAST], string_agg(stream_test.s,Utf8("|")) ORDER BY [stream_test.s ASC NULLS LAST], string_agg(DISTINCT stream_test.s,Utf8("|")) ORDER BY [stream_test.s ASC NULLS LAST]], ordering_mode=Sorted +01)AggregateExec: mode=Single, gby=[g@0 as g], aggr=[array_agg(stream_test.s) ORDER BY [stream_test.s ASC NULLS LAST], string_agg(stream_test.s,Utf8("|")) ORDER BY [stream_test.s ASC NULLS LAST], string_agg(DISTINCT stream_test.s,Utf8("|")) ORDER BY [stream_test.s ASC NULLS LAST]], group_clustering_mode=Full 02)--SortExec: TopK(fetch=10000), expr=[g@0 ASC NULLS LAST, s@1 ASC NULLS LAST], preserve_partitioning=[false] 03)----DataSourceExec: partitions=1, partition_sizes=[1] @@ -9718,7 +9718,7 @@ logical_plan 05)--------TableScan: stream_test projection=[g, x, y] physical_plan 01)ProjectionExec: expr=[g@0 as g, corr(stream_test.x,stream_test.y)@1 as corr(stream_test.x,stream_test.y), covar_samp(stream_test.x,stream_test.y)@2 as covar(stream_test.x,stream_test.y), covar_samp(stream_test.x,stream_test.y)@2 as covar_samp(stream_test.x,stream_test.y), covar_pop(stream_test.x,stream_test.y)@3 as covar_pop(stream_test.x,stream_test.y), regr_sxx(stream_test.x,stream_test.y)@4 as regr_sxx(stream_test.x,stream_test.y), regr_sxy(stream_test.x,stream_test.y)@5 as regr_sxy(stream_test.x,stream_test.y), regr_syy(stream_test.x,stream_test.y)@6 as regr_syy(stream_test.x,stream_test.y), regr_avgx(stream_test.x,stream_test.y)@7 as regr_avgx(stream_test.x,stream_test.y), regr_avgy(stream_test.x,stream_test.y)@8 as regr_avgy(stream_test.x,stream_test.y), regr_count(stream_test.x,stream_test.y)@9 as regr_count(stream_test.x,stream_test.y), regr_slope(stream_test.x,stream_test.y)@10 as regr_slope(stream_test.x,stream_test.y), regr_intercept(stream_test.x,stream_test.y)@11 as regr_intercept(stream_test.x,stream_test.y), regr_r2(stream_test.x,stream_test.y)@12 as regr_r2(stream_test.x,stream_test.y)] -02)--AggregateExec: mode=Single, gby=[g@0 as g], aggr=[corr(stream_test.x,stream_test.y), covar_samp(stream_test.x,stream_test.y), covar_pop(stream_test.x,stream_test.y), regr_sxx(stream_test.x,stream_test.y), regr_sxy(stream_test.x,stream_test.y), regr_syy(stream_test.x,stream_test.y), regr_avgx(stream_test.x,stream_test.y), regr_avgy(stream_test.x,stream_test.y), regr_count(stream_test.x,stream_test.y), regr_slope(stream_test.x,stream_test.y), regr_intercept(stream_test.x,stream_test.y), regr_r2(stream_test.x,stream_test.y)], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[g@0 as g], aggr=[corr(stream_test.x,stream_test.y), covar_samp(stream_test.x,stream_test.y), covar_pop(stream_test.x,stream_test.y), regr_sxx(stream_test.x,stream_test.y), regr_sxy(stream_test.x,stream_test.y), regr_syy(stream_test.x,stream_test.y), regr_avgx(stream_test.x,stream_test.y), regr_avgy(stream_test.x,stream_test.y), regr_count(stream_test.x,stream_test.y), regr_slope(stream_test.x,stream_test.y), regr_intercept(stream_test.x,stream_test.y), regr_r2(stream_test.x,stream_test.y)], group_clustering_mode=Full 03)----SortExec: TopK(fetch=10000), expr=[g@0 ASC NULLS LAST], preserve_partitioning=[false] 04)------DataSourceExec: partitions=1, partition_sizes=[1] @@ -9771,7 +9771,7 @@ logical_plan 05)--------TableScan: stream_test projection=[g, x, i] physical_plan 01)ProjectionExec: expr=[g@0 as g, approx_distinct(stream_test.i)@1 as approx_distinct(stream_test.i), approx_median(stream_test.x)@2 as approx_median(stream_test.x), percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST]@3 as percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST]@3 as quantile_cont(stream_test.x,Float64(0.5)), approx_percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST]@4 as approx_percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], approx_percentile_cont_with_weight(Float64(1),Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST]@5 as approx_percentile_cont_with_weight(Float64(1),Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], percentile_cont(stream_test.x,Float64(0.5))@6 as percentile_cont(stream_test.x,Float64(0.5)), approx_percentile_cont(stream_test.x,Float64(0.5))@7 as approx_percentile_cont(stream_test.x,Float64(0.5)), approx_percentile_cont_with_weight(stream_test.x,Float64(1),Float64(0.5))@8 as approx_percentile_cont_with_weight(stream_test.x,Float64(1),Float64(0.5))] -02)--AggregateExec: mode=Single, gby=[g@0 as g], aggr=[approx_distinct(stream_test.i), approx_median(stream_test.x), percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], approx_percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], approx_percentile_cont_with_weight(Float64(1),Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], percentile_cont(stream_test.x,Float64(0.5)), approx_percentile_cont(stream_test.x,Float64(0.5)), approx_percentile_cont_with_weight(stream_test.x,Float64(1),Float64(0.5))], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[g@0 as g], aggr=[approx_distinct(stream_test.i), approx_median(stream_test.x), percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], approx_percentile_cont(Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], approx_percentile_cont_with_weight(Float64(1),Float64(0.5)) WITHIN GROUP [stream_test.x ASC NULLS LAST], percentile_cont(stream_test.x,Float64(0.5)), approx_percentile_cont(stream_test.x,Float64(0.5)), approx_percentile_cont_with_weight(stream_test.x,Float64(1),Float64(0.5))], group_clustering_mode=Full 03)----SortExec: TopK(fetch=10000), expr=[g@0 ASC NULLS LAST], preserve_partitioning=[false] 04)------DataSourceExec: partitions=1, partition_sizes=[1] diff --git a/datafusion/sqllogictest/test_files/group_by.slt b/datafusion/sqllogictest/test_files/group_by.slt index 9eb1865ecf107..281038f62ee9c 100644 --- a/datafusion/sqllogictest/test_files/group_by.slt +++ b/datafusion/sqllogictest/test_files/group_by.slt @@ -2112,7 +2112,7 @@ logical_plan 03)----TableScan: annotated_data_infinite2 projection=[a, b, c] physical_plan 01)ProjectionExec: expr=[a@1 as a, b@0 as b, sum(annotated_data_infinite2.c)@2 as summation1] -02)--AggregateExec: mode=Single, gby=[b@1 as b, a@0 as a], aggr=[sum(annotated_data_infinite2.c)], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[b@1 as b, a@0 as a], aggr=[sum(annotated_data_infinite2.c)], group_clustering_mode=Full 03)----StreamingTableExec: partition_sizes=1, projection=[a, b, c], infinite_source=true, output_ordering=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, c@2 ASC NULLS LAST] @@ -2143,7 +2143,7 @@ logical_plan 03)----TableScan: annotated_data_infinite2 projection=[a, c, d] physical_plan 01)ProjectionExec: expr=[a@1 as a, d@0 as d, sum(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]@2 as summation1] -02)--AggregateExec: mode=Single, gby=[d@2 as d, a@0 as a], aggr=[sum(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]], ordering_mode=PartiallySorted([1]) +02)--AggregateExec: mode=Single, gby=[d@2 as d, a@0 as a], aggr=[sum(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]], group_clustering_mode=Partial([1]) 03)----StreamingTableExec: partition_sizes=1, projection=[a, c, d], infinite_source=true, output_ordering=[a@0 ASC NULLS LAST] query III @@ -2176,7 +2176,7 @@ logical_plan 03)----TableScan: annotated_data_infinite2 projection=[a, b, c] physical_plan 01)ProjectionExec: expr=[a@0 as a, b@1 as b, first_value(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]@2 as first_c] -02)--AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[first_value(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[first_value(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]], group_clustering_mode=Full 03)----StreamingTableExec: partition_sizes=1, projection=[a, b, c], infinite_source=true, output_ordering=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, c@2 ASC NULLS LAST] query III @@ -2202,7 +2202,7 @@ logical_plan 03)----TableScan: annotated_data_infinite2 projection=[a, b, c] physical_plan 01)ProjectionExec: expr=[a@0 as a, b@1 as b, last_value(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]@2 as last_c] -02)--AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[last_value(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[last_value(annotated_data_infinite2.c) ORDER BY [annotated_data_infinite2.a DESC NULLS FIRST]], group_clustering_mode=Full 03)----StreamingTableExec: partition_sizes=1, projection=[a, b, c], infinite_source=true, output_ordering=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, c@2 ASC NULLS LAST] query III @@ -2229,7 +2229,7 @@ logical_plan 03)----TableScan: annotated_data_infinite2 projection=[a, b, c] physical_plan 01)ProjectionExec: expr=[a@0 as a, b@1 as b, last_value(annotated_data_infinite2.c)@2 as last_c] -02)--AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[last_value(annotated_data_infinite2.c)], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[last_value(annotated_data_infinite2.c)], group_clustering_mode=Full 03)----StreamingTableExec: partition_sizes=1, projection=[a, b, c], infinite_source=true, output_ordering=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, c@2 ASC NULLS LAST] query III @@ -2288,7 +2288,7 @@ logical_plan 01)Aggregate: groupBy=[[annotated_data_infinite2.a, annotated_data_infinite2.b]], aggr=[[array_agg(annotated_data_infinite2.d) ORDER BY [annotated_data_infinite2.d ASC NULLS LAST]]] 02)--TableScan: annotated_data_infinite2 projection=[a, b, d] physical_plan -01)AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[array_agg(annotated_data_infinite2.d) ORDER BY [annotated_data_infinite2.d ASC NULLS LAST]], ordering_mode=Sorted +01)AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[array_agg(annotated_data_infinite2.d) ORDER BY [annotated_data_infinite2.d ASC NULLS LAST]], group_clustering_mode=Full 02)--PartialSortExec: expr=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST, d@2 ASC NULLS LAST], common_prefix_length=[2] 03)----StreamingTableExec: partition_sizes=1, projection=[a, b, d], infinite_source=true, output_ordering=[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST] @@ -2612,7 +2612,7 @@ logical_plan 05)--------TableScan: sales_global projection=[country, amount] physical_plan 01)ProjectionExec: expr=[country@0 as country, array_agg(s.amount) ORDER BY [s.amount DESC NULLS FIRST]@1 as amounts, sum(s.amount)@2 as sum1] -02)--AggregateExec: mode=Single, gby=[country@0 as country], aggr=[array_agg(s.amount) ORDER BY [s.amount DESC NULLS FIRST], sum(s.amount)], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[country@0 as country], aggr=[array_agg(s.amount) ORDER BY [s.amount DESC NULLS FIRST], sum(s.amount)], group_clustering_mode=Full 03)----SortExec: TopK(fetch=10), expr=[country@0 ASC NULLS LAST, amount@1 DESC], preserve_partitioning=[false] 04)------DataSourceExec: partitions=1, partition_sizes=[1] @@ -2650,7 +2650,7 @@ logical_plan 05)--------TableScan: sales_global projection=[zip_code, country, amount] physical_plan 01)ProjectionExec: expr=[country@0 as country, zip_code@1 as zip_code, array_agg(s.amount) ORDER BY [s.amount DESC NULLS FIRST]@2 as amounts, sum(s.amount)@3 as sum1] -02)--AggregateExec: mode=Single, gby=[country@1 as country, zip_code@0 as zip_code], aggr=[array_agg(s.amount) ORDER BY [s.amount DESC NULLS FIRST], sum(s.amount)], ordering_mode=PartiallySorted([0]) +02)--AggregateExec: mode=Single, gby=[country@1 as country, zip_code@0 as zip_code], aggr=[array_agg(s.amount) ORDER BY [s.amount DESC NULLS FIRST], sum(s.amount)], group_clustering_mode=Partial([0]) 03)----SortExec: TopK(fetch=10), expr=[country@1 ASC NULLS LAST, amount@2 DESC], preserve_partitioning=[false] 04)------DataSourceExec: partitions=1, partition_sizes=[1] @@ -2687,7 +2687,7 @@ logical_plan 05)--------TableScan: sales_global projection=[country, amount] physical_plan 01)ProjectionExec: expr=[country@0 as country, array_agg(s.amount) ORDER BY [s.country DESC NULLS FIRST]@1 as amounts, sum(s.amount)@2 as sum1] -02)--AggregateExec: mode=Single, gby=[country@0 as country], aggr=[array_agg(s.amount) ORDER BY [s.country DESC NULLS FIRST], sum(s.amount)], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[country@0 as country], aggr=[array_agg(s.amount) ORDER BY [s.country DESC NULLS FIRST], sum(s.amount)], group_clustering_mode=Full 03)----SortExec: TopK(fetch=10), expr=[country@0 ASC NULLS LAST], preserve_partitioning=[false] 04)------DataSourceExec: partitions=1, partition_sizes=[1] @@ -2723,7 +2723,7 @@ logical_plan 05)--------TableScan: sales_global projection=[country, amount] physical_plan 01)ProjectionExec: expr=[country@0 as country, array_agg(s.amount) ORDER BY [s.country DESC NULLS FIRST, s.amount DESC NULLS FIRST]@1 as amounts, sum(s.amount)@2 as sum1] -02)--AggregateExec: mode=Single, gby=[country@0 as country], aggr=[array_agg(s.amount) ORDER BY [s.country DESC NULLS FIRST, s.amount DESC NULLS FIRST], sum(s.amount)], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[country@0 as country], aggr=[array_agg(s.amount) ORDER BY [s.country DESC NULLS FIRST, s.amount DESC NULLS FIRST], sum(s.amount)], group_clustering_mode=Full 03)----SortExec: TopK(fetch=10), expr=[country@0 ASC NULLS LAST, amount@1 DESC], preserve_partitioning=[false] 04)------DataSourceExec: partitions=1, partition_sizes=[1] @@ -4027,7 +4027,7 @@ logical_plan 12)------------------TableScan: multiple_ordered_table projection=[a, d] physical_plan 01)ProjectionExec: expr=[last_value(l.d) ORDER BY [l.a ASC NULLS LAST]@1 as amount_usd] -02)--AggregateExec: mode=Single, gby=[row_n@2 as row_n], aggr=[last_value(l.d) ORDER BY [l.a ASC NULLS LAST]], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[row_n@2 as row_n], aggr=[last_value(l.d) ORDER BY [l.a ASC NULLS LAST]], group_clustering_mode=Full 03)----HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(d@1, d@1)], filter=CAST(a@0 AS Int64) >= CAST(a@1 AS Int64) - 10, projection=[a@0, d@1, row_n@4] 04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, d], output_ordering=[a@0 ASC NULLS LAST], file_type=csv, has_header=true 05)------ProjectionExec: expr=[a@0 as a, d@1 as d, row_number() ORDER BY [r.a ASC NULLS LAST] RANGE BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW@2 as row_n] @@ -4067,9 +4067,9 @@ logical_plan 01)Aggregate: groupBy=[[multiple_ordered_table_with_pk.c, multiple_ordered_table_with_pk.b]], aggr=[[sum(CAST(multiple_ordered_table_with_pk.d AS Int64))]] 02)--TableScan: multiple_ordered_table_with_pk projection=[b, c, d] physical_plan -01)AggregateExec: mode=FinalPartitioned, gby=[c@0 as c, b@1 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=PartiallySorted([0]) +01)AggregateExec: mode=FinalPartitioned, gby=[c@0 as c, b@1 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Partial([0]) 02)--RepartitionExec: partitioning=Hash([c@0, b@1], 8), input_partitions=8, preserve_order=true, sort_exprs=c@0 ASC NULLS LAST -03)----AggregateExec: mode=Partial, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=PartiallySorted([0]) +03)----AggregateExec: mode=Partial, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Partial([0]) 04)------RepartitionExec: partitioning=RoundRobinBatch(8), input_partitions=1, maintains_sort_order=true 05)--------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[b, c, d], output_ordering=[c@1 ASC NULLS LAST], constraints=[PrimaryKey([3])], file_type=csv, has_header=true @@ -4106,9 +4106,9 @@ logical_plan 01)Aggregate: groupBy=[[multiple_ordered_table_with_pk.c, multiple_ordered_table_with_pk.b]], aggr=[[sum(CAST(multiple_ordered_table_with_pk.d AS Int64))]] 02)--TableScan: multiple_ordered_table_with_pk projection=[b, c, d] physical_plan -01)AggregateExec: mode=FinalPartitioned, gby=[c@0 as c, b@1 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=PartiallySorted([0]) +01)AggregateExec: mode=FinalPartitioned, gby=[c@0 as c, b@1 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Partial([0]) 02)--RepartitionExec: partitioning=Hash([c@0, b@1], 8), input_partitions=8, preserve_order=true, sort_exprs=c@0 ASC NULLS LAST -03)----AggregateExec: mode=Partial, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=PartiallySorted([0]) +03)----AggregateExec: mode=Partial, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Partial([0]) 04)------RepartitionExec: partitioning=RoundRobinBatch(8), input_partitions=1, maintains_sort_order=true 05)--------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[b, c, d], output_ordering=[c@1 ASC NULLS LAST], constraints=[PrimaryKey([3])], file_type=csv, has_header=true @@ -4129,9 +4129,9 @@ logical_plan 03)----Aggregate: groupBy=[[multiple_ordered_table_with_pk.c]], aggr=[[sum(CAST(multiple_ordered_table_with_pk.d AS Int64))]] 04)------TableScan: multiple_ordered_table_with_pk projection=[c, d] physical_plan -01)AggregateExec: mode=Single, gby=[c@0 as c, sum1@1 as sum1], aggr=[], ordering_mode=PartiallySorted([0]) +01)AggregateExec: mode=Single, gby=[c@0 as c, sum1@1 as sum1], aggr=[], group_clustering_mode=Partial([0]) 02)--ProjectionExec: expr=[c@0 as c, sum(multiple_ordered_table_with_pk.d)@1 as sum1] -03)----AggregateExec: mode=Single, gby=[c@0 as c], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=Sorted +03)----AggregateExec: mode=Single, gby=[c@0 as c], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Full 04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[c, d], output_ordering=[c@0 ASC NULLS LAST], constraints=[PrimaryKey([3])], file_type=csv, has_header=true query TT @@ -4151,7 +4151,7 @@ physical_plan 01)ProjectionExec: expr=[c@0 as c, sum1@2 as sum1, sum(multiple_ordered_table_with_pk.b) ROWS BETWEEN UNBOUNDED PRECEDING AND UNBOUNDED FOLLOWING@3 as sumb] 02)--WindowAggExec: wdw=[sum(multiple_ordered_table_with_pk.b) ROWS BETWEEN UNBOUNDED PRECEDING AND UNBOUNDED FOLLOWING: Ok(Field { name: "sum(multiple_ordered_table_with_pk.b) ROWS BETWEEN UNBOUNDED PRECEDING AND UNBOUNDED FOLLOWING", data_type: Int64, nullable: true }), frame: WindowFrame { units: Rows, start_bound: Preceding(UInt64(NULL)), end_bound: Following(UInt64(NULL)), is_causal: false }] 03)----ProjectionExec: expr=[c@0 as c, b@1 as b, sum(multiple_ordered_table_with_pk.d)@2 as sum1] -04)------AggregateExec: mode=Single, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=PartiallySorted([0]) +04)------AggregateExec: mode=Single, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Partial([0]) 05)--------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[b, c, d], output_ordering=[c@1 ASC NULLS LAST], constraints=[PrimaryKey([3])], file_type=csv, has_header=true query TT @@ -4180,10 +4180,10 @@ logical_plan physical_plan 01)HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(b@1, b@1)], projection=[c@0, c@3, sum1@2, sum1@5] 02)--ProjectionExec: expr=[c@0 as c, b@1 as b, sum(multiple_ordered_table_with_pk.d)@2 as sum1] -03)----AggregateExec: mode=Single, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=PartiallySorted([0]) +03)----AggregateExec: mode=Single, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Partial([0]) 04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[b, c, d], output_ordering=[c@1 ASC NULLS LAST], constraints=[PrimaryKey([3])], file_type=csv, has_header=true 05)--ProjectionExec: expr=[c@0 as c, b@1 as b, sum(multiple_ordered_table_with_pk.d)@2 as sum1] -06)----AggregateExec: mode=Single, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=PartiallySorted([0]) +06)----AggregateExec: mode=Single, gby=[c@1 as c, b@0 as b], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Partial([0]) 07)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[b, c, d], output_ordering=[c@1 ASC NULLS LAST], constraints=[PrimaryKey([3])], file_type=csv, has_header=true query TT @@ -4212,10 +4212,10 @@ physical_plan 01)ProjectionExec: expr=[c@0 as c, c@2 as c, sum1@1 as sum1, sum1@3 as sum1] 02)--CrossJoinExec 03)----ProjectionExec: expr=[c@0 as c, sum(multiple_ordered_table_with_pk.d)@1 as sum1] -04)------AggregateExec: mode=Single, gby=[c@0 as c], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=Sorted +04)------AggregateExec: mode=Single, gby=[c@0 as c], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Full 05)--------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[c, d], output_ordering=[c@0 ASC NULLS LAST], constraints=[PrimaryKey([3])], file_type=csv, has_header=true 06)----ProjectionExec: expr=[c@0 as c, sum(multiple_ordered_table_with_pk.d)@1 as sum1] -07)------AggregateExec: mode=Single, gby=[c@0 as c], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=Sorted +07)------AggregateExec: mode=Single, gby=[c@0 as c], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Full 08)--------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[c, d], output_ordering=[c@0 ASC NULLS LAST], constraints=[PrimaryKey([3])], file_type=csv, has_header=true # we do not generate physical plan for Repartition yet (e.g Distribute By queries). @@ -4254,10 +4254,10 @@ logical_plan physical_plan 01)UnionExec 02)--ProjectionExec: expr=[c@0 as c, a@1 as a, sum(multiple_ordered_table_with_pk.d)@2 as sum1] -03)----AggregateExec: mode=Single, gby=[c@1 as c, a@0 as a], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=Sorted +03)----AggregateExec: mode=Single, gby=[c@1 as c, a@0 as a], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Full 04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, c, d], output_orderings=[[a@0 ASC NULLS LAST], [c@1 ASC NULLS LAST]], constraints=[PrimaryKey([3])], file_type=csv, has_header=true 05)--ProjectionExec: expr=[c@0 as c, a@1 as a, sum(multiple_ordered_table_with_pk.d)@2 as sum1] -06)----AggregateExec: mode=Single, gby=[c@1 as c, a@0 as a], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=Sorted +06)----AggregateExec: mode=Single, gby=[c@1 as c, a@0 as a], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Full 07)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, c, d], output_orderings=[[a@0 ASC NULLS LAST], [c@1 ASC NULLS LAST]], constraints=[PrimaryKey([3])], file_type=csv, has_header=true # table scan should be simplified. @@ -4272,7 +4272,7 @@ logical_plan 03)----TableScan: multiple_ordered_table_with_pk projection=[a, c, d] physical_plan 01)ProjectionExec: expr=[c@0 as c, a@1 as a, sum(multiple_ordered_table_with_pk.d)@2 as sum1] -02)--AggregateExec: mode=Single, gby=[c@1 as c, a@0 as a], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[c@1 as c, a@0 as a], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Full 03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, c, d], output_orderings=[[a@0 ASC NULLS LAST], [c@1 ASC NULLS LAST]], constraints=[PrimaryKey([3])], file_type=csv, has_header=true # limit should be simplified @@ -4291,7 +4291,7 @@ logical_plan physical_plan 01)ProjectionExec: expr=[c@0 as c, a@1 as a, sum(multiple_ordered_table_with_pk.d)@2 as sum1] 02)--GlobalLimitExec: skip=0, fetch=5 -03)----AggregateExec: mode=Single, gby=[c@1 as c, a@0 as a], aggr=[sum(multiple_ordered_table_with_pk.d)], ordering_mode=Sorted +03)----AggregateExec: mode=Single, gby=[c@1 as c, a@0 as a], aggr=[sum(multiple_ordered_table_with_pk.d)], group_clustering_mode=Full 04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, c, d], output_orderings=[[a@0 ASC NULLS LAST], [c@1 ASC NULLS LAST]], constraints=[PrimaryKey([3])], file_type=csv, has_header=true statement ok @@ -4374,9 +4374,9 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [time_chunks@0 DESC], fetch=5 02)--ProjectionExec: expr=[date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)@0 as time_chunks] -03)----AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)@0 as date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)], aggr=[], ordering_mode=Sorted +03)----AggregateExec: mode=FinalPartitioned, gby=[date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)@0 as date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)], aggr=[], group_clustering_mode=Full 04)------RepartitionExec: partitioning=Hash([date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)@0], 8), input_partitions=8, preserve_order=true, sort_exprs=date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)@0 DESC -05)--------AggregateExec: mode=Partial, gby=[date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 900000000000 }, ts@0) as date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)], aggr=[], ordering_mode=Sorted +05)--------AggregateExec: mode=Partial, gby=[date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 900000000000 }, ts@0) as date_bin(Utf8("15 minutes"),unbounded_csv_with_timestamps.ts)], aggr=[], group_clustering_mode=Full 06)----------RepartitionExec: partitioning=RoundRobinBatch(8), input_partitions=1, maintains_sort_order=true 07)------------StreamingTableExec: partition_sizes=1, projection=[ts], infinite_source=true, output_ordering=[ts@0 DESC] @@ -5099,7 +5099,7 @@ logical_plan 02)--Aggregate: groupBy=[[multiple_ordered_table.a, multiple_ordered_table.b]], aggr=[[array_agg(multiple_ordered_table.c) ORDER BY [multiple_ordered_table.c DESC NULLS FIRST]]] 03)----TableScan: multiple_ordered_table projection=[a, b, c] physical_plan -01)AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[array_agg(multiple_ordered_table.c) ORDER BY [multiple_ordered_table.c DESC NULLS FIRST]], ordering_mode=Sorted +01)AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b], aggr=[array_agg(multiple_ordered_table.c) ORDER BY [multiple_ordered_table.c DESC NULLS FIRST]], group_clustering_mode=Full 02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, b, c], output_orderings=[[a@0 ASC NULLS LAST, b@1 ASC NULLS LAST], [c@2 ASC NULLS LAST]], file_type=csv, has_header=true query II? diff --git a/datafusion/sqllogictest/test_files/joins.slt b/datafusion/sqllogictest/test_files/joins.slt index f4dbd212eb8ee..5d2ba6025d42c 100644 --- a/datafusion/sqllogictest/test_files/joins.slt +++ b/datafusion/sqllogictest/test_files/joins.slt @@ -3510,7 +3510,7 @@ logical_plan 08)----------TableScan: annotated_data projection=[a, b] physical_plan 01)ProjectionExec: expr=[a@0 as a, last_value(r.b) ORDER BY [r.a ASC NULLS FIRST]@3 as last_col1] -02)--AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b, c@2 as c], aggr=[last_value(r.b) ORDER BY [r.a ASC NULLS FIRST]], ordering_mode=PartiallySorted([0]) +02)--AggregateExec: mode=Single, gby=[a@0 as a, b@1 as b, c@2 as c], aggr=[last_value(r.b) ORDER BY [r.a ASC NULLS FIRST]], group_clustering_mode=Partial([0]) 03)----HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(a@0, a@0)] 04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, b, c], output_ordering=[a@0 ASC, b@1 ASC NULLS LAST, c@2 ASC NULLS LAST], file_type=csv, has_header=true 05)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, b], output_ordering=[a@0 ASC, b@1 ASC NULLS LAST], file_type=csv, has_header=true @@ -3557,7 +3557,7 @@ logical_plan 12)------------------TableScan: multiple_ordered_table projection=[a, d] physical_plan 01)ProjectionExec: expr=[last_value(l.d) ORDER BY [l.a ASC NULLS LAST]@1 as amount_usd] -02)--AggregateExec: mode=Single, gby=[row_n@2 as row_n], aggr=[last_value(l.d) ORDER BY [l.a ASC NULLS LAST]], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[row_n@2 as row_n], aggr=[last_value(l.d) ORDER BY [l.a ASC NULLS LAST]], group_clustering_mode=Full 03)----HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(d@1, d@1)], filter=CAST(a@0 AS Int64) >= CAST(a@1 AS Int64) - 10, projection=[a@0, d@1, row_n@4] 04)------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, d], output_ordering=[a@0 ASC NULLS LAST], file_type=csv, has_header=true 05)------ProjectionExec: expr=[a@0 as a, d@1 as d, row_number() ORDER BY [r.a ASC NULLS LAST] RANGE BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW@2 as row_n] @@ -3592,9 +3592,9 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [a@0 ASC] 02)--ProjectionExec: expr=[a@0 as a, last_value(r.b) ORDER BY [r.a ASC NULLS FIRST]@3 as last_col1] -03)----AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b, c@2 as c], aggr=[last_value(r.b) ORDER BY [r.a ASC NULLS FIRST]], ordering_mode=PartiallySorted([0]) +03)----AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b, c@2 as c], aggr=[last_value(r.b) ORDER BY [r.a ASC NULLS FIRST]], group_clustering_mode=Partial([0]) 04)------RepartitionExec: partitioning=Hash([a@0, b@1, c@2], 2), input_partitions=2, preserve_order=true, sort_exprs=a@0 ASC -05)--------AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b, c@2 as c], aggr=[last_value(r.b) ORDER BY [r.a ASC NULLS FIRST]], ordering_mode=PartiallySorted([0]) +05)--------AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b, c@2 as c], aggr=[last_value(r.b) ORDER BY [r.a ASC NULLS FIRST]], group_clustering_mode=Partial([0]) 06)----------HashJoinExec: mode=Partitioned, join_type=Inner, on=[(a@0, a@0)] 07)------------RepartitionExec: partitioning=Hash([a@0], 2), input_partitions=1, maintains_sort_order=true 08)--------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/window_2.csv]]}, projection=[a, b, c], output_ordering=[a@0 ASC, b@1 ASC NULLS LAST, c@2 ASC NULLS LAST], file_type=csv, has_header=true diff --git a/datafusion/sqllogictest/test_files/order.slt b/datafusion/sqllogictest/test_files/order.slt index d974eb9dd06fc..bcd069fb03134 100644 --- a/datafusion/sqllogictest/test_files/order.slt +++ b/datafusion/sqllogictest/test_files/order.slt @@ -1898,7 +1898,7 @@ EXPLAIN SELECT c1, SUM(c2) as sum_c2 FROM table_with_ordered_not_null GROUP BY c ---- physical_plan 01)ProjectionExec: expr=[c1@0 as c1, sum(table_with_ordered_not_null.c2)@1 as sum_c2] -02)--AggregateExec: mode=Single, gby=[c1@0 as c1], aggr=[sum(table_with_ordered_not_null.c2)], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[c1@0 as c1], aggr=[sum(table_with_ordered_not_null.c2)], group_clustering_mode=Full 03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/aggregate_agg_multi_order.csv]]}, projection=[c1, c2], output_ordering=[c1@0 ASC NULLS LAST], file_type=csv, has_header=true statement ok diff --git a/datafusion/sqllogictest/test_files/ordered_aggregate_spill.slt b/datafusion/sqllogictest/test_files/ordered_aggregate_spill.slt index 548dfc69b1b9a..6e0c72e58ea12 100644 --- a/datafusion/sqllogictest/test_files/ordered_aggregate_spill.slt +++ b/datafusion/sqllogictest/test_files/ordered_aggregate_spill.slt @@ -55,9 +55,9 @@ FROM generate_series(20000) AS t1(v1) GROUP BY v1 ---- Plan with Metrics -01)AggregateExec: mode=FinalPartitioned,ordering_mode=Sorted, metrics=[spill_count=0,] +01)AggregateExec: mode=FinalPartitioned,group_clustering_mode=Full, metrics=[spill_count=0,] 02)--RepartitionExec:preserve_order=true -03)----AggregateExec: mode=Partial,ordering_mode=Sorted, metrics=[spill_count=0,] +03)----AggregateExec: mode=Partial,group_clustering_mode=Full, metrics=[spill_count=0,] query II rowsort @@ -100,9 +100,9 @@ FROM generate_series(20000) AS t1(v1) GROUP BY round(v1, -4), v1 % 5000 ---- Plan with Metrics -01)AggregateExec: mode=FinalPartitioned,aggr=[sum(t1.v1 * Int64(2))], ordering_mode=PartiallySorted([0]), metrics=[spill_count=0,] +01)AggregateExec: mode=FinalPartitioned,aggr=[sum(t1.v1 * Int64(2))], group_clustering_mode=Partial([0]), metrics=[spill_count=0,] 02)--RepartitionExec:input_partitions=1, maintains_sort_order=true -03)----AggregateExec: mode=Partial,ordering_mode=PartiallySorted([0]), metrics=[spill_count=0,] +03)----AggregateExec: mode=Partial,group_clustering_mode=Partial([0]), metrics=[spill_count=0,] # All rounds should have the same result hash @@ -124,9 +124,9 @@ FROM generate_series(20000) AS t1(v1) GROUP BY round(v1, -4), v1 % 5000 ---- Plan with Metrics -01)AggregateExec: mode=FinalPartitioned,aggr=[sum(t1.v1 * Int64(2))], ordering_mode=PartiallySorted([0]), metrics=[spilled_bytes= KB,] +01)AggregateExec: mode=FinalPartitioned,aggr=[sum(t1.v1 * Int64(2))], group_clustering_mode=Partial([0]), metrics=[spilled_bytes= KB,] 02)--RepartitionExec:input_partitions=1, maintains_sort_order=true -03)----AggregateExec: mode=Partial,ordering_mode=PartiallySorted([0]), metrics=[spill_count=0,] +03)----AggregateExec: mode=Partial,group_clustering_mode=Partial([0]), metrics=[spill_count=0,] # All rounds should have the same result hash @@ -148,9 +148,9 @@ FROM generate_series(20000) AS t1(v1) GROUP BY round(v1, -4), v1 % 5000 ---- Plan with Metrics -01)AggregateExec: mode=FinalPartitioned,aggr=[sum(t1.v1 * Int64(2))], ordering_mode=PartiallySorted([0]), metrics=[spilled_bytes= KB,] +01)AggregateExec: mode=FinalPartitioned,aggr=[sum(t1.v1 * Int64(2))], group_clustering_mode=Partial([0]), metrics=[spilled_bytes= KB,] 02)--RepartitionExec:input_partitions=1, maintains_sort_order=true -03)----AggregateExec: mode=Partial,ordering_mode=PartiallySorted([0]), metrics=[spill_count=0,] +03)----AggregateExec: mode=Partial,group_clustering_mode=Partial([0]), metrics=[spill_count=0,] # All rounds should have the same result hash @@ -176,9 +176,9 @@ FROM generate_series(20000) AS t1(v1) GROUP BY round(v1, -4), v1 % 5000 ---- Plan with Metrics -01)AggregateExec: mode=FinalPartitioned,aggr=[sum(t1.v1 * Int64(2)), min(t1.v1 % Int64(2))], ordering_mode=PartiallySorted([0]), metrics=[spilled_rows= K,] +01)AggregateExec: mode=FinalPartitioned,aggr=[sum(t1.v1 * Int64(2)), min(t1.v1 % Int64(2))], group_clustering_mode=Partial([0]), metrics=[spilled_rows= K,] 02)--RepartitionExec:input_partitions=1, maintains_sort_order=true -03)----AggregateExec: mode=Partial,aggr=[sum(t1.v1 * Int64(2)), min(t1.v1 % Int64(2))], ordering_mode=PartiallySorted([0]), metrics=[spill_count=0,] +03)----AggregateExec: mode=Partial,aggr=[sum(t1.v1 * Int64(2)), min(t1.v1 % Int64(2))], group_clustering_mode=Partial([0]), metrics=[spill_count=0,] # ================================================================================== @@ -201,7 +201,7 @@ FROM generate_series(20000) AS t1(v1) GROUP BY round(v1, -4), v1 % 5000 ---- Plan with Metrics -01)AggregateExec: mode=Single,aggr=[min(t1.v1 * Int64(2)), avg(t1.v1)], ordering_mode=PartiallySorted([0]), metrics=[spill_count=0,] +01)AggregateExec: mode=Single,aggr=[min(t1.v1 * Int64(2)), avg(t1.v1)], group_clustering_mode=Partial([0]), metrics=[spill_count=0,] query IIIR rowsort @@ -222,7 +222,7 @@ FROM generate_series(20000) AS t1(v1) GROUP BY round(v1, -4), v1 % 5000 ---- Plan with Metrics -01)AggregateExec: mode=Single,aggr=[min(t1.v1 * Int64(2)), avg(t1.v1)], ordering_mode=PartiallySorted([0]), metrics=[spilled_bytes= KB,] +01)AggregateExec: mode=Single,aggr=[min(t1.v1 * Int64(2)), avg(t1.v1)], group_clustering_mode=Partial([0]), metrics=[spilled_bytes= KB,] # Same result hash as the no-spill round above diff --git a/datafusion/sqllogictest/test_files/preserve_file_partitioning.slt b/datafusion/sqllogictest/test_files/preserve_file_partitioning.slt index e2dd22cc82bba..f4aab94ea627d 100644 --- a/datafusion/sqllogictest/test_files/preserve_file_partitioning.slt +++ b/datafusion/sqllogictest/test_files/preserve_file_partitioning.slt @@ -287,9 +287,9 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, count(Int64(1))@1 as count(*), avg(fact_table_ordered.value)@2 as avg(fact_table_ordered.value)] -03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted +03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], group_clustering_mode=Full 04)------RepartitionExec: partitioning=Hash([f_dkey@0], 3), input_partitions=3, preserve_order=true, sort_exprs=f_dkey@0 ASC NULLS LAST -05)--------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted +05)--------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], group_clustering_mode=Full 06)----------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=C/data.parquet]]}, projection=[value, f_dkey], output_ordering=[f_dkey@1 ASC NULLS LAST], file_type=parquet # Verify results without optimization @@ -319,7 +319,7 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, count(Int64(1))@1 as count(*), avg(fact_table_ordered.value)@2 as avg(fact_table_ordered.value)] -03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted +03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], group_clustering_mode=Full 04)------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=C/data.parquet]]}, projection=[value, f_dkey], output_ordering=[f_dkey@1 ASC NULLS LAST], output_partitioning=Hash([f_dkey@1], 3), file_type=parquet query TIR @@ -359,9 +359,9 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, max(d.env)@1 as max(d.env), max(d.service)@2 as max(d.service), count(Int64(1))@3 as count(*), sum(f.value)@4 as sum(f.value)] -03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[max(d.env), max(d.service), count(Int64(1)), sum(f.value)], ordering_mode=Sorted +03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[max(d.env), max(d.service), count(Int64(1)), sum(f.value)], group_clustering_mode=Full 04)------RepartitionExec: partitioning=Hash([f_dkey@0], 3), input_partitions=3, preserve_order=true, sort_exprs=f_dkey@0 ASC NULLS LAST -05)--------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[max(d.env), max(d.service), count(Int64(1)), sum(f.value)], ordering_mode=Sorted +05)--------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[max(d.env), max(d.service), count(Int64(1)), sum(f.value)], group_clustering_mode=Full 06)----------HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(d_dkey@0, f_dkey@1)], projection=[value@3, f_dkey@4, env@1, service@2] 07)------------CoalescePartitionsExec 08)--------------FilterExec: service@2 = log @@ -412,7 +412,7 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, max(d.env)@1 as max(d.env), max(d.service)@2 as max(d.service), count(Int64(1))@3 as count(*), sum(f.value)@4 as sum(f.value)] -03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[max(d.env), max(d.service), count(Int64(1)), sum(f.value)], ordering_mode=Sorted +03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[max(d.env), max(d.service), count(Int64(1)), sum(f.value)], group_clustering_mode=Full 04)------HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(d_dkey@0, f_dkey@1)], projection=[value@3, f_dkey@4, env@1, service@2] 05)--------CoalescePartitionsExec 06)----------FilterExec: service@2 = log diff --git a/datafusion/sqllogictest/test_files/range_sorted_time_bin_agg.slt b/datafusion/sqllogictest/test_files/range_sorted_time_bin_agg.slt index 18123a492dbd6..400bbab1a3592 100644 --- a/datafusion/sqllogictest/test_files/range_sorted_time_bin_agg.slt +++ b/datafusion/sqllogictest/test_files/range_sorted_time_bin_agg.slt @@ -35,9 +35,9 @@ # single streaming SinglePartitioned step with no hash shuffle. # # Today's plan still hash-repartitions: -# Partial AggregateExec (ordering_mode=Sorted) +# Partial AggregateExec (group_clustering_mode=Full) # -> RepartitionExec Hash([key, date_bin(...)]) -# -> FinalPartitioned AggregateExec (ordering_mode=Sorted) +# -> FinalPartitioned AggregateExec (group_clustering_mode=Full) statement ok set datafusion.explain.physical_plan_only = true; @@ -86,7 +86,7 @@ physical_plan DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion # overlap across the two 60-minute streams. # # Today this is still Partial + hash RepartitionExec + Final, even though -# ordering_mode=Sorted is already recognized. +# group_clustering_mode=Full is already recognized. ########## query TT @@ -97,9 +97,9 @@ GROUP BY key, time_bin; ---- physical_plan 01)ProjectionExec: expr=[key@0 as key, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1 as time_bin, sum(range_sorted_time_bin.value)@2 as sum(range_sorted_time_bin.value)] -02)--AggregateExec: mode=FinalPartitioned, gby=[key@0 as key, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)], aggr=[sum(range_sorted_time_bin.value)], ordering_mode=Sorted +02)--AggregateExec: mode=FinalPartitioned, gby=[key@0 as key, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)], aggr=[sum(range_sorted_time_bin.value)], group_clustering_mode=Full 03)----RepartitionExec: partitioning=Hash([key@0, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1], 2), input_partitions=2, preserve_order=true, sort_exprs=key@0 ASC, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1 ASC -04)------AggregateExec: mode=Partial, gby=[key@0 as key, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }, timestamp@1) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)], aggr=[sum(range_sorted_time_bin.value)], ordering_mode=Sorted +04)------AggregateExec: mode=Partial, gby=[key@0 as key, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }, timestamp@1) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)], aggr=[sum(range_sorted_time_bin.value)], group_clustering_mode=Full 05)--------FilterExec: col4@1 = a, projection=[key@0, timestamp@2, value@3] 06)----------DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch_range_partitioning/range_sorted_time_bin/part-0.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch_range_partitioning/range_sorted_time_bin/part-1.parquet]]}, projection=[key, col4, timestamp, value], output_ordering=[key@0 ASC, timestamp@2 ASC], output_partitioning=Range([timestamp@2 ASC], [(1704070800000000000)], 2), file_type=parquet, predicate=col4@4 = a, pruning_predicate=col4_null_count@2 != row_count@3 AND col4_min@0 <= a AND a <= col4_max@1, required_guarantees=[col4 in (a)] @@ -128,9 +128,9 @@ GROUP BY key, time_bin; ---- physical_plan 01)ProjectionExec: expr=[key@0 as key, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1 as time_bin, sum(range_sorted_time_bin.value)@2 as sum(range_sorted_time_bin.value)] -02)--AggregateExec: mode=FinalPartitioned, gby=[key@0 as key, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)], aggr=[sum(range_sorted_time_bin.value)], ordering_mode=Sorted +02)--AggregateExec: mode=FinalPartitioned, gby=[key@0 as key, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)], aggr=[sum(range_sorted_time_bin.value)], group_clustering_mode=Full 03)----RepartitionExec: partitioning=Hash([key@0, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1], 2), input_partitions=2, preserve_order=true, sort_exprs=key@0 ASC, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)@1 ASC -04)------AggregateExec: mode=Partial, gby=[key@0 as key, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }, timestamp@1) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)], aggr=[sum(range_sorted_time_bin.value)], ordering_mode=Sorted +04)------AggregateExec: mode=Partial, gby=[key@0 as key, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }, timestamp@1) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 60000000000 }"),range_sorted_time_bin.timestamp)], aggr=[sum(range_sorted_time_bin.value)], group_clustering_mode=Full 05)--------DataSourceExec: file_groups={2 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch_range_partitioning/range_sorted_time_bin/part-0.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch_range_partitioning/range_sorted_time_bin/part-1.parquet]]}, projection=[key, timestamp, value], output_ordering=[key@0 ASC, timestamp@1 ASC], output_partitioning=Range([timestamp@1 ASC], [(1704070800000000000)], 2), file_type=parquet query TPI diff --git a/datafusion/sqllogictest/test_files/repartition_subset_satisfaction.slt b/datafusion/sqllogictest/test_files/repartition_subset_satisfaction.slt index 5371ca59beea1..09bf131e5e95f 100644 --- a/datafusion/sqllogictest/test_files/repartition_subset_satisfaction.slt +++ b/datafusion/sqllogictest/test_files/repartition_subset_satisfaction.slt @@ -161,9 +161,9 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST, time_bin@1 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1 as time_bin, count(Int64(1))@2 as count(*), avg(fact_table_ordered.value)@3 as avg(fact_table_ordered.value)] -03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted +03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], group_clustering_mode=Full 04)------RepartitionExec: partitioning=Hash([f_dkey@0, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1], 3), input_partitions=3, preserve_order=true, sort_exprs=f_dkey@0 ASC NULLS LAST, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1 ASC NULLS LAST -05)--------AggregateExec: mode=Partial, gby=[f_dkey@2 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@0) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted +05)--------AggregateExec: mode=Partial, gby=[f_dkey@2 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@0) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], group_clustering_mode=Full 06)----------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=C/data.parquet]]}, projection=[timestamp, value, f_dkey], output_ordering=[f_dkey@2 ASC NULLS LAST, timestamp@0 ASC NULLS LAST], output_partitioning=Hash([f_dkey@2], 3), file_type=parquet # Verify results without subset satisfaction @@ -203,7 +203,7 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST, time_bin@1 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1 as time_bin, count(Int64(1))@2 as count(*), avg(fact_table_ordered.value)@3 as avg(fact_table_ordered.value)] -03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@2 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@0) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted +03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@2 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@0) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], group_clustering_mode=Full 04)------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=C/data.parquet]]}, projection=[timestamp, value, f_dkey], output_ordering=[f_dkey@2 ASC NULLS LAST, timestamp@0 ASC NULLS LAST], output_partitioning=Hash([f_dkey@2], 3), file_type=parquet # Verify results match with subset satisfaction @@ -373,9 +373,9 @@ physical_plan 05)--------RepartitionExec: partitioning=Hash([env@0, time_bin@1], 3), input_partitions=3 06)----------AggregateExec: mode=Partial, gby=[env@1 as env, time_bin@0 as time_bin], aggr=[avg(a.max_bin_value)] 07)------------ProjectionExec: expr=[date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp)@1 as time_bin, env@2 as env, max(j.value)@3 as max_bin_value] -08)--------------AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp), env@2 as env], aggr=[max(j.value)], ordering_mode=PartiallySorted([0, 1]) +08)--------------AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp), env@2 as env], aggr=[max(j.value)], group_clustering_mode=Partial([0, 1]) 09)----------------RepartitionExec: partitioning=Hash([f_dkey@0, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp)@1, env@2], 3), input_partitions=3, preserve_order=true, sort_exprs=f_dkey@0 ASC NULLS LAST, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp)@1 ASC NULLS LAST -10)------------------AggregateExec: mode=Partial, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@2) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp), env@1 as env], aggr=[max(j.value)], ordering_mode=PartiallySorted([0, 1]) +10)------------------AggregateExec: mode=Partial, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@2) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp), env@1 as env], aggr=[max(j.value)], group_clustering_mode=Partial([0, 1]) 11)--------------------HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(d_dkey@1, f_dkey@2)], projection=[f_dkey@4, env@0, timestamp@2, value@3] 12)----------------------CoalescePartitionsExec 13)------------------------FilterExec: service@1 = log, projection=[env@0, d_dkey@2] @@ -470,7 +470,7 @@ physical_plan 05)--------RepartitionExec: partitioning=Hash([env@0, time_bin@1], 3), input_partitions=3 06)----------AggregateExec: mode=Partial, gby=[env@1 as env, time_bin@0 as time_bin], aggr=[avg(a.max_bin_value)] 07)------------ProjectionExec: expr=[date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp)@1 as time_bin, env@2 as env, max(j.value)@3 as max_bin_value] -08)--------------AggregateExec: mode=SinglePartitioned, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@2) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp), env@1 as env], aggr=[max(j.value)], ordering_mode=PartiallySorted([0, 1]) +08)--------------AggregateExec: mode=SinglePartitioned, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@2) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),j.timestamp), env@1 as env], aggr=[max(j.value)], group_clustering_mode=Partial([0, 1]) 09)----------------HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(d_dkey@1, f_dkey@2)], projection=[f_dkey@4, env@0, timestamp@2, value@3] 10)------------------CoalescePartitionsExec 11)--------------------FilterExec: service@1 = log, projection=[env@0, d_dkey@2] diff --git a/datafusion/sqllogictest/test_files/sort_pushdown.slt b/datafusion/sqllogictest/test_files/sort_pushdown.slt index a173e76d6c262..da583f127c2b7 100644 --- a/datafusion/sqllogictest/test_files/sort_pushdown.slt +++ b/datafusion/sqllogictest/test_files/sort_pushdown.slt @@ -912,7 +912,7 @@ logical_plan 03)----TableScan: agg_expr_parquet projection=[x, y, v] physical_plan 01)SortExec: expr=[x@0 ASC NULLS LAST, agg_expr_parquet.y % Int64(2)@1 ASC NULLS LAST], preserve_partitioning=[false] -02)--AggregateExec: mode=Single, gby=[x@0 as x, CAST(y@1 AS Int64) % 2 as agg_expr_parquet.y % Int64(2)], aggr=[sum(agg_expr_parquet.v)], ordering_mode=PartiallySorted([0]) +02)--AggregateExec: mode=Single, gby=[x@0 as x, CAST(y@1 AS Int64) % 2 as agg_expr_parquet.y % Int64(2)], aggr=[sum(agg_expr_parquet.v)], group_clustering_mode=Partial([0]) 03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/sort_pushdown/agg_expr_sorted.parquet]]}, projection=[x, y, v], output_ordering=[x@0 ASC NULLS LAST, y@1 ASC NULLS LAST], file_type=parquet # Expected output pattern from ORDER BY [x, bucket]: @@ -946,7 +946,7 @@ logical_plan 02)--Aggregate: groupBy=[[agg_expr_parquet.x, CAST(agg_expr_parquet.y AS Int64)]], aggr=[[sum(CAST(agg_expr_parquet.v AS Int64))]] 03)----TableScan: agg_expr_parquet projection=[x, y, v] physical_plan -01)AggregateExec: mode=Single, gby=[x@0 as x, CAST(y@1 AS Int64) as agg_expr_parquet.y], aggr=[sum(agg_expr_parquet.v)], ordering_mode=Sorted +01)AggregateExec: mode=Single, gby=[x@0 as x, CAST(y@1 AS Int64) as agg_expr_parquet.y], aggr=[sum(agg_expr_parquet.v)], group_clustering_mode=Full 02)--DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/sort_pushdown/agg_expr_sorted.parquet]]}, projection=[x, y, v], output_ordering=[x@0 ASC NULLS LAST, y@1 ASC NULLS LAST], file_type=parquet query III @@ -978,7 +978,7 @@ logical_plan 03)----TableScan: agg_expr_parquet projection=[x, v] physical_plan 01)SortExec: expr=[sum(agg_expr_parquet.v)@1 ASC NULLS LAST], preserve_partitioning=[false] -02)--AggregateExec: mode=Single, gby=[x@0 as x], aggr=[sum(agg_expr_parquet.v)], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[x@0 as x], aggr=[sum(agg_expr_parquet.v)], group_clustering_mode=Full 03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/sort_pushdown/agg_expr_sorted.parquet]]}, projection=[x, v], output_ordering=[x@0 ASC NULLS LAST], file_type=parquet query II @@ -1034,7 +1034,7 @@ logical_plan 03)----TableScan: agg_expr_parquet projection=[x, v] physical_plan 01)SortExec: expr=[CAST(x@0 AS Int64) + 1 DESC], preserve_partitioning=[false] -02)--AggregateExec: mode=Single, gby=[x@0 as x], aggr=[sum(agg_expr_parquet.v)], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[x@0 as x], aggr=[sum(agg_expr_parquet.v)], group_clustering_mode=Full 03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/sort_pushdown/agg_expr_sorted.parquet]]}, projection=[x, v], output_ordering=[x@0 ASC NULLS LAST], file_type=parquet query II @@ -1060,7 +1060,7 @@ logical_plan 03)----TableScan: agg_expr_parquet projection=[x, v] physical_plan 01)SortExec: expr=[2 * CAST(x@0 AS Int64) ASC NULLS LAST], preserve_partitioning=[false] -02)--AggregateExec: mode=Single, gby=[x@0 as x], aggr=[sum(agg_expr_parquet.v)], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[x@0 as x], aggr=[sum(agg_expr_parquet.v)], group_clustering_mode=Full 03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/sort_pushdown/agg_expr_sorted.parquet]]}, projection=[x, v], output_ordering=[x@0 ASC NULLS LAST], file_type=parquet query II diff --git a/datafusion/sqllogictest/test_files/unnest.slt b/datafusion/sqllogictest/test_files/unnest.slt index 8e6013328f98f..39ed64a526755 100644 --- a/datafusion/sqllogictest/test_files/unnest.slt +++ b/datafusion/sqllogictest/test_files/unnest.slt @@ -988,9 +988,9 @@ logical_plan 08)--------------TableScan: range() projection=[value] physical_plan 01)ProjectionExec: expr=[array_agg(unnested.ar)@1 as array_agg(unnested.ar)] -02)--AggregateExec: mode=FinalPartitioned, gby=[generated_id@0 as generated_id], aggr=[array_agg(unnested.ar)], ordering_mode=Sorted +02)--AggregateExec: mode=FinalPartitioned, gby=[generated_id@0 as generated_id], aggr=[array_agg(unnested.ar)], group_clustering_mode=Full 03)----RepartitionExec: partitioning=Hash([generated_id@0], 4), input_partitions=4, preserve_order=true, sort_exprs=generated_id@0 ASC NULLS LAST -04)------AggregateExec: mode=Partial, gby=[generated_id@0 as generated_id], aggr=[array_agg(unnested.ar)], ordering_mode=Sorted +04)------AggregateExec: mode=Partial, gby=[generated_id@0 as generated_id], aggr=[array_agg(unnested.ar)], group_clustering_mode=Full 05)--------ProjectionExec: expr=[generated_id@0 as generated_id, __unnest_placeholder(make_array(range().value),depth=1)@1 as ar] 06)----------UnnestExec 07)------------ProjectionExec: expr=[row_number() ROWS BETWEEN UNBOUNDED PRECEDING AND UNBOUNDED FOLLOWING@1 as generated_id, make_array(value@0) as __unnest_placeholder(make_array(range().value))] diff --git a/datafusion/sqllogictest/test_files/window.slt b/datafusion/sqllogictest/test_files/window.slt index a6b64b98de0e3..254fdf20e5124 100644 --- a/datafusion/sqllogictest/test_files/window.slt +++ b/datafusion/sqllogictest/test_files/window.slt @@ -357,7 +357,7 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [b@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[b@0 as b, max(d.a)@1 as max_a, max(d.seq)@2 as max(d.seq)] -03)----AggregateExec: mode=SinglePartitioned, gby=[b@2 as b], aggr=[max(d.a), max(d.seq)], ordering_mode=Sorted +03)----AggregateExec: mode=SinglePartitioned, gby=[b@2 as b], aggr=[max(d.a), max(d.seq)], group_clustering_mode=Full 04)------ProjectionExec: expr=[row_number() PARTITION BY [s.b] ORDER BY [s.a ASC NULLS LAST] RANGE BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW@2 as seq, a@0 as a, b@1 as b] 05)--------BoundedWindowAggExec: wdw=[row_number() PARTITION BY [s.b] ORDER BY [s.a ASC NULLS LAST] RANGE BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW: Field { "row_number() PARTITION BY [s.b] ORDER BY [s.a ASC NULLS LAST] RANGE BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW": UInt64 }, frame: RANGE BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW], mode=[Sorted] 06)----------SortExec: expr=[b@1 ASC NULLS LAST, a@0 ASC NULLS LAST], preserve_partitioning=[true] diff --git a/docs/source/library-user-guide/upgrading/56.0.0.md b/docs/source/library-user-guide/upgrading/56.0.0.md index 1ea0f4671009d..be1a3007d94c1 100644 --- a/docs/source/library-user-guide/upgrading/56.0.0.md +++ b/docs/source/library-user-guide/upgrading/56.0.0.md @@ -75,6 +75,27 @@ The Minimum Supported Rust Version (MSRV) has been updated to [`1.95.0`]. [`1.95.0`]: https://releases.rs/docs/1.95.0/ +### Aggregate group-clustering APIs + +`AggregateExec::input_order_mode()` has been replaced by +`AggregateExec::group_clustering_mode()`. It returns a `GroupClusteringMode`: +`None`, `Partial(indices)`, or `Full`, corresponding to the previous aggregate +modes `Linear`, `PartiallySorted(indices)`, and `Sorted`. +`AggregateExec::compute_properties` now accepts `&GroupClusteringMode` in place of +`&InputOrderMode`. + +`GroupClusteringMode` describes the input's group-clustering guarantees, which +determine when groups can be emitted during aggregation. + +In `datafusion_physical_plan::aggregates::order`, `GroupOrdering`, +`GroupOrderingPartial`, and `GroupOrderingFull` have been renamed to +`GroupClustering`, `GroupClusteringPartial`, and `GroupClusteringFull`. +`GroupClustering::try_new` accepts `&GroupClusteringMode`. + +Aggregate plans now display `group_clustering_mode=Full` or +`group_clustering_mode=Partial(indices)` instead of `ordering_mode=Sorted` or +`ordering_mode=PartiallySorted(indices)`. + ### Upgrade arrow/parquet to 60.0.0 and object_store to 0.14.2 DataFusion 56.0.0 uses `arrow` and `parquet` 60.0.0, and `object_store` 0.14.2.