Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 4 additions & 4 deletions datafusion/core/tests/dataframe/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -3422,9 +3422,9 @@ async fn union_with_mix_of_presorted_and_explicitly_resorted_inputs_with_reparti
assert_snapshot!(
union_with_mix_of_presorted_and_explicitly_resorted_inputs_impl(false).await?,
@r"
AggregateExec: mode=Final, gby=[id@0 as id], aggr=[], ordering_mode=Sorted
AggregateExec: mode=Final, gby=[id@0 as id], aggr=[], group_clustering_mode=Full
SortPreservingMergeExec: [id@0 ASC NULLS LAST]
AggregateExec: mode=Partial, gby=[id@0 as id], aggr=[], ordering_mode=Sorted
AggregateExec: mode=Partial, gby=[id@0 as id], aggr=[], group_clustering_mode=Full
UnionExec
DataSourceExec: file_groups={1 group: [[{testdata}/alltypes_tiny_pages.parquet]]}, projection=[id], output_ordering=[id@0 ASC NULLS LAST], file_type=parquet
SortExec: expr=[id@0 ASC NULLS LAST], preserve_partitioning=[false]
Expand All @@ -3440,9 +3440,9 @@ async fn union_with_mix_of_presorted_and_explicitly_resorted_inputs_with_reparti
assert_snapshot!(
union_with_mix_of_presorted_and_explicitly_resorted_inputs_impl(true).await?,
@r"
AggregateExec: mode=Final, gby=[id@0 as id], aggr=[], ordering_mode=Sorted
AggregateExec: mode=Final, gby=[id@0 as id], aggr=[], group_clustering_mode=Full
SortPreservingMergeExec: [id@0 ASC NULLS LAST]
AggregateExec: mode=Partial, gby=[id@0 as id], aggr=[], ordering_mode=Sorted
AggregateExec: mode=Partial, gby=[id@0 as id], aggr=[], group_clustering_mode=Full
UnionExec
DataSourceExec: file_groups={1 group: [[{testdata}/alltypes_tiny_pages.parquet]]}, projection=[id], output_ordering=[id@0 ASC NULLS LAST], file_type=parquet
SortExec: expr=[id@0 ASC NULLS LAST], preserve_partitioning=[false]
Expand Down
19 changes: 11 additions & 8 deletions datafusion/core/tests/fuzz_cases/aggregate_fuzz.rs
Original file line number Diff line number Diff line change
Expand Up @@ -41,15 +41,14 @@ use datafusion_common_runtime::JoinSet;
use datafusion_functions_aggregate::sum::sum_udaf;
use datafusion_physical_expr::PhysicalSortExpr;
use datafusion_physical_expr::expressions::{Column, col, lit};
use datafusion_physical_plan::InputOrderMode;
use test_utils::{StringBatchGenerator, add_empty_batches};

use datafusion_execution::TaskContext;
use datafusion_execution::memory_pool::FairSpillPool;
use datafusion_execution::runtime_env::RuntimeEnvBuilder;
use datafusion_physical_expr::aggregate::AggregateExprBuilder;
use datafusion_physical_plan::aggregates::{
AggregateExec, AggregateMode, PhysicalGroupBy,
AggregateExec, AggregateMode, GroupClusteringMode, PhysicalGroupBy,
};
use datafusion_physical_plan::metrics::MetricValue;
use datafusion_physical_plan::{ExecutionPlan, collect, displayable};
Expand Down Expand Up @@ -303,7 +302,7 @@ async fn streaming_aggregate_test() {
/// two `AggregateExec` variants produce the same result: the pipeline breaking
/// one over unordered input (`PartialHashAggregateStream`) and the
/// non-pipeline breaking one over ordered input
/// (`OrderedPartialAggregateStream`).
/// (`ClusteredPartialAggregateStream`).
async fn run_aggregate_test(input1: Vec<RecordBatch>, group_by_columns: Vec<&str>) {
let schema = input1[0].schema();
let session_config = SessionConfig::new().with_batch_size(50);
Expand Down Expand Up @@ -354,8 +353,8 @@ async fn run_aggregate_test(input1: Vec<RecordBatch>, group_by_columns: Vec<&str
.unwrap(),
);
assert_ne!(
aggregate_exec_running.input_order_mode(),
&InputOrderMode::Linear,
aggregate_exec_running.group_clustering_mode(),
&GroupClusteringMode::None,
"running aggregate should observe ordered input for group_by: {group_by:?}"
);

Expand Down Expand Up @@ -555,13 +554,17 @@ async fn verify_ordered_aggregate(frame: &DataFrame, expected_sort: bool) {

fn f_down(&mut self, node: &'n Self::Node) -> Result<TreeNodeRecursion> {
if let Some(exec) = node.downcast_ref::<AggregateExec>() {
assert_eq!(
exec.properties().output_ordering().is_some(),
self.expected_sort
);
if self.expected_sort {
assert!(matches!(
exec.input_order_mode(),
InputOrderMode::PartiallySorted(_) | InputOrderMode::Sorted
exec.group_clustering_mode(),
GroupClusteringMode::Partial(_) | GroupClusteringMode::Full
));
} else {
assert_eq!(*exec.input_order_mode(), InputOrderMode::Linear);
assert_eq!(*exec.group_clustering_mode(), GroupClusteringMode::None);
}
}
Ok(TreeNodeRecursion::Continue)
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -93,9 +93,9 @@ pub struct QueryBuilder {
/// ...
/// ```
///
/// More details can see [`GroupOrdering`].
/// More details can see [`GroupClustering`].
///
/// [`GroupOrdering`]: datafusion_physical_plan::aggregates::order::GroupOrdering
/// [`GroupClustering`]: datafusion_physical_plan::aggregates::order::GroupClustering
dataset_sort_keys: Vec<Vec<String>>,

/// If we will also test the no grouping case like:
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -4867,9 +4867,9 @@ fn preserve_ordering_for_streaming_sorted_aggregate() -> Result<()> {

let plan_distrib = test_config.to_plan(physical_plan.clone(), &DISTRIB_DISTRIB_SORT);
assert_plan!(plan_distrib, @r"
AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[COUNT(b)], ordering_mode=Sorted
AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[COUNT(b)], group_clustering_mode=Full
RepartitionExec: partitioning=Hash([a@0], 2), input_partitions=2, preserve_order=true, sort_exprs=a@0 ASC
AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[COUNT(b)], ordering_mode=Sorted
AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[COUNT(b)], group_clustering_mode=Full
DataSourceExec: file_groups={2 groups: [[x], [y]]}, projection=[a, b, c, d, e], output_ordering=[a@0 ASC], file_type=parquet
");

Expand Down Expand Up @@ -4901,9 +4901,9 @@ fn preserve_ordering_for_streaming_partially_sorted_aggregate() -> Result<()> {

let plan_distrib = test_config.to_plan(physical_plan.clone(), &DISTRIB_DISTRIB_SORT);
assert_plan!(plan_distrib, @r"
AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b], aggr=[COUNT(c)], ordering_mode=PartiallySorted([0])
AggregateExec: mode=FinalPartitioned, gby=[a@0 as a, b@1 as b], aggr=[COUNT(c)], group_clustering_mode=Partial([0])
RepartitionExec: partitioning=Hash([a@0, b@1], 2), input_partitions=2, preserve_order=true, sort_exprs=a@0 ASC
AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b], aggr=[COUNT(c)], ordering_mode=PartiallySorted([0])
AggregateExec: mode=Partial, gby=[a@0 as a, b@1 as b], aggr=[COUNT(c)], group_clustering_mode=Partial([0])
DataSourceExec: file_groups={2 groups: [[x], [y]]}, projection=[a, b, c, d, e], output_ordering=[a@0 ASC], file_type=parquet
");

Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -520,7 +520,7 @@ fn test_has_order_by() -> Result<()> {
actual,
@r"
LocalLimitExec: fetch=10
AggregateExec: mode=Single, gby=[a@0 as a], aggr=[], ordering_mode=Sorted
AggregateExec: mode=Single, gby=[a@0 as a], aggr=[], group_clustering_mode=Full
DataSourceExec: file_groups={1 group: [[x]]}, projection=[a, b, c, d, e], output_ordering=[a@0 ASC], file_type=parquet
"
);
Expand Down
4 changes: 2 additions & 2 deletions datafusion/core/tests/physical_optimizer/pushdown_sort.rs
Original file line number Diff line number Diff line change
Expand Up @@ -731,13 +731,13 @@ fn test_pushdown_through_blocking_node() {
OptimizationTest:
input:
- SortExec: expr=[a@0 ASC], preserve_partitioning=[false]
- AggregateExec: mode=Final, gby=[a@0 as a], aggr=[COUNT(b)], ordering_mode=Sorted
- AggregateExec: mode=Final, gby=[a@0 as a], aggr=[COUNT(b)], group_clustering_mode=Full
- SortExec: expr=[a@0 DESC NULLS LAST], preserve_partitioning=[false]
- DataSourceExec: file_groups={1 group: [[x]]}, projection=[a, b, c, d, e], output_ordering=[a@0 ASC], file_type=parquet
output:
Ok:
- SortExec: expr=[a@0 ASC], preserve_partitioning=[false]
- AggregateExec: mode=Final, gby=[a@0 as a], aggr=[COUNT(b)], ordering_mode=Sorted
- AggregateExec: mode=Final, gby=[a@0 as a], aggr=[COUNT(b)], group_clustering_mode=Full
- SortExec: expr=[a@0 DESC NULLS LAST], preserve_partitioning=[false]
- DataSourceExec: file_groups={1 group: [[x]]}, projection=[a, b, c, d, e], file_type=parquet, sort_order_for_reorder=[a@0 DESC NULLS LAST], reverse_row_groups=true
"
Expand Down
14 changes: 7 additions & 7 deletions datafusion/physical-plan/benches/dictionary_group_values.rs
Original file line number Diff line number Diff line change
Expand Up @@ -29,7 +29,7 @@ use criterion::{
};
use datafusion_expr::EmitTo;
use datafusion_physical_plan::aggregates::group_values::new_group_values;
use datafusion_physical_plan::aggregates::order::{GroupOrdering, GroupOrderingFull};
use datafusion_physical_plan::aggregates::order::{GroupClustering, GroupClusteringFull};
use rand::rngs::StdRng;
use rand::seq::SliceRandom;
use rand::{Rng, SeedableRng};
Expand Down Expand Up @@ -106,7 +106,7 @@ fn bench_intern_emit(c: &mut Criterion) {
b.iter_batched_ref(
|| {
(
new_group_values(schema.clone(), &GroupOrdering::None)
new_group_values(schema.clone(), &GroupClustering::None)
.unwrap(),
Vec::<usize>::with_capacity(size),
)
Expand Down Expand Up @@ -151,7 +151,7 @@ fn bench_repeated_intern_emit(c: &mut Criterion) {
b.iter_batched_ref(
|| {
(
new_group_values(schema.clone(), &GroupOrdering::None)
new_group_values(schema.clone(), &GroupClustering::None)
.unwrap(),
Vec::<usize>::with_capacity(size),
)
Expand All @@ -172,7 +172,7 @@ fn bench_repeated_intern_emit(c: &mut Criterion) {
group.finish();
}

// GroupOrdering::Full -> GroupValuesColumn::<true>: scalar append_val/equal_to path.
// GroupClustering::Full -> GroupValuesColumn::<true>: scalar append_val/equal_to path.
fn bench_scalar_append_equal(c: &mut Criterion) {
let mut group = c.benchmark_group("dict_scalar_append_equal");
let schema = dict_schema();
Expand All @@ -192,7 +192,7 @@ fn bench_scalar_append_equal(c: &mut Criterion) {
(
new_group_values(
schema.clone(),
&GroupOrdering::Full(GroupOrderingFull::new()),
&GroupClustering::Full(GroupClusteringFull::new()),
)
.unwrap(),
Vec::<usize>::with_capacity(size),
Expand Down Expand Up @@ -227,7 +227,7 @@ fn bench_take_n(c: &mut Criterion) {
b.iter_batched_ref(
|| {
(
new_group_values(schema.clone(), &GroupOrdering::None).unwrap(),
new_group_values(schema.clone(), &GroupClustering::None).unwrap(),
Vec::<usize>::with_capacity(size),
)
},
Expand Down Expand Up @@ -298,7 +298,7 @@ fn bench_shared_values_arc(c: &mut Criterion) {
b.iter_batched_ref(
|| {
(
new_group_values(schema.clone(), &GroupOrdering::None)
new_group_values(schema.clone(), &GroupClustering::None)
.unwrap(),
Vec::<usize>::with_capacity(size),
)
Expand Down
22 changes: 12 additions & 10 deletions datafusion/physical-plan/benches/ordered_group_values.rs
Original file line number Diff line number Diff line change
Expand Up @@ -38,12 +38,12 @@ use datafusion_physical_expr::expressions::col;
use datafusion_physical_expr::{LexOrdering, PhysicalSortExpr};
use datafusion_physical_plan::aggregates::group_values::multi_group_by::GroupValuesColumn;
use datafusion_physical_plan::aggregates::group_values::{GroupValues, new_group_values};
use datafusion_physical_plan::aggregates::order::GroupOrdering;
use datafusion_physical_plan::aggregates::order::GroupClustering;
use datafusion_physical_plan::aggregates::{
AggregateExec, AggregateMode, PhysicalGroupBy,
AggregateExec, AggregateMode, GroupClusteringMode, PhysicalGroupBy,
};
use datafusion_physical_plan::test::TestMemoryExec;
use datafusion_physical_plan::{ExecutionPlan, InputOrderMode, collect};
use datafusion_physical_plan::{ExecutionPlan, collect};
use tokio::runtime::Runtime;

const ROWS: usize = 131_072;
Expand Down Expand Up @@ -112,8 +112,10 @@ fn grouping(c: &mut Criterion) {
let values: Box<dyn GroupValues> = if selected {
new_group_values(
Arc::clone(&schema),
&GroupOrdering::try_new(&InputOrderMode::Sorted)
.unwrap(),
&GroupClustering::try_new(
&GroupClusteringMode::Full,
)
.unwrap(),
)
.unwrap()
} else {
Expand Down Expand Up @@ -158,7 +160,7 @@ fn check_case(schema: &SchemaRef, batches: &[Vec<ArrayRef>]) -> (usize, usize) {
let mut hashed = GroupValuesColumn::<true>::try_new(Arc::clone(schema)).unwrap();
let mut selected = new_group_values(
Arc::clone(schema),
&GroupOrdering::try_new(&InputOrderMode::Sorted).unwrap(),
&GroupClustering::try_new(&GroupClusteringMode::Full).unwrap(),
)
.unwrap();
let mut expected = Vec::new();
Expand Down Expand Up @@ -188,7 +190,7 @@ fn aggregate_plan(
schema: &SchemaRef,
keys: Vec<Vec<ArrayRef>>,
sort_columns: &[&str],
input_order_mode: &InputOrderMode,
group_clustering_mode: &GroupClusteringMode,
) -> Arc<dyn ExecutionPlan> {
let mut fields = schema.fields().to_vec();
fields.push(Arc::new(Field::new("v", DataType::Int64, false)));
Expand Down Expand Up @@ -232,7 +234,7 @@ fn aggregate_plan(
schema,
)
.unwrap();
assert_eq!(plan.input_order_mode(), input_order_mode);
assert_eq!(plan.group_clustering_mode(), group_clustering_mode);
Arc::new(plan)
}

Expand All @@ -246,7 +248,7 @@ fn aggregation(c: &mut Criterion) {
for run_length in [1, 8, 128, 8192] {
let (schema, keys) = inputs(run_length, 8192, strings);
let plan =
aggregate_plan(&schema, keys, &["a", "b"], &InputOrderMode::Sorted);
aggregate_plan(&schema, keys, &["a", "b"], &GroupClusteringMode::Full);
let name =
format!("{}_run{run_length}", if strings { "string" } else { "int" });
group.bench_function(name, |b| {
Expand Down Expand Up @@ -291,7 +293,7 @@ fn partially_ordered_aggregation(c: &mut Criterion) {
&schema,
keys,
&["a"],
&InputOrderMode::PartiallySorted(vec![0]),
&GroupClusteringMode::Partial(vec![0]),
);
let runtime = Runtime::new().unwrap();
let mut group = c.benchmark_group("partially_ordered_aggregate_exec");
Expand Down
16 changes: 8 additions & 8 deletions datafusion/physical-plan/benches/partial_ordering.rs
Original file line number Diff line number Diff line change
Expand Up @@ -18,7 +18,7 @@
use std::sync::Arc;

use arrow::array::{ArrayRef, Int32Array};
use datafusion_physical_plan::aggregates::order::GroupOrderingPartial;
use datafusion_physical_plan::aggregates::order::GroupClusteringPartial;

use criterion::{Criterion, criterion_group, criterion_main};

Expand All @@ -34,20 +34,20 @@ fn create_test_arrays(num_columns: usize) -> Vec<ArrayRef> {
.collect()
}
fn bench_new_groups(c: &mut Criterion) {
let mut group = c.benchmark_group("group_ordering_partial");
let mut group = c.benchmark_group("group_clustering_partial");

// Test with 1, 2, 4, and 8 order indices
// Test with 1, 2, 4, and 8 grouping indices
for num_columns in [1, 2, 4, 8] {
let order_indices: Vec<usize> = (0..num_columns).collect();
let grouping_indices: Vec<usize> = (0..num_columns).collect();

group.bench_function(format!("order_indices_{num_columns}"), |b| {
group.bench_function(format!("grouping_indices_{num_columns}"), |b| {
let batch_group_values = create_test_arrays(num_columns);
let group_indices: Vec<usize> = (0..BATCH_SIZE).collect();

b.iter(|| {
let mut ordering =
GroupOrderingPartial::try_new(order_indices.clone()).unwrap();
ordering
let mut clustering =
GroupClusteringPartial::try_new(grouping_indices.clone()).unwrap();
clustering
.new_groups(&batch_group_values, &group_indices, BATCH_SIZE)
.unwrap();
});
Expand Down
Loading