From 41316e857293bc6a074868d1ab806a7b7e03ac67 Mon Sep 17 00:00:00 2001 From: wudidapaopao <664920313@qq.com> Date: Tue, 29 Sep 2026 04:24:57 +0800 Subject: [PATCH 1/6] feat: optimize count over non-null arguments --- datafusion/core/src/physical_planner.rs | 4 +- datafusion/core/tests/dataframe/mod.rs | 114 +++++------ .../tests/datasource/object_store_access.rs | 4 +- datafusion/core/tests/sql/explain_analyze.rs | 6 +- datafusion/core/tests/sql/unparser.rs | 5 +- datafusion/expr-common/src/accumulator.rs | 13 ++ .../expr-common/src/groups_accumulator.rs | 14 ++ .../src/type_coercion/aggregates.rs | 8 + datafusion/functions-aggregate/src/count.rs | 188 ++++++++++++++++-- .../simplify_expressions/expr_simplifier.rs | 48 ++++- .../simplify_expressions/simplify_exprs.rs | 2 +- .../src/single_distinct_to_groupby.rs | 66 +++++- .../optimizer/tests/optimizer_integration.rs | 8 +- datafusion/physical-expr/src/aggregate.rs | 6 +- .../aggregates/aggregate_hash_table/common.rs | 15 +- .../src/aggregates/aggregate_stream.rs | 11 +- .../src/aggregates/grouped_hash_stream.rs | 8 +- datafusion/physical-plan/src/windows/mod.rs | 31 ++- datafusion/sql/src/unparser/expr.rs | 6 + .../sqllogictest/test_files/aggregate.slt | 30 +-- .../test_files/aggregate_memory_spill.slt | 2 +- .../test_files/aggregate_repartition.slt | 16 +- .../test_files/array/array_has.slt | 36 ++-- datafusion/sqllogictest/test_files/avro.slt | 6 +- .../sqllogictest/test_files/clickbench.slt | 168 ++++++++-------- .../test_files/count_star_rule.slt | 14 +- .../dynamic_filter_pushdown_config.slt | 6 +- .../test_files/explain_analyze.slt | 2 +- .../sqllogictest/test_files/explain_tree.slt | 168 ++++++++-------- .../test_files/functional_dependencies.slt | 6 +- .../sqllogictest/test_files/group_by.slt | 21 +- datafusion/sqllogictest/test_files/joins.slt | 20 +- datafusion/sqllogictest/test_files/json.slt | 6 +- .../sqllogictest/test_files/lateral_join.slt | 6 +- datafusion/sqllogictest/test_files/limit.slt | 12 +- .../test_files/nested_loop_join_spill.slt | 8 +- .../optimizer_group_by_constant.slt | 8 +- .../piecewise_merge_join_batches.slt | 12 +- .../test_files/preserve_file_partitioning.slt | 44 ++-- .../test_files/projection_pushdown.slt | 4 +- .../push_down_filter_regression.slt | 6 +- .../repartition_subset_satisfaction.slt | 10 +- datafusion/sqllogictest/test_files/select.slt | 6 +- .../test_files/single_distinct_to_groupby.slt | 14 +- .../sqllogictest/test_files/subquery.slt | 34 ++-- .../test_files/tpch/plans/q1.slt.part | 6 +- .../test_files/tpch/plans/q13.slt.part | 6 +- .../test_files/tpch/plans/q21.slt.part | 6 +- .../test_files/tpch/plans/q22.slt.part | 6 +- .../test_files/tpch/plans/q4.slt.part | 6 +- datafusion/sqllogictest/test_files/union.slt | 14 +- datafusion/sqllogictest/test_files/window.slt | 8 +- .../tests/cases/roundtrip_logical_plan.rs | 24 +-- 53 files changed, 818 insertions(+), 480 deletions(-) diff --git a/datafusion/core/src/physical_planner.rs b/datafusion/core/src/physical_planner.rs index c88c34645b6de..ce17dd902b5c7 100644 --- a/datafusion/core/src/physical_planner.rs +++ b/datafusion/core/src/physical_planner.rs @@ -4800,7 +4800,7 @@ mod tests { assert_contains!( aggregate_explain(&logical_plan).await?, - "aggr=[count(1) as count(*)]" + "aggr=[count() as count(*)]" ); Ok(()) @@ -4815,7 +4815,7 @@ mod tests { assert_contains!( aggregate_explain(&logical_plan).await?, - "aggr=[count(1) as total_rows]" + "aggr=[count() as total_rows]" ); Ok(()) diff --git a/datafusion/core/tests/dataframe/mod.rs b/datafusion/core/tests/dataframe/mod.rs index 43ceee3444ced..6328c2bf8eb14 100644 --- a/datafusion/core/tests/dataframe/mod.rs +++ b/datafusion/core/tests/dataframe/mod.rs @@ -3160,42 +3160,42 @@ async fn test_count_wildcard_on_sort() -> Result<()> { assert_snapshot!( pretty_format_batches(&sql_results).unwrap(), @r" - +---------------+-------------------------------------------------------------------------------------+ - | plan_type | plan | - +---------------+-------------------------------------------------------------------------------------+ - | logical_plan | Sort: count(*) ASC NULLS LAST | - | | Projection: t1.b, count(Int64(1)) AS count(*) | - | | Aggregate: groupBy=[[t1.b]], aggr=[[count(Int64(1))]] | - | | TableScan: t1 projection=[b] | - | physical_plan | SortPreservingMergeExec: [count(*)@1 ASC NULLS LAST] | - | | ProjectionExec: expr=[b@0 as b, count(Int64(1))@1 as count(*)] | - | | SortExec: expr=[count(Int64(1))@1 ASC NULLS LAST], preserve_partitioning=[true] | - | | AggregateExec: mode=FinalPartitioned, gby=[b@0 as b], aggr=[count(Int64(1))] | - | | RepartitionExec: partitioning=Hash([b@0], 4), input_partitions=1 | - | | AggregateExec: mode=Partial, gby=[b@0 as b], aggr=[count(Int64(1))] | - | | DataSourceExec: partitions=1, partition_sizes=[1] | - | | | - +---------------+-------------------------------------------------------------------------------------+ + +---------------+-----------------------------------------------------------------------------------------------+ + | plan_type | plan | + +---------------+-----------------------------------------------------------------------------------------------+ + | logical_plan | Sort: count(*) ASC NULLS LAST | + | | Projection: t1.b, count(Int64(1)) AS count(*) | + | | Aggregate: groupBy=[[t1.b]], aggr=[[count() AS count(Int64(1))]] | + | | TableScan: t1 projection=[b] | + | physical_plan | SortPreservingMergeExec: [count(*)@1 ASC NULLS LAST] | + | | ProjectionExec: expr=[b@0 as b, count(Int64(1))@1 as count(*)] | + | | SortExec: expr=[count(Int64(1))@1 ASC NULLS LAST], preserve_partitioning=[true] | + | | AggregateExec: mode=FinalPartitioned, gby=[b@0 as b], aggr=[count() as count(Int64(1))] | + | | RepartitionExec: partitioning=Hash([b@0], 4), input_partitions=1 | + | | AggregateExec: mode=Partial, gby=[b@0 as b], aggr=[count() as count(Int64(1))] | + | | DataSourceExec: partitions=1, partition_sizes=[1] | + | | | + +---------------+-----------------------------------------------------------------------------------------------+ " ); assert_snapshot!( pretty_format_batches(&df_results).unwrap(), @r" - +---------------+---------------------------------------------------------------------------------------+ - | plan_type | plan | - +---------------+---------------------------------------------------------------------------------------+ - | logical_plan | Sort: count(*) AS count(*) ASC NULLS LAST | - | | Aggregate: groupBy=[[t1.b]], aggr=[[count(Int64(1)) AS count(*)]] | - | | TableScan: t1 projection=[b] | - | physical_plan | SortPreservingMergeExec: [count(*)@1 ASC NULLS LAST] | - | | SortExec: expr=[count(*)@1 ASC NULLS LAST], preserve_partitioning=[true] | - | | AggregateExec: mode=FinalPartitioned, gby=[b@0 as b], aggr=[count(1) as count(*)] | - | | RepartitionExec: partitioning=Hash([b@0], 4), input_partitions=1 | - | | AggregateExec: mode=Partial, gby=[b@0 as b], aggr=[count(1) as count(*)] | - | | DataSourceExec: partitions=1, partition_sizes=[1] | - | | | - +---------------+---------------------------------------------------------------------------------------+ + +---------------+--------------------------------------------------------------------------------------+ + | plan_type | plan | + +---------------+--------------------------------------------------------------------------------------+ + | logical_plan | Sort: count(*) AS count(*) ASC NULLS LAST | + | | Aggregate: groupBy=[[t1.b]], aggr=[[count() AS count(*)]] | + | | TableScan: t1 projection=[b] | + | physical_plan | SortPreservingMergeExec: [count(*)@1 ASC NULLS LAST] | + | | SortExec: expr=[count(*)@1 ASC NULLS LAST], preserve_partitioning=[true] | + | | AggregateExec: mode=FinalPartitioned, gby=[b@0 as b], aggr=[count() as count(*)] | + | | RepartitionExec: partitioning=Hash([b@0], 4), input_partitions=1 | + | | AggregateExec: mode=Partial, gby=[b@0 as b], aggr=[count() as count(*)] | + | | DataSourceExec: partitions=1, partition_sizes=[1] | + | | | + +---------------+--------------------------------------------------------------------------------------+ " ); Ok(()) @@ -3221,7 +3221,7 @@ async fn test_count_wildcard_on_where_in() -> Result<()> { | | TableScan: t1 projection=[a, b] | | | SubqueryAlias: __correlated_sq_1 | | | Projection: count(Int64(1)) AS count(*) | - | | Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] | + | | Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] | | | TableScan: t2 projection=[] | | physical_plan | HashJoinExec: mode=CollectLeft, join_type=RightSemi, on=[(count(*)@0, CAST(t1.a AS Int64)@2)], projection=[a@0, b@1] | | | ProjectionExec: expr=[4 as count(*)] | @@ -3265,7 +3265,7 @@ async fn test_count_wildcard_on_where_in() -> Result<()> { | logical_plan | LeftSemi Join: CAST(t1.a AS Int64) = __correlated_sq_1.count(*) | | | TableScan: t1 projection=[a, b] | | | SubqueryAlias: __correlated_sq_1 | - | | Aggregate: groupBy=[[]], aggr=[[count(Int64(1)) AS count(*)]] | + | | Aggregate: groupBy=[[]], aggr=[[count() AS count(*)]] | | | TableScan: t2 projection=[] | | physical_plan | HashJoinExec: mode=CollectLeft, join_type=RightSemi, on=[(count(*)@0, CAST(t1.a AS Int64)@2)], projection=[a@0, b@1] | | | ProjectionExec: expr=[4 as count(*)] | @@ -3541,16 +3541,16 @@ async fn test_count_wildcard_on_aggregate() -> Result<()> { assert_snapshot!( pretty_format_batches(&sql_results).unwrap(), @r" - +---------------+-----------------------------------------------------+ - | plan_type | plan | - +---------------+-----------------------------------------------------+ - | logical_plan | Projection: count(Int64(1)) AS count(*) | - | | Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] | - | | TableScan: t1 projection=[] | - | physical_plan | ProjectionExec: expr=[4 as count(*)] | - | | PlaceholderRowExec | - | | | - +---------------+-----------------------------------------------------+ + +---------------+----------------------------------------------------------------+ + | plan_type | plan | + +---------------+----------------------------------------------------------------+ + | logical_plan | Projection: count(Int64(1)) AS count(*) | + | | Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] | + | | TableScan: t1 projection=[] | + | physical_plan | ProjectionExec: expr=[4 as count(*)] | + | | PlaceholderRowExec | + | | | + +---------------+----------------------------------------------------------------+ " ); @@ -3567,15 +3567,15 @@ async fn test_count_wildcard_on_aggregate() -> Result<()> { assert_snapshot!( pretty_format_batches(&df_results).unwrap(), @r" - +---------------+---------------------------------------------------------------+ - | plan_type | plan | - +---------------+---------------------------------------------------------------+ - | logical_plan | Aggregate: groupBy=[[]], aggr=[[count(Int64(1)) AS count(*)]] | - | | TableScan: t1 projection=[] | - | physical_plan | ProjectionExec: expr=[4 as count(*)] | - | | PlaceholderRowExec | - | | | - +---------------+---------------------------------------------------------------+ + +---------------+-------------------------------------------------------+ + | plan_type | plan | + +---------------+-------------------------------------------------------+ + | logical_plan | Aggregate: groupBy=[[]], aggr=[[count() AS count(*)]] | + | | TableScan: t1 projection=[] | + | physical_plan | ProjectionExec: expr=[4 as count(*)] | + | | PlaceholderRowExec | + | | | + +---------------+-------------------------------------------------------+ " ); @@ -3606,16 +3606,16 @@ async fn test_count_wildcard_on_where_scalar_subquery() -> Result<()> { | | TableScan: t1 projection=[a, b] | | | SubqueryAlias: __scalar_sq_1 | | | Projection: count(Int64(1)) AS count(*), t2.a, Boolean(true) AS __always_true | - | | Aggregate: groupBy=[[t2.a]], aggr=[[count(Int64(1))]] | + | | Aggregate: groupBy=[[t2.a]], aggr=[[count() AS count(Int64(1))]] | | | TableScan: t2 projection=[a] | | physical_plan | FilterExec: CASE WHEN __always_true@3 IS NULL THEN 0 ELSE count(*)@2 END > 0, projection=[a@0, b@1] | | | RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 | | | HashJoinExec: mode=CollectLeft, join_type=Right, on=[(a@1, a@0)], projection=[a@3, b@4, count(*)@0, __always_true@2] | | | CoalescePartitionsExec | | | ProjectionExec: expr=[count(Int64(1))@1 as count(*), a@0 as a, true as __always_true] | - | | AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[count(Int64(1))] | + | | AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[count() as count(Int64(1))] | | | RepartitionExec: partitioning=Hash([a@0], 4), input_partitions=1 | - | | AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[count(Int64(1))] | + | | AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[count() as count(Int64(1))] | | | DataSourceExec: partitions=1, partition_sizes=[1] | | | DataSourceExec: partitions=1, partition_sizes=[1] | | | | @@ -3661,16 +3661,16 @@ async fn test_count_wildcard_on_where_scalar_subquery() -> Result<()> { | | TableScan: t1 projection=[a, b] | | | SubqueryAlias: __scalar_sq_1 | | | Projection: count(*), t2.a, Boolean(true) AS __always_true | - | | Aggregate: groupBy=[[t2.a]], aggr=[[count(Int64(1)) AS count(*)]] | + | | Aggregate: groupBy=[[t2.a]], aggr=[[count() AS count(*)]] | | | TableScan: t2 projection=[a] | | physical_plan | FilterExec: CASE WHEN __always_true@3 IS NULL THEN 0 ELSE count(*)@2 END > 0, projection=[a@0, b@1] | | | RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 | | | HashJoinExec: mode=CollectLeft, join_type=Right, on=[(a@1, a@0)], projection=[a@3, b@4, count(*)@0, __always_true@2] | | | CoalescePartitionsExec | | | ProjectionExec: expr=[count(*)@1 as count(*), a@0 as a, true as __always_true] | - | | AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[count(1) as count(*)] | + | | AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[count() as count(*)] | | | RepartitionExec: partitioning=Hash([a@0], 4), input_partitions=1 | - | | AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[count(1) as count(*)] | + | | AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[count() as count(*)] | | | DataSourceExec: partitions=1, partition_sizes=[1] | | | DataSourceExec: partitions=1, partition_sizes=[1] | | | | diff --git a/datafusion/core/tests/datasource/object_store_access.rs b/datafusion/core/tests/datasource/object_store_access.rs index 16d894bde1303..9d51834f28719 100644 --- a/datafusion/core/tests/datasource/object_store_access.rs +++ b/datafusion/core/tests/datasource/object_store_access.rs @@ -866,8 +866,8 @@ async fn query_single_parquet_file() { RequestCountingObjectStore() Total Requests: 3 - GET (opts) path=parquet_table.parquet head=true - - GET (ranges) path=parquet_table.parquet ranges=4-534,534-1064 - - GET (ranges) path=parquet_table.parquet ranges=1064-1594,1594-2124 + - GET (ranges) path=parquet_table.parquet ranges=4-534 + - GET (ranges) path=parquet_table.parquet ranges=1064-1594 " ); } diff --git a/datafusion/core/tests/sql/explain_analyze.rs b/datafusion/core/tests/sql/explain_analyze.rs index 67614cfbdea9a..e1dc9badb9c9f 100644 --- a/datafusion/core/tests/sql/explain_analyze.rs +++ b/datafusion/core/tests/sql/explain_analyze.rs @@ -1137,7 +1137,7 @@ async fn explain_analyze_aggregate_metrics_map_indices_to_expressions() { .to_string(); assert_contains!( normal.as_str(), - "aggr=[sum(aggregate_test_100.c5), sum(aggregate_test_100.c6), count(aggregate_test_100.c7)]" + "aggr=[sum(aggregate_test_100.c5), sum(aggregate_test_100.c6), count() as count(aggregate_test_100.c7)]" ); assert_contains!(normal.as_str(), "agg_expr_0_arguments_time"); assert_contains!(normal.as_str(), "agg_expr_1_arguments_time"); @@ -1159,7 +1159,7 @@ async fn explain_analyze_aggregate_metrics_map_indices_to_expressions() { ); assert_contains!( verbose.as_str(), - "agg_expr_2_arguments_time{partition=0, aggregate=count(aggregate_test_100.c7)}" + "agg_expr_2_arguments_time{partition=0, aggregate=count() as count(aggregate_test_100.c7)}" ); } @@ -1178,7 +1178,7 @@ async fn explain_logical_plan_only() { @r#" logical_plan Projection: count(Int64(1)) AS count(*) - Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] + Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] SubqueryAlias: t Projection: Values: (Utf8("a"), Int64(1), Int64(100)), (Utf8("a"), Int64(2), Int64(150)) diff --git a/datafusion/core/tests/sql/unparser.rs b/datafusion/core/tests/sql/unparser.rs index b77f53cb46932..46c81f133dee0 100644 --- a/datafusion/core/tests/sql/unparser.rs +++ b/datafusion/core/tests/sql/unparser.rs @@ -567,9 +567,8 @@ async fn optimized_duckdb_unparse_preserves_nested_aggregate_scope() -> Result<( assert!( sql.contains(concat!( - r#"FROM (SELECT sum("total_revenue") AS "alias2", "#, - r#"date_part('year', "signup_date") AS "group_alias_0", "#, - r#""customer_id" AS "alias1" "# + r#"FROM (SELECT date_part('year', "signup_date") AS "group_alias_0", "#, + r#"sum("total_revenue") AS "alias2" "# )), "inner aggregate should define the aliases before the outer aggregate uses them: {sql}", ); diff --git a/datafusion/expr-common/src/accumulator.rs b/datafusion/expr-common/src/accumulator.rs index 95cee870e2f0b..660d32699ffe7 100644 --- a/datafusion/expr-common/src/accumulator.rs +++ b/datafusion/expr-common/src/accumulator.rs @@ -138,6 +138,19 @@ pub trait Accumulator: Send + Sync + Debug + std::any::Any { /// running sum. fn update_batch(&mut self, values: &[ArrayRef]) -> Result<()>; + /// Like [`Self::update_batch`], but also receives the input row count. + /// + /// Aggregates without arguments cannot derive this count from an empty + /// `values` slice and must override this method. The default delegates to + /// [`Self::update_batch`]. + fn update_batch_with_num_rows( + &mut self, + values: &[ArrayRef], + _num_rows: usize, + ) -> Result<()> { + self.update_batch(values) + } + /// Returns an optional metric timed once per grouped adapter input batch. /// /// A grouped accumulator adapter uses this for aggregate-owned work it diff --git a/datafusion/expr-common/src/groups_accumulator.rs b/datafusion/expr-common/src/groups_accumulator.rs index 1c004cb70b931..7c2e34d5519e4 100644 --- a/datafusion/expr-common/src/groups_accumulator.rs +++ b/datafusion/expr-common/src/groups_accumulator.rs @@ -373,6 +373,20 @@ pub trait GroupsAccumulator: Send + std::any::Any { opt_filter: Option<&BooleanArray>, ) -> Result>; + /// Like [`Self::convert_to_state`], but also receives the input row count. + /// + /// Aggregates without arguments cannot derive this count from an empty + /// `values` slice and must override this method. The default delegates to + /// [`Self::convert_to_state`]. + fn convert_to_state_with_num_rows( + &self, + values: &[ArrayRef], + opt_filter: Option<&BooleanArray>, + _num_rows: usize, + ) -> Result> { + self.convert_to_state(values, opt_filter) + } + /// Amount of memory used to store the state of this accumulator, /// in bytes. /// diff --git a/datafusion/expr-common/src/type_coercion/aggregates.rs b/datafusion/expr-common/src/type_coercion/aggregates.rs index ada0bd26b8d06..623e34bdda0b3 100644 --- a/datafusion/expr-common/src/type_coercion/aggregates.rs +++ b/datafusion/expr-common/src/type_coercion/aggregates.rs @@ -75,6 +75,14 @@ pub fn check_arg_count( ); } } + TypeSignature::Nullary => { + if !input_fields.is_empty() { + return plan_err!( + "The function {func_name} expects 0 arguments, but {} were provided", + input_fields.len() + ); + } + } TypeSignature::OneOf(variants) => { let ok = variants .iter() diff --git a/datafusion/functions-aggregate/src/count.rs b/datafusion/functions-aggregate/src/count.rs index ed2abe88644ae..83abb5f3c0c90 100644 --- a/datafusion/functions-aggregate/src/count.rs +++ b/datafusion/functions-aggregate/src/count.rs @@ -39,7 +39,8 @@ use datafusion_expr::{ GroupsAccumulator, ReversedUDAF, SetMonotonicity, Signature, StatisticsArgs, TypeSignature, Volatility, WindowFunctionDefinition, expr::WindowFunction, - function::{AccumulatorArgs, StateFieldsArgs}, + function::{AccumulatorArgs, AggregateFunctionSimplification, StateFieldsArgs}, + simplify::SimplifyContext, utils::{AggregateOrderSensitivity, format_state_name}, }; use datafusion_functions_aggregate_common::aggregate::count_distinct::PrimitiveDistinctCountGroupsAccumulator; @@ -362,12 +363,12 @@ impl AggregateUDFImpl for Count { arg_types: &[DataType], is_distinct: bool, ) -> Option { + if !is_distinct { + return Some(arg_types.len() <= 1); + } if arg_types.len() != 1 { return Some(false); } - if !is_distinct { - return Some(true); - } // Keep in step with `create_distinct_count_groups_accumulator`. Some(matches!( arg_types[0], @@ -400,17 +401,34 @@ impl AggregateUDFImpl for Count { AggregateOrderSensitivity::Insensitive } + fn simplify(&self) -> Option { + Some(Box::new(|mut aggregate_function, info| { + let params = &aggregate_function.params; + // Every row is counted when none of the arguments can be null + if !params.distinct + && !params.args.is_empty() + && params + .args + .iter() + .all(|arg| is_safe_non_null_count_arg(arg, info)) + { + aggregate_function.params.args.clear(); + } + Ok(Expr::AggregateFunction(aggregate_function)) + })) + } + fn default_value(&self, _data_type: &DataType) -> Result { Ok(ScalarValue::Int64(Some(0))) } fn value_from_stats(&self, statistics_args: &StatisticsArgs) -> Option { - let [expr] = statistics_args.exprs else { - return None; - }; let col_stats = &statistics_args.statistics.column_statistics; if statistics_args.is_distinct { + let [expr] = statistics_args.exprs else { + return None; + }; // Only column references can be resolved from statistics; // expressions like casts or literals are not supported. let col_expr = expr.downcast_ref::()?; @@ -425,6 +443,15 @@ impl AggregateUDFImpl for Count { return None; }; + if statistics_args.exprs.is_empty() { + let num_rows = i64::try_from(num_rows).ok()?; + return Some(ScalarValue::Int64(Some(num_rows))); + } + + let [expr] = statistics_args.exprs else { + return None; + }; + // TODO optimize with exprs other than Column if let Some(col_expr) = expr.downcast_ref::() { if let Precision::Exact(val) = col_stats[col_expr.index()].null_count { @@ -466,6 +493,16 @@ impl AggregateUDFImpl for Count { } } +/// Returns true if `expr` is a non-null literal or non-nullable column that is +/// safe to elide from `COUNT`. +fn is_safe_non_null_count_arg(expr: &Expr, info: &SimplifyContext) -> bool { + match expr { + Expr::Literal(value, _) => !value.is_null(), + Expr::Column(_) => matches!(info.nullable(expr), Ok(false)), + _ => false, + } +} + #[cold] fn create_distinct_count_groups_accumulator( args: &AccumulatorArgs, @@ -614,6 +651,19 @@ impl Accumulator for CountAccumulator { Ok(()) } + fn update_batch_with_num_rows( + &mut self, + values: &[ArrayRef], + num_rows: usize, + ) -> Result<()> { + if values.is_empty() { + self.count += num_rows as i64; + Ok(()) + } else { + self.update_batch(values) + } + } + fn retract_batch(&mut self, values: &[ArrayRef]) -> Result<()> { let array = &values[0]; self.count -= (array.len() - null_count_for_multiple_cols(values)) as i64; @@ -673,15 +723,18 @@ impl GroupsAccumulator for CountGroupsAccumulator { opt_filter: Option<&BooleanArray>, total_num_groups: usize, ) -> Result<()> { - assert_eq!(values.len(), 1, "single argument to update_batch"); - let values = &values[0]; + assert!( + values.len() <= 1, + "COUNT expects zero or one argument to update_batch" + ); + let logical_nulls = values.first().and_then(|values| values.logical_nulls()); // Add one to each group's counter for each non null, non // filtered value self.counts.resize(total_num_groups, 0); accumulate_indices( group_indices, - values.logical_nulls().as_ref(), + logical_nulls.as_ref(), opt_filter, |group_index| { // SAFETY: group_index is guaranteed to be in bounds @@ -769,12 +822,35 @@ impl GroupsAccumulator for CountGroupsAccumulator { values: &[ArrayRef], opt_filter: Option<&BooleanArray>, ) -> Result> { - let values = &values[0]; + let Some(values) = values.first() else { + return internal_err!( + "Nullary COUNT requires convert_to_state_with_num_rows" + ); + }; + self.convert_to_state_with_num_rows( + std::slice::from_ref(values), + opt_filter, + values.len(), + ) + } - let state_array = match (values.logical_nulls(), opt_filter) { + fn convert_to_state_with_num_rows( + &self, + values: &[ArrayRef], + opt_filter: Option<&BooleanArray>, + num_rows: usize, + ) -> Result> { + if values.len() > 1 { + return internal_err!( + "COUNT expects zero or one argument to convert_to_state" + ); + } + let logical_nulls = values.first().and_then(|values| values.logical_nulls()); + + let state_array = match (logical_nulls, opt_filter) { (None, None) => { // In case there is no nulls in input and no filter, returning array of 1 - Arc::new(Int64Array::from_value(1, values.len())) + Arc::new(Int64Array::from_value(1, num_rows)) } (Some(nulls), None) => { // If there are any nulls in input values -- casting `nulls` (true for values, false for nulls) @@ -958,10 +1034,14 @@ mod tests { use super::*; use arrow::{ - array::{DictionaryArray, Int32Array, Int64Array, NullArray, StringArray}, + array::{ + BooleanArray, DictionaryArray, Int32Array, Int64Array, NullArray, StringArray, + }, datatypes::{DataType, Field, Int32Type, Schema}, }; + use datafusion_common::DFSchema; use datafusion_expr::function::AccumulatorArgs; + use datafusion_expr::{col, lit}; use datafusion_physical_expr::{PhysicalExpr, expressions::Column}; use std::sync::Arc; /// Helper function to create a dictionary array with non-null keys but some null values @@ -1003,6 +1083,86 @@ mod tests { Ok(()) } + #[test] + fn count_accumulator_nullary() -> Result<()> { + let mut accumulator = CountAccumulator::new(); + accumulator.update_batch_with_num_rows(&[], 10)?; + assert_eq!(accumulator.evaluate()?, ScalarValue::Int64(Some(10))); + Ok(()) + } + + #[test] + fn count_groups_accumulator_nullary() -> Result<()> { + let mut accumulator = CountGroupsAccumulator::new(); + accumulator.update_batch(&[], &[0, 1, 0, 2], None, 3)?; + + let result = accumulator.evaluate(EmitTo::All)?; + let expected = Int64Array::from(vec![2, 1, 1]); + assert_eq!(result.as_primitive::(), &expected); + + let state = accumulator.convert_to_state_with_num_rows(&[], None, 3)?; + let expected = Int64Array::from(vec![1, 1, 1]); + assert_eq!(state[0].as_primitive::(), &expected); + + let filter = BooleanArray::from(vec![Some(true), None, Some(false), Some(true)]); + let mut filtered_accumulator = CountGroupsAccumulator::new(); + filtered_accumulator.update_batch(&[], &[0, 1, 0, 2], Some(&filter), 3)?; + let result = filtered_accumulator.evaluate(EmitTo::All)?; + let expected = Int64Array::from(vec![1, 0, 1]); + assert_eq!(result.as_primitive::(), &expected); + + let state = accumulator.convert_to_state_with_num_rows(&[], Some(&filter), 4)?; + let expected = Int64Array::from(vec![1, 0, 0, 1]); + assert_eq!(state[0].as_primitive::(), &expected); + Ok(()) + } + + #[test] + fn simplify_count_safe_non_null_args() -> Result<()> { + let schema = DFSchema::try_from(Schema::new(vec![ + Field::new("a", DataType::Int32, false), + Field::new("b", DataType::Int32, true), + ]))?; + let info = SimplifyContext::builder() + .with_schema(Arc::new(schema)) + .build(); + let simplify = Count::new().simplify().unwrap(); + let simplified_args = |args: Vec, distinct: bool| -> Result> { + let aggregate_function = datafusion_expr::expr::AggregateFunction::new_udf( + count_udaf(), + args, + distinct, + None, + vec![], + None, + ); + match simplify(aggregate_function, &info)? { + Expr::AggregateFunction(f) => Ok(f.params.args), + other => internal_err!("unexpected expression {other}"), + } + }; + + for args in [ + vec![lit(1i64)], + vec![lit("x")], + vec![col("a")], + vec![col("a"), lit(2)], + ] { + assert!(simplified_args(args, false)?.is_empty()); + } + for args in [ + vec![lit(ScalarValue::Null)], + vec![col("b")], + vec![col("a"), col("b")], + vec![col("a") + lit(1)], + ] { + assert_eq!(simplified_args(args.clone(), false)?, args); + } + assert!(simplified_args(vec![], false)?.is_empty()); + assert_eq!(simplified_args(vec![lit(1i64)], true)?, vec![lit(1i64)]); + Ok(()) + } + #[test] fn count_groups_preserving_reads() -> Result<()> { let mut accumulator = CountGroupsAccumulator::new(); diff --git a/datafusion/optimizer/src/simplify_expressions/expr_simplifier.rs b/datafusion/optimizer/src/simplify_expressions/expr_simplifier.rs index 506692f0dfa6d..b6e219e9426b7 100644 --- a/datafusion/optimizer/src/simplify_expressions/expr_simplifier.rs +++ b/datafusion/optimizer/src/simplify_expressions/expr_simplifier.rs @@ -1679,7 +1679,13 @@ impl TreeNodeRewriter for Simplifier<'_> { .. }) => match (func.simplify(), expr) { (Some(simplify_function), Expr::AggregateFunction(af)) => { - Transformed::yes(simplify_function(af, info)?) + let original = Expr::AggregateFunction(af.clone()); + let simplified = simplify_function(af, info)?; + if simplified == original { + Transformed::no(simplified) + } else { + Transformed::yes(simplified) + } } (_, expr) => Transformed::no(expr), }, @@ -5628,23 +5634,47 @@ mod tests { let expected = aggregate_function_expr.clone(); assert_eq!(simplify(aggregate_function_expr), expected); + + let udaf = AggregateUDF::new_from_impl(SimplifyMockUdaf::new_with_noop()); + let aggregate_function_expr = + Expr::AggregateFunction(expr::AggregateFunction::new_udf( + udaf.into(), + vec![], + false, + None, + vec![], + None, + )); + + let expected = aggregate_function_expr.clone(); + let (actual, cycles) = simplify_with_cycle_count(aggregate_function_expr); + assert_eq!(actual, expected); + assert_eq!(cycles, 1); } /// A Mock UDAF which defines `simplify` to be used in tests /// related to UDAF simplification #[derive(Debug, Clone, PartialEq, Eq, Hash)] struct SimplifyMockUdaf { - simplify: bool, + simplify: Option, } impl SimplifyMockUdaf { /// make simplify method return new expression fn new_with_simplify() -> Self { - Self { simplify: true } + Self { + simplify: Some(true), + } } /// make simplify method return no change fn new_without_simplify() -> Self { - Self { simplify: false } + Self { simplify: None } + } + /// make simplify method return the original expression + fn new_with_noop() -> Self { + Self { + simplify: Some(false), + } } } @@ -5680,10 +5710,12 @@ mod tests { } fn simplify(&self) -> Option { - if self.simplify { - Some(Box::new(|_, _| Ok(col("result_column")))) - } else { - None + match self.simplify { + Some(true) => Some(Box::new(|_, _| Ok(col("result_column")))), + Some(false) => Some(Box::new(|aggregate_function, _| { + Ok(Expr::AggregateFunction(aggregate_function)) + })), + None => None, } } } diff --git a/datafusion/optimizer/src/simplify_expressions/simplify_exprs.rs b/datafusion/optimizer/src/simplify_expressions/simplify_exprs.rs index 41dd24b6d857f..65dccc7126e53 100644 --- a/datafusion/optimizer/src/simplify_expressions/simplify_exprs.rs +++ b/datafusion/optimizer/src/simplify_expressions/simplify_exprs.rs @@ -396,7 +396,7 @@ mod tests { plan, @r" Projection: sum(test.a) + Int64(2) * CAST(count(test.a) AS Int64) AS sum(test.a + Int64(2)), sum(test.a) + Int64(3) * CAST(count(test.a) AS Int64) AS sum(test.a + Int64(3)) - Aggregate: groupBy=[[]], aggr=[[sum(test.a), count(test.a)]] + Aggregate: groupBy=[[]], aggr=[[sum(test.a), count() AS count(test.a)]] TableScan: test " )?; diff --git a/datafusion/optimizer/src/single_distinct_to_groupby.rs b/datafusion/optimizer/src/single_distinct_to_groupby.rs index f663003a9e4a4..d366532d29e0e 100644 --- a/datafusion/optimizer/src/single_distinct_to_groupby.rs +++ b/datafusion/optimizer/src/single_distinct_to_groupby.rs @@ -104,6 +104,32 @@ struct CountRollup { sum: Arc, } +fn unalias_top(mut expr: &Expr) -> &Expr { + while let Expr::Alias(alias) = expr + && alias + .metadata + .as_ref() + .is_none_or(|metadata| metadata.is_empty()) + { + expr = &alias.expr; + } + expr +} + +fn into_unaliased_top(expr: Expr) -> Expr { + match expr { + Expr::Alias(alias) + if alias + .metadata + .as_ref() + .is_none_or(|metadata| metadata.is_empty()) => + { + into_unaliased_top(*alias.expr) + } + expr => expr, + } +} + impl CountRollup { fn try_new(config: &dyn OptimizerConfig) -> Option { let registry = config.function_registry()?; @@ -131,6 +157,7 @@ fn is_single_distinct_agg( let mut distinct_aggs = vec![]; let mut has_count_rollup = false; for expr in aggr_expr { + let expr = unalias_top(expr); if let Expr::AggregateFunction(AggregateFunction { func, params: @@ -301,7 +328,7 @@ impl OptimizerRule for SingleDistinctToGroupBy { // zero that `sum` reports as NULL over an empty input. let (outer_aggr_exprs, outer_proj_exprs): (Vec, Vec) = aggr_expr .into_iter() - .map(|aggr_expr| match aggr_expr { + .map(|aggr_expr| match into_unaliased_top(aggr_expr) { Expr::AggregateFunction(AggregateFunction { func, params: @@ -381,7 +408,7 @@ impl OptimizerRule for SingleDistinctToGroupBy { Ok((outer, proj)) } } - _ => Ok((aggr_expr.clone(), aggr_expr)), + aggr_expr => Ok((aggr_expr.clone(), aggr_expr)), }) .collect::>>()? .into_iter() @@ -1157,6 +1184,41 @@ mod tests { ) } + #[test] + fn aliased_count_star_and_distinct_without_groupby() -> Result<()> { + let table_scan = test_table_scan_utf8_b()?; + + // Simplifying `count(1)` to `count()` preserves its old name with an + // alias before this rule runs. + let plan = LogicalPlanBuilder::from(table_scan) + .aggregate( + Vec::::new(), + vec![ + Expr::AggregateFunction(AggregateFunction::new_udf( + count_udaf(), + vec![], + false, + None, + vec![], + None, + )) + .alias("count(Int64(1))"), + count_distinct(col("b")), + ], + )? + .build()?; + + assert_optimized_plan_equal!( + plan, + @r" + Projection: CASE WHEN sum(alias2) IS NOT NULL THEN sum(alias2) ELSE Int64(0) END AS count(Int64(1)), count(alias1) AS count(DISTINCT test.b) [count(Int64(1)):Int64, count(DISTINCT test.b):Int64] + Aggregate: groupBy=[[]], aggr=[[sum(alias2), count(alias1)]] [sum(alias2):Int64;N, count(alias1):Int64] + Aggregate: groupBy=[[test.b AS alias1]], aggr=[[count() AS alias2]] [alias1:Utf8, alias2:Int64] + TableScan: test [a:UInt32, b:Utf8, c:UInt32] + " + ) + } + #[test] fn count_star_min_max_sum_and_distinct_with_groupby() -> Result<()> { let table_scan = test_table_scan_utf8_b()?; diff --git a/datafusion/optimizer/tests/optimizer_integration.rs b/datafusion/optimizer/tests/optimizer_integration.rs index 26b48c5e1f352..0db3131c29827 100644 --- a/datafusion/optimizer/tests/optimizer_integration.rs +++ b/datafusion/optimizer/tests/optimizer_integration.rs @@ -296,7 +296,7 @@ fn between_date32_plus_interval() -> Result<()> { assert_snapshot!( format!("{plan}"), @r#" - Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] + Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] Projection: Filter: test.col_date32 >= Date32("1998-03-18") AND test.col_date32 <= Date32("1998-06-16") TableScan: test projection=[col_date32] @@ -314,7 +314,7 @@ fn between_date64_plus_interval() -> Result<()> { assert_snapshot!( format!("{plan}"), @r#" - Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] + Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] Projection: Filter: test.col_date64 >= Date64("1998-03-18") AND test.col_date64 <= Date64("1998-06-16") TableScan: test projection=[col_date64] @@ -388,7 +388,7 @@ fn push_down_filter_groupby_expr_contains_alias() { format!("{plan}"), @r" Projection: test.col_int32 + test.col_uint32 AS c, count(Int64(1)) AS count(*) - Aggregate: groupBy=[[CAST(test.col_int32 AS Int64) + CAST(test.col_uint32 AS Int64)]], aggr=[[count(Int64(1))]] + Aggregate: groupBy=[[CAST(test.col_int32 AS Int64) + CAST(test.col_uint32 AS Int64)]], aggr=[[count() AS count(Int64(1))]] Filter: CAST(test.col_int32 AS Int64) + CAST(test.col_uint32 AS Int64) > Int64(3) TableScan: test projection=[col_int32, col_uint32] " @@ -445,7 +445,7 @@ fn eliminate_redundant_null_check_on_count() { format!("{plan}"), @r" Projection: test.col_int32, count(Int64(1)) AS count(*) AS c - Aggregate: groupBy=[[test.col_int32]], aggr=[[count(Int64(1))]] + Aggregate: groupBy=[[test.col_int32]], aggr=[[count() AS count(Int64(1))]] TableScan: test projection=[col_int32] " ); diff --git a/datafusion/physical-expr/src/aggregate.rs b/datafusion/physical-expr/src/aggregate.rs index 6d95d8ea12bd8..9f1dfaac3a4a1 100644 --- a/datafusion/physical-expr/src/aggregate.rs +++ b/datafusion/physical-expr/src/aggregate.rs @@ -44,9 +44,7 @@ use crate::planner::{create_physical_expr, create_physical_exprs}; use arrow::compute::SortOptions; use arrow::datatypes::{DataType, FieldRef, Schema, SchemaRef}; use datafusion_common::metadata::FieldMetadata; -use datafusion_common::{ - DFSchema, Result, ScalarValue, assert_or_internal_err, internal_err, not_impl_err, -}; +use datafusion_common::{DFSchema, Result, ScalarValue, internal_err, not_impl_err}; use datafusion_expr::execution_props::ExecutionProps; use datafusion_expr::expr::{ AggregateFunction, AggregateFunctionParams, NullTreatment, physical_name, @@ -262,8 +260,6 @@ impl AggregateExprBuilder { is_distinct, is_reversed, } = self; - assert_or_internal_err!(!args.is_empty(), "args should not be empty"); - // An order-insensitive aggregate ignores its ORDER BY, so drop it here. // Everything derived from `order_bys` below, such as the ordering fields // in the aggregate's state, then agrees that there is no ordering. diff --git a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common.rs b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common.rs index 0c5c4185e2f92..6d3c67078bf73 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common.rs @@ -583,6 +583,8 @@ pub(super) struct RowAlignedAccumulatorArgs { pub(super) arguments: Vec, /// Original row-aligned filter passed through to state conversion. pub(super) filter: Option, + /// Number of rows represented by the arguments. + pub(super) num_rows: usize, } /// Evaluated all group by keys and accumulator args. @@ -813,7 +815,11 @@ impl HashAggregateAccumulator { }) .collect::>()?; - Ok(RowAlignedAccumulatorArgs { arguments, filter }) + Ok(RowAlignedAccumulatorArgs { + arguments, + filter, + num_rows: batch.num_rows(), + }) } fn evaluate_filter(&self, batch: &RecordBatch) -> Result> { @@ -888,8 +894,11 @@ impl HashAggregateAccumulator { &self, values: &RowAlignedAccumulatorArgs, ) -> Result> { - self.accumulator - .convert_to_state(&values.arguments, values.filter.as_ref()) + self.accumulator.convert_to_state_with_num_rows( + &values.arguments, + values.filter.as_ref(), + values.num_rows, + ) } pub(super) fn null_arguments( diff --git a/datafusion/physical-plan/src/aggregates/aggregate_stream.rs b/datafusion/physical-plan/src/aggregates/aggregate_stream.rs index 3ce7915806196..49f6c2f0cf216 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_stream.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_stream.rs @@ -32,7 +32,9 @@ use crate::{RecordBatchStream, SendableRecordBatchStream}; use arrow::array::ArrayRef; use arrow::datatypes::SchemaRef; use arrow::record_batch::RecordBatch; -use datafusion_common::{Result, ScalarValue, internal_datafusion_err, internal_err}; +use datafusion_common::{ + DataFusionError, Result, ScalarValue, internal_datafusion_err, internal_err, +}; use datafusion_execution::TaskContext; use datafusion_expr::Operator; use datafusion_physical_expr::PhysicalExpr; @@ -496,12 +498,13 @@ fn aggregate_batch( .enumerate() .try_for_each(|(index, ((accum, expr), filter))| { // 1.2 and 1.3 - let values = aggregate_argument_metrics.time(index, || { + let (values, num_rows) = aggregate_argument_metrics.time(index, || { let batch = match filter { Some(filter) => Cow::Owned(batch_filter(batch, filter)?), None => Cow::Borrowed(batch), }; - evaluate_expressions_to_arrays(expr, batch.as_ref()) + let values = evaluate_expressions_to_arrays(expr, batch.as_ref())?; + Ok::<_, DataFusionError>((values, batch.num_rows())) })?; // 1.4 @@ -510,7 +513,7 @@ fn aggregate_batch( AggregateInputMode::Raw => aggregate_accumulator_metrics.time( index, AccumulatorPhase::Update, - || accum.update_batch(&values), + || accum.update_batch_with_num_rows(&values, num_rows), ), AggregateInputMode::Partial => aggregate_accumulator_metrics.time( index, diff --git a/datafusion/physical-plan/src/aggregates/grouped_hash_stream.rs b/datafusion/physical-plan/src/aggregates/grouped_hash_stream.rs index 042f4a90449dc..e3861a7a70cb9 100644 --- a/datafusion/physical-plan/src/aggregates/grouped_hash_stream.rs +++ b/datafusion/physical-plan/src/aggregates/grouped_hash_stream.rs @@ -1469,7 +1469,13 @@ impl GroupedHashAggregateStream { output.extend(self.aggregate_accumulator_metrics.time( idx, AccumulatorPhase::ConvertToState, - || acc.convert_to_state(values, opt_filter), + || { + acc.convert_to_state_with_num_rows( + values, + opt_filter, + batch.num_rows(), + ) + }, )?); } diff --git a/datafusion/physical-plan/src/windows/mod.rs b/datafusion/physical-plan/src/windows/mod.rs index 7c5f55f661f38..d000d8e980e63 100644 --- a/datafusion/physical-plan/src/windows/mod.rs +++ b/datafusion/physical-plan/src/windows/mod.rs @@ -33,7 +33,7 @@ use crate::{ use arrow::datatypes::{Schema, SchemaRef}; use arrow_schema::{FieldRef, SortOptions}; -use datafusion_common::{Result, assert_or_internal_err, exec_err}; +use datafusion_common::{Result, assert_or_internal_err, exec_err, not_impl_err}; use datafusion_expr::{ LimitEffect, PartitionEvaluator, ReversedUDWF, SetMonotonicity, WindowFrame, WindowFunctionDefinition, WindowUDF, @@ -103,6 +103,12 @@ pub fn create_window_expr( ) -> Result> { Ok(match fun { WindowFunctionDefinition::AggregateUDF(fun) => { + if args.is_empty() { + return not_impl_err!( + "Aggregate window function {} without arguments is not supported", + fun.name() + ); + } let aggregate = if distinct { AggregateExprBuilder::new(Arc::clone(fun), args.to_vec()) .schema(input_schema) @@ -957,6 +963,29 @@ mod tests { Ok(()) } + #[test] + fn create_window_expr_rejects_aggregate_without_args() { + let schema = Arc::new(Schema::new(vec![Field::new("a", DataType::Int32, true)])); + let err = create_window_expr( + &WindowFunctionDefinition::AggregateUDF(count_udaf()), + "count".to_owned(), + &[], + &[], + &[], + Arc::new(WindowFrame::new(None)), + schema, + false, + false, + None, + ) + .unwrap_err(); + assert!( + err.to_string() + .contains("without arguments is not supported"), + "{err}" + ); + } + #[tokio::test] async fn get_best_fitting_window_preserves_state_observer() -> Result<()> { // `EnforceSorting`/`EnforceDistribution` call `get_best_fitting_window` diff --git a/datafusion/sql/src/unparser/expr.rs b/datafusion/sql/src/unparser/expr.rs index 0f2cef167cc9d..ba73ee98f33e1 100644 --- a/datafusion/sql/src/unparser/expr.rs +++ b/datafusion/sql/src/unparser/expr.rs @@ -417,6 +417,11 @@ impl Unparser<'_> { .map(|sort_expr| self.sort_to_sql(sort_expr)) .collect::>>()?; (args_to_use, within_group) + } else if args.is_empty() && func_name == "count" { + // Many dialects only accept `count(*)` for a parameterless count + let wildcard = + ast::FunctionArg::Unnamed(ast::FunctionArgExpr::Wildcard); + (vec![wildcard], Vec::new()) } else { (self.function_args_to_sql(args)?, Vec::new()) }; @@ -2278,6 +2283,7 @@ mod tests { .unwrap(), "count(*) FILTER (WHERE true)", ), + (count_udaf().call(vec![]), "count(*)"), ( Expr::from(WindowFunction { fun: WindowFunctionDefinition::WindowUDF(row_number_udwf()), diff --git a/datafusion/sqllogictest/test_files/aggregate.slt b/datafusion/sqllogictest/test_files/aggregate.slt index 2be4e173fc701..8477cc92431dc 100644 --- a/datafusion/sqllogictest/test_files/aggregate.slt +++ b/datafusion/sqllogictest/test_files/aggregate.slt @@ -8912,11 +8912,13 @@ query TT explain select count(1), count(2) from t; ---- logical_plan -01)Aggregate: groupBy=[[]], aggr=[[count(Int64(1)), count(Int64(2))]] -02)--TableScan: t projection=[] +01)Projection: __common_expr_1 AS count(Int64(1)), __common_expr_1 AS count(Int64(2)) +02)--Aggregate: groupBy=[[]], aggr=[[count() AS __common_expr_1]] +03)----TableScan: t projection=[] physical_plan -01)AggregateExec: mode=Single, gby=[], aggr=[count(Int64(1)), count(Int64(2))] -02)--DataSourceExec: partitions=1, partition_sizes=[1] +01)ProjectionExec: expr=[__common_expr_1@0 as count(Int64(1)), __common_expr_1@0 as count(Int64(2))] +02)--ProjectionExec: expr=[2 as __common_expr_1] +03)----PlaceholderRowExec query II select count(1), count() from t; @@ -8928,7 +8930,7 @@ explain select count(1), count() from t; ---- logical_plan 01)Projection: count(Int64(1)), count(Int64(1)) AS count() -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 03)----TableScan: t projection=[] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(Int64(1)), count(Int64(1))@0 as count()] @@ -8945,7 +8947,7 @@ explain select count(1), count(*) from t; ---- logical_plan 01)Projection: count(Int64(1)), count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 03)----TableScan: t projection=[] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(Int64(1)), count(Int64(1))@0 as count(*)] @@ -8962,7 +8964,7 @@ explain select count(), count(*) from t; ---- logical_plan 01)Projection: count(Int64(1)) AS count(), count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 03)----TableScan: t projection=[] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(), count(Int64(1))@0 as count(*)] @@ -8973,13 +8975,13 @@ query TT explain select count(1) * count(2) from t; ---- logical_plan -01)Projection: count(Int64(1)) * count(Int64(2)) -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1)), count(Int64(2))]] +01)Projection: __common_expr_1 * __common_expr_1 AS count(Int64(1)) * count(Int64(2)) +02)--Aggregate: groupBy=[[]], aggr=[[count() AS __common_expr_1]] 03)----TableScan: t projection=[] physical_plan -01)ProjectionExec: expr=[count(Int64(1))@0 * count(Int64(2))@1 as count(Int64(1)) * count(Int64(2))] -02)--AggregateExec: mode=Single, gby=[], aggr=[count(Int64(1)), count(Int64(2))] -03)----DataSourceExec: partitions=1, partition_sizes=[1] +01)ProjectionExec: expr=[__common_expr_1@0 * __common_expr_1@0 as count(Int64(1)) * count(Int64(2))] +02)--ProjectionExec: expr=[2 as __common_expr_1] +03)----PlaceholderRowExec statement count 0 drop table t; @@ -9491,12 +9493,12 @@ ORDER BY g; logical_plan 01)Sort: stream_test.g ASC NULLS LAST 02)--Projection: stream_test.g, count(Int64(1)) AS count(*), sum(stream_test.x), avg(stream_test.x), avg(stream_test.x) AS mean(stream_test.x), min(stream_test.x), max(stream_test.y), bit_and(stream_test.i), bit_or(stream_test.i), bit_xor(stream_test.i), bool_and(stream_test.b), bool_or(stream_test.b), median(stream_test.x), Int32(0) AS grouping(stream_test.g), var(stream_test.x), var(stream_test.x) AS var_samp(stream_test.x), var_pop(stream_test.x), var(stream_test.x) AS var_sample(stream_test.x), var_pop(stream_test.x) AS var_population(stream_test.x), stddev(stream_test.x), stddev(stream_test.x) AS stddev_samp(stream_test.x), stddev_pop(stream_test.x) -03)----Aggregate: groupBy=[[stream_test.g]], aggr=[[count(Int64(1)), sum(stream_test.x), avg(stream_test.x), min(stream_test.x), max(stream_test.y), bit_and(stream_test.i), bit_or(stream_test.i), bit_xor(stream_test.i), bool_and(stream_test.b), bool_or(stream_test.b), median(stream_test.x), var(stream_test.x), var_pop(stream_test.x), stddev(stream_test.x), stddev_pop(stream_test.x)]] +03)----Aggregate: groupBy=[[stream_test.g]], aggr=[[count() AS count(Int64(1)), sum(stream_test.x), avg(stream_test.x), min(stream_test.x), max(stream_test.y), bit_and(stream_test.i), bit_or(stream_test.i), bit_xor(stream_test.i), bool_and(stream_test.b), bool_or(stream_test.b), median(stream_test.x), var(stream_test.x), var_pop(stream_test.x), stddev(stream_test.x), stddev_pop(stream_test.x)]] 04)------Sort: stream_test.g ASC NULLS LAST, fetch=10000 05)--------TableScan: stream_test projection=[g, x, y, i, b] physical_plan 01)ProjectionExec: expr=[g@0 as g, count(Int64(1))@1 as count(*), sum(stream_test.x)@2 as sum(stream_test.x), avg(stream_test.x)@3 as avg(stream_test.x), avg(stream_test.x)@3 as mean(stream_test.x), min(stream_test.x)@4 as min(stream_test.x), max(stream_test.y)@5 as max(stream_test.y), bit_and(stream_test.i)@6 as bit_and(stream_test.i), bit_or(stream_test.i)@7 as bit_or(stream_test.i), bit_xor(stream_test.i)@8 as bit_xor(stream_test.i), bool_and(stream_test.b)@9 as bool_and(stream_test.b), bool_or(stream_test.b)@10 as bool_or(stream_test.b), median(stream_test.x)@11 as median(stream_test.x), 0 as grouping(stream_test.g), var(stream_test.x)@12 as var(stream_test.x), var(stream_test.x)@12 as var_samp(stream_test.x), var_pop(stream_test.x)@13 as var_pop(stream_test.x), var(stream_test.x)@12 as var_sample(stream_test.x), var_pop(stream_test.x)@13 as var_population(stream_test.x), stddev(stream_test.x)@14 as stddev(stream_test.x), stddev(stream_test.x)@14 as stddev_samp(stream_test.x), stddev_pop(stream_test.x)@15 as stddev_pop(stream_test.x)] -02)--AggregateExec: mode=Single, gby=[g@0 as g], aggr=[count(Int64(1)), sum(stream_test.x), avg(stream_test.x), min(stream_test.x), max(stream_test.y), bit_and(stream_test.i), bit_or(stream_test.i), bit_xor(stream_test.i), bool_and(stream_test.b), bool_or(stream_test.b), median(stream_test.x), var(stream_test.x), var_pop(stream_test.x), stddev(stream_test.x), stddev_pop(stream_test.x)], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[g@0 as g], aggr=[count() as count(Int64(1)), sum(stream_test.x), avg(stream_test.x), min(stream_test.x), max(stream_test.y), bit_and(stream_test.i), bit_or(stream_test.i), bit_xor(stream_test.i), bool_and(stream_test.b), bool_or(stream_test.b), median(stream_test.x), var(stream_test.x), var_pop(stream_test.x), stddev(stream_test.x), stddev_pop(stream_test.x)], ordering_mode=Sorted 03)----SortExec: TopK(fetch=10000), expr=[g@0 ASC NULLS LAST], preserve_partitioning=[false] 04)------DataSourceExec: partitions=1, partition_sizes=[1] diff --git a/datafusion/sqllogictest/test_files/aggregate_memory_spill.slt b/datafusion/sqllogictest/test_files/aggregate_memory_spill.slt index 48622127bce99..f8b05506800e8 100644 --- a/datafusion/sqllogictest/test_files/aggregate_memory_spill.slt +++ b/datafusion/sqllogictest/test_files/aggregate_memory_spill.slt @@ -113,7 +113,7 @@ FROM ( ) ---- -04)------AggregateExec: mode=Single, gby=[group_alias_0@0 as group_alias_0], aggr=[count(alias1)], metrics=[spill_count=7,] +04)------AggregateExec: mode=Single, gby=[group_alias_0@0 as group_alias_0], aggr=[count() as count(alias1)], metrics=[spill_count=7,] # --- Case D: multiple aggregates (sum/min/max) under memory limit --- diff --git a/datafusion/sqllogictest/test_files/aggregate_repartition.slt b/datafusion/sqllogictest/test_files/aggregate_repartition.slt index 2302e161bfe72..acf6553ac7e18 100644 --- a/datafusion/sqllogictest/test_files/aggregate_repartition.slt +++ b/datafusion/sqllogictest/test_files/aggregate_repartition.slt @@ -72,13 +72,13 @@ EXPLAIN SELECT env, count(*) FROM dim_csv GROUP BY env; ---- logical_plan 01)Projection: dim_csv.env, count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[dim_csv.env]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[dim_csv.env]], aggr=[[count() AS count(Int64(1))]] 03)----TableScan: dim_csv projection=[env] physical_plan 01)ProjectionExec: expr=[env@0 as env, count(Int64(1))@1 as count(*)] -02)--AggregateExec: mode=FinalPartitioned, gby=[env@0 as env], aggr=[count(Int64(1))] +02)--AggregateExec: mode=FinalPartitioned, gby=[env@0 as env], aggr=[count() as count(Int64(1))] 03)----RepartitionExec: partitioning=Hash([env@0], 4), input_partitions=4 -04)------AggregateExec: mode=Partial, gby=[env@0 as env], aggr=[count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[env@0 as env], aggr=[count() as count(Int64(1))] 05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/aggregate_repartition/dim.csv]]}, projection=[env], file_type=csv, has_header=true @@ -89,13 +89,13 @@ EXPLAIN SELECT env, count(*) FROM dim_parquet GROUP BY env; ---- logical_plan 01)Projection: dim_parquet.env, count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[dim_parquet.env]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[dim_parquet.env]], aggr=[[count() AS count(Int64(1))]] 03)----TableScan: dim_parquet projection=[env] physical_plan 01)ProjectionExec: expr=[env@0 as env, count(Int64(1))@1 as count(*)] -02)--AggregateExec: mode=FinalPartitioned, gby=[env@0 as env], aggr=[count(Int64(1))] +02)--AggregateExec: mode=FinalPartitioned, gby=[env@0 as env], aggr=[count() as count(Int64(1))] 03)----RepartitionExec: partitioning=Hash([env@0], 4), input_partitions=1 -04)------AggregateExec: mode=Partial, gby=[env@0 as env], aggr=[count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[env@0 as env], aggr=[count() as count(Int64(1))] 05)--------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/aggregate_repartition/dim.parquet]]}, projection=[env], file_type=parquet # Verify the queries actually work and return the same results @@ -122,11 +122,11 @@ EXPLAIN SELECT env, count(*) FROM dim_parquet GROUP BY env; ---- logical_plan 01)Projection: dim_parquet.env, count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[dim_parquet.env]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[dim_parquet.env]], aggr=[[count() AS count(Int64(1))]] 03)----TableScan: dim_parquet projection=[env] physical_plan 01)ProjectionExec: expr=[env@0 as env, count(Int64(1))@1 as count(*)] -02)--AggregateExec: mode=Single, gby=[env@0 as env], aggr=[count(Int64(1))] +02)--AggregateExec: mode=Single, gby=[env@0 as env], aggr=[count() as count(Int64(1))] 03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/aggregate_repartition/dim.parquet]]}, projection=[env], file_type=parquet # Config reset diff --git a/datafusion/sqllogictest/test_files/array/array_has.slt b/datafusion/sqllogictest/test_files/array/array_has.slt index d7b6680fab062..4fda9253239c7 100644 --- a/datafusion/sqllogictest/test_files/array/array_has.slt +++ b/datafusion/sqllogictest/test_files/array/array_has.slt @@ -505,7 +505,7 @@ select count(*) from test WHERE needle IN ('7f4b18de3cfeb9b4ac78c381ee2ad278', ' ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 03)----SubqueryAlias: test 04)------SubqueryAlias: t 05)--------Projection: @@ -513,9 +513,9 @@ logical_plan 07)------------TableScan: generate_series() projection=[value] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] 05)--------FilterExec: substr(md5(CAST(value@0 AS Utf8View)), 1, 32) IN (SET) ([7f4b18de3cfeb9b4ac78c381ee2ad278, a, b, c]), projection=[] 06)----------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 07)------------LazyMemoryExec: partitions=1, batch_generators=[generate_series: start=1, end=100000, batch_size=8192] @@ -532,7 +532,7 @@ select count(*) from test WHERE needle = ANY(['7f4b18de3cfeb9b4ac78c381ee2ad278' ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 03)----SubqueryAlias: test 04)------SubqueryAlias: t 05)--------Projection: @@ -540,9 +540,9 @@ logical_plan 07)------------TableScan: generate_series() projection=[value] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] 05)--------FilterExec: substr(md5(CAST(value@0 AS Utf8View)), 1, 32) IN (SET) ([7f4b18de3cfeb9b4ac78c381ee2ad278, a, b, c]), projection=[] 06)----------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 07)------------LazyMemoryExec: partitions=1, batch_generators=[generate_series: start=1, end=100000, batch_size=8192] @@ -559,7 +559,7 @@ select count(*) from test WHERE array_has(['7f4b18de3cfeb9b4ac78c381ee2ad278', ' ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 03)----SubqueryAlias: test 04)------SubqueryAlias: t 05)--------Projection: @@ -567,9 +567,9 @@ logical_plan 07)------------TableScan: generate_series() projection=[value] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] 05)--------FilterExec: substr(md5(CAST(value@0 AS Utf8View)), 1, 32) IN (SET) ([7f4b18de3cfeb9b4ac78c381ee2ad278, a, b, c]), projection=[] 06)----------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 07)------------LazyMemoryExec: partitions=1, batch_generators=[generate_series: start=1, end=100000, batch_size=8192] @@ -586,7 +586,7 @@ select count(*) from test WHERE array_has(arrow_cast(['7f4b18de3cfeb9b4ac78c381e ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 03)----SubqueryAlias: test 04)------SubqueryAlias: t 05)--------Projection: @@ -594,9 +594,9 @@ logical_plan 07)------------TableScan: generate_series() projection=[value] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] 05)--------FilterExec: substr(md5(CAST(value@0 AS Utf8View)), 1, 32) IN (SET) ([7f4b18de3cfeb9b4ac78c381ee2ad278, a, b, c]), projection=[] 06)----------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 07)------------LazyMemoryExec: partitions=1, batch_generators=[generate_series: start=1, end=100000, batch_size=8192] @@ -613,7 +613,7 @@ select count(*) from test WHERE array_has(arrow_cast(['7f4b18de3cfeb9b4ac78c381e ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 03)----SubqueryAlias: test 04)------SubqueryAlias: t 05)--------Projection: @@ -621,9 +621,9 @@ logical_plan 07)------------TableScan: generate_series() projection=[value] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] 05)--------FilterExec: substr(md5(CAST(value@0 AS Utf8View)), 1, 32) IN (SET) ([7f4b18de3cfeb9b4ac78c381ee2ad278, a, b, c]), projection=[] 06)----------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 07)------------LazyMemoryExec: partitions=1, batch_generators=[generate_series: start=1, end=100000, batch_size=8192] @@ -641,7 +641,7 @@ select count(*) from test WHERE array_has([needle], needle); ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 03)----SubqueryAlias: test 04)------SubqueryAlias: t 05)--------Projection: @@ -649,9 +649,9 @@ logical_plan 07)------------TableScan: generate_series() projection=[value] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] 05)--------FilterExec: substr(md5(CAST(value@0 AS Utf8View)), 1, 32) IS NOT NULL, projection=[] 06)----------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 07)------------LazyMemoryExec: partitions=1, batch_generators=[generate_series: start=1, end=100000, batch_size=8192] diff --git a/datafusion/sqllogictest/test_files/avro.slt b/datafusion/sqllogictest/test_files/avro.slt index 04e8fb7a8796d..9e2382aa343f8 100644 --- a/datafusion/sqllogictest/test_files/avro.slt +++ b/datafusion/sqllogictest/test_files/avro.slt @@ -265,13 +265,13 @@ EXPLAIN SELECT count(*) from alltypes_plain ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 03)----TableScan: alltypes_plain projection=[] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] 05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/testing/data/avro/alltypes_plain.avro]]}, file_type=avro diff --git a/datafusion/sqllogictest/test_files/clickbench.slt b/datafusion/sqllogictest/test_files/clickbench.slt index 7cb5547383c38..192d97ee2dd02 100644 --- a/datafusion/sqllogictest/test_files/clickbench.slt +++ b/datafusion/sqllogictest/test_files/clickbench.slt @@ -60,7 +60,7 @@ EXPLAIN SELECT COUNT(*) FROM hits; ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 03)----SubqueryAlias: hits 04)------TableScan: hits_raw projection=[] physical_plan @@ -78,16 +78,16 @@ EXPLAIN SELECT COUNT(*) FROM hits WHERE "AdvEngineID" <> 0; ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 03)----SubqueryAlias: hits 04)------Projection: 05)--------Filter: hits_raw.AdvEngineID != Int16(0) 06)----------TableScan: hits_raw projection=[AdvEngineID], partial_filters=[hits_raw.AdvEngineID != Int16(0)] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] 05)--------FilterExec: AdvEngineID@0 != 0, projection=[] 06)----------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 07)------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[AdvEngineID], file_type=parquet, predicate=AdvEngineID@40 != 0, pruning_predicate=AdvEngineID_null_count@2 != row_count@3 AND (AdvEngineID_min@0 != 0 OR 0 != AdvEngineID_max@1), required_guarantees=[AdvEngineID not in (0)] @@ -102,12 +102,12 @@ EXPLAIN SELECT SUM("AdvEngineID"), COUNT(*), AVG("ResolutionWidth") FROM hits; ---- logical_plan 01)Projection: sum(hits.AdvEngineID), count(Int64(1)) AS count(*), avg(hits.ResolutionWidth) -02)--Aggregate: groupBy=[[]], aggr=[[sum(CAST(hits.AdvEngineID AS Int64)), count(Int64(1)), avg(CAST(hits.ResolutionWidth AS Float64))]] +02)--Aggregate: groupBy=[[]], aggr=[[sum(CAST(hits.AdvEngineID AS Int64)), count() AS count(Int64(1)), avg(CAST(hits.ResolutionWidth AS Float64))]] 03)----SubqueryAlias: hits 04)------TableScan: hits_raw projection=[ResolutionWidth, AdvEngineID] physical_plan 01)ProjectionExec: expr=[sum(hits.AdvEngineID)@0 as sum(hits.AdvEngineID), count(Int64(1))@1 as count(*), avg(hits.ResolutionWidth)@2 as avg(hits.ResolutionWidth)] -02)--AggregateExec: mode=Single, gby=[], aggr=[sum(hits.AdvEngineID), count(Int64(1)), avg(hits.ResolutionWidth)] +02)--AggregateExec: mode=Single, gby=[], aggr=[sum(hits.AdvEngineID), count() as count(Int64(1)), avg(hits.ResolutionWidth)] 03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[ResolutionWidth, AdvEngineID], file_type=parquet query IIR @@ -207,7 +207,7 @@ EXPLAIN SELECT "AdvEngineID", COUNT(*) FROM hits WHERE "AdvEngineID" <> 0 GROUP logical_plan 01)Sort: count(*) DESC NULLS FIRST 02)--Projection: hits.AdvEngineID, count(Int64(1)) AS count(*) -03)----Aggregate: groupBy=[[hits.AdvEngineID]], aggr=[[count(Int64(1))]] +03)----Aggregate: groupBy=[[hits.AdvEngineID]], aggr=[[count() AS count(Int64(1))]] 04)------SubqueryAlias: hits 05)--------Filter: hits_raw.AdvEngineID != Int16(0) 06)----------TableScan: hits_raw projection=[AdvEngineID], partial_filters=[hits_raw.AdvEngineID != Int16(0)] @@ -215,9 +215,9 @@ physical_plan 01)SortPreservingMergeExec: [count(*)@1 DESC] 02)--ProjectionExec: expr=[AdvEngineID@0 as AdvEngineID, count(Int64(1))@1 as count(*)] 03)----SortExec: expr=[count(Int64(1))@1 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[AdvEngineID@0 as AdvEngineID], aggr=[count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[AdvEngineID@0 as AdvEngineID], aggr=[count() as count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([AdvEngineID@0], 4), input_partitions=4 -06)----------AggregateExec: mode=Partial, gby=[AdvEngineID@0 as AdvEngineID], aggr=[count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[AdvEngineID@0 as AdvEngineID], aggr=[count() as count(Int64(1))] 07)------------FilterExec: AdvEngineID@0 != 0 08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[AdvEngineID], file_type=parquet, predicate=AdvEngineID@40 != 0, pruning_predicate=AdvEngineID_null_count@2 != row_count@3 AND (AdvEngineID_min@0 != 0 OR 0 != AdvEngineID_max@1), required_guarantees=[AdvEngineID not in (0)] @@ -264,16 +264,16 @@ EXPLAIN SELECT "RegionID", SUM("AdvEngineID"), COUNT(*) AS c, AVG("ResolutionWid logical_plan 01)Sort: c DESC NULLS FIRST, fetch=10 02)--Projection: hits.RegionID, sum(hits.AdvEngineID), count(Int64(1)) AS count(*) AS c, avg(hits.ResolutionWidth), count(DISTINCT hits.UserID) -03)----Aggregate: groupBy=[[hits.RegionID]], aggr=[[sum(CAST(hits.AdvEngineID AS Int64)), count(Int64(1)), avg(CAST(hits.ResolutionWidth AS Float64)), count(DISTINCT hits.UserID)]] +03)----Aggregate: groupBy=[[hits.RegionID]], aggr=[[sum(CAST(hits.AdvEngineID AS Int64)), count() AS count(Int64(1)), avg(CAST(hits.ResolutionWidth AS Float64)), count(DISTINCT hits.UserID)]] 04)------SubqueryAlias: hits 05)--------TableScan: hits_raw projection=[RegionID, UserID, ResolutionWidth, AdvEngineID] physical_plan 01)SortPreservingMergeExec: [c@2 DESC], fetch=10 02)--ProjectionExec: expr=[RegionID@0 as RegionID, sum(hits.AdvEngineID)@1 as sum(hits.AdvEngineID), count(Int64(1))@2 as c, avg(hits.ResolutionWidth)@3 as avg(hits.ResolutionWidth), count(DISTINCT hits.UserID)@4 as count(DISTINCT hits.UserID)] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@2 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[RegionID@0 as RegionID], aggr=[sum(hits.AdvEngineID), count(Int64(1)), avg(hits.ResolutionWidth), count(DISTINCT hits.UserID)] +04)------AggregateExec: mode=FinalPartitioned, gby=[RegionID@0 as RegionID], aggr=[sum(hits.AdvEngineID), count() as count(Int64(1)), avg(hits.ResolutionWidth), count(DISTINCT hits.UserID)] 05)--------RepartitionExec: partitioning=Hash([RegionID@0], 4), input_partitions=1 -06)----------AggregateExec: mode=Partial, gby=[RegionID@0 as RegionID], aggr=[sum(hits.AdvEngineID), count(Int64(1)), avg(hits.ResolutionWidth), count(DISTINCT hits.UserID)] +06)----------AggregateExec: mode=Partial, gby=[RegionID@0 as RegionID], aggr=[sum(hits.AdvEngineID), count() as count(Int64(1)), avg(hits.ResolutionWidth), count(DISTINCT hits.UserID)] 07)------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[RegionID, UserID, ResolutionWidth, AdvEngineID], file_type=parquet query IIIRI rowsort @@ -351,7 +351,7 @@ EXPLAIN SELECT "SearchPhrase", COUNT(*) AS c FROM hits WHERE "SearchPhrase" <> ' logical_plan 01)Sort: c DESC NULLS FIRST, fetch=10 02)--Projection: hits.SearchPhrase, count(Int64(1)) AS count(*) AS c -03)----Aggregate: groupBy=[[hits.SearchPhrase]], aggr=[[count(Int64(1))]] +03)----Aggregate: groupBy=[[hits.SearchPhrase]], aggr=[[count() AS count(Int64(1))]] 04)------SubqueryAlias: hits 05)--------Filter: hits_raw.SearchPhrase != Utf8View("") 06)----------TableScan: hits_raw projection=[SearchPhrase], partial_filters=[hits_raw.SearchPhrase != Utf8View("")] @@ -359,9 +359,9 @@ physical_plan 01)SortPreservingMergeExec: [c@1 DESC], fetch=10 02)--ProjectionExec: expr=[SearchPhrase@0 as SearchPhrase, count(Int64(1))@1 as c] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@1 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[SearchPhrase@0 as SearchPhrase], aggr=[count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[SearchPhrase@0 as SearchPhrase], aggr=[count() as count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([SearchPhrase@0], 4), input_partitions=4 -06)----------AggregateExec: mode=Partial, gby=[SearchPhrase@0 as SearchPhrase], aggr=[count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[SearchPhrase@0 as SearchPhrase], aggr=[count() as count(Int64(1))] 07)------------FilterExec: SearchPhrase@0 != 08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != , pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] @@ -407,7 +407,7 @@ EXPLAIN SELECT "SearchEngineID", "SearchPhrase", COUNT(*) AS c FROM hits WHERE " logical_plan 01)Sort: c DESC NULLS FIRST, fetch=10 02)--Projection: hits.SearchEngineID, hits.SearchPhrase, count(Int64(1)) AS count(*) AS c -03)----Aggregate: groupBy=[[hits.SearchEngineID, hits.SearchPhrase]], aggr=[[count(Int64(1))]] +03)----Aggregate: groupBy=[[hits.SearchEngineID, hits.SearchPhrase]], aggr=[[count() AS count(Int64(1))]] 04)------SubqueryAlias: hits 05)--------Filter: hits_raw.SearchPhrase != Utf8View("") 06)----------TableScan: hits_raw projection=[SearchEngineID, SearchPhrase], partial_filters=[hits_raw.SearchPhrase != Utf8View("")] @@ -415,9 +415,9 @@ physical_plan 01)SortPreservingMergeExec: [c@2 DESC], fetch=10 02)--ProjectionExec: expr=[SearchEngineID@0 as SearchEngineID, SearchPhrase@1 as SearchPhrase, count(Int64(1))@2 as c] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@2 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[SearchEngineID@0 as SearchEngineID, SearchPhrase@1 as SearchPhrase], aggr=[count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[SearchEngineID@0 as SearchEngineID, SearchPhrase@1 as SearchPhrase], aggr=[count() as count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([SearchEngineID@0, SearchPhrase@1], 4), input_partitions=4 -06)----------AggregateExec: mode=Partial, gby=[SearchEngineID@0 as SearchEngineID, SearchPhrase@1 as SearchPhrase], aggr=[count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[SearchEngineID@0 as SearchEngineID, SearchPhrase@1 as SearchPhrase], aggr=[count() as count(Int64(1))] 07)------------FilterExec: SearchPhrase@1 != 08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[SearchEngineID, SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != , pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] @@ -433,16 +433,16 @@ EXPLAIN SELECT "UserID", COUNT(*) FROM hits GROUP BY "UserID" ORDER BY COUNT(*) logical_plan 01)Sort: count(*) DESC NULLS FIRST, fetch=10 02)--Projection: hits.UserID, count(Int64(1)) AS count(*) -03)----Aggregate: groupBy=[[hits.UserID]], aggr=[[count(Int64(1))]] +03)----Aggregate: groupBy=[[hits.UserID]], aggr=[[count() AS count(Int64(1))]] 04)------SubqueryAlias: hits 05)--------TableScan: hits_raw projection=[UserID] physical_plan 01)SortPreservingMergeExec: [count(*)@1 DESC], fetch=10 02)--ProjectionExec: expr=[UserID@0 as UserID, count(Int64(1))@1 as count(*)] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@1 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[UserID@0 as UserID], aggr=[count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[UserID@0 as UserID], aggr=[count() as count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([UserID@0], 4), input_partitions=1 -06)----------AggregateExec: mode=Partial, gby=[UserID@0 as UserID], aggr=[count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[UserID@0 as UserID], aggr=[count() as count(Int64(1))] 07)------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[UserID], file_type=parquet query II rowsort @@ -461,16 +461,16 @@ EXPLAIN SELECT "UserID", "SearchPhrase", COUNT(*) FROM hits GROUP BY "UserID", " logical_plan 01)Sort: count(*) DESC NULLS FIRST, fetch=10 02)--Projection: hits.UserID, hits.SearchPhrase, count(Int64(1)) AS count(*) -03)----Aggregate: groupBy=[[hits.UserID, hits.SearchPhrase]], aggr=[[count(Int64(1))]] +03)----Aggregate: groupBy=[[hits.UserID, hits.SearchPhrase]], aggr=[[count() AS count(Int64(1))]] 04)------SubqueryAlias: hits 05)--------TableScan: hits_raw projection=[UserID, SearchPhrase] physical_plan 01)SortPreservingMergeExec: [count(*)@2 DESC], fetch=10 02)--ProjectionExec: expr=[UserID@0 as UserID, SearchPhrase@1 as SearchPhrase, count(Int64(1))@2 as count(*)] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@2 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[UserID@0 as UserID, SearchPhrase@1 as SearchPhrase], aggr=[count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[UserID@0 as UserID, SearchPhrase@1 as SearchPhrase], aggr=[count() as count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([UserID@0, SearchPhrase@1], 4), input_partitions=1 -06)----------AggregateExec: mode=Partial, gby=[UserID@0 as UserID, SearchPhrase@1 as SearchPhrase], aggr=[count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[UserID@0 as UserID, SearchPhrase@1 as SearchPhrase], aggr=[count() as count(Int64(1))] 07)------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[UserID, SearchPhrase], file_type=parquet query ITI rowsort @@ -489,15 +489,15 @@ EXPLAIN SELECT "UserID", "SearchPhrase", COUNT(*) FROM hits GROUP BY "UserID", " logical_plan 01)Projection: hits.UserID, hits.SearchPhrase, count(Int64(1)) AS count(*) 02)--Limit: skip=0, fetch=10 -03)----Aggregate: groupBy=[[hits.UserID, hits.SearchPhrase]], aggr=[[count(Int64(1))]] +03)----Aggregate: groupBy=[[hits.UserID, hits.SearchPhrase]], aggr=[[count() AS count(Int64(1))]] 04)------SubqueryAlias: hits 05)--------TableScan: hits_raw projection=[UserID, SearchPhrase] physical_plan 01)ProjectionExec: expr=[UserID@0 as UserID, SearchPhrase@1 as SearchPhrase, count(Int64(1))@2 as count(*)] 02)--CoalescePartitionsExec: fetch=10 -03)----AggregateExec: mode=FinalPartitioned, gby=[UserID@0 as UserID, SearchPhrase@1 as SearchPhrase], aggr=[count(Int64(1))] +03)----AggregateExec: mode=FinalPartitioned, gby=[UserID@0 as UserID, SearchPhrase@1 as SearchPhrase], aggr=[count() as count(Int64(1))] 04)------RepartitionExec: partitioning=Hash([UserID@0, SearchPhrase@1], 4), input_partitions=1 -05)--------AggregateExec: mode=Partial, gby=[UserID@0 as UserID, SearchPhrase@1 as SearchPhrase], aggr=[count(Int64(1))] +05)--------AggregateExec: mode=Partial, gby=[UserID@0 as UserID, SearchPhrase@1 as SearchPhrase], aggr=[count() as count(Int64(1))] 06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[UserID, SearchPhrase], file_type=parquet query ITI rowsort @@ -516,16 +516,16 @@ EXPLAIN SELECT "UserID", extract(minute FROM to_timestamp_seconds("EventTime")) logical_plan 01)Sort: count(*) DESC NULLS FIRST, fetch=10 02)--Projection: hits.UserID, date_part(Utf8("MINUTE"),to_timestamp_seconds(hits.EventTime)) AS m, hits.SearchPhrase, count(Int64(1)) AS count(*) -03)----Aggregate: groupBy=[[hits.UserID, date_part(Utf8("MINUTE"), to_timestamp_seconds(hits.EventTime)), hits.SearchPhrase]], aggr=[[count(Int64(1))]] +03)----Aggregate: groupBy=[[hits.UserID, date_part(Utf8("MINUTE"), to_timestamp_seconds(hits.EventTime)), hits.SearchPhrase]], aggr=[[count() AS count(Int64(1))]] 04)------SubqueryAlias: hits 05)--------TableScan: hits_raw projection=[EventTime, UserID, SearchPhrase] physical_plan 01)SortPreservingMergeExec: [count(*)@3 DESC], fetch=10 02)--ProjectionExec: expr=[UserID@0 as UserID, date_part(Utf8("MINUTE"),to_timestamp_seconds(hits.EventTime))@1 as m, SearchPhrase@2 as SearchPhrase, count(Int64(1))@3 as count(*)] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@3 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[UserID@0 as UserID, date_part(Utf8("MINUTE"),to_timestamp_seconds(hits.EventTime))@1 as date_part(Utf8("MINUTE"),to_timestamp_seconds(hits.EventTime)), SearchPhrase@2 as SearchPhrase], aggr=[count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[UserID@0 as UserID, date_part(Utf8("MINUTE"),to_timestamp_seconds(hits.EventTime))@1 as date_part(Utf8("MINUTE"),to_timestamp_seconds(hits.EventTime)), SearchPhrase@2 as SearchPhrase], aggr=[count() as count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([UserID@0, date_part(Utf8("MINUTE"),to_timestamp_seconds(hits.EventTime))@1, SearchPhrase@2], 4), input_partitions=1 -06)----------AggregateExec: mode=Partial, gby=[UserID@1 as UserID, date_part(MINUTE, to_timestamp_seconds(EventTime@0)) as date_part(Utf8("MINUTE"),to_timestamp_seconds(hits.EventTime)), SearchPhrase@2 as SearchPhrase], aggr=[count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[UserID@1 as UserID, date_part(MINUTE, to_timestamp_seconds(EventTime@0)) as date_part(Utf8("MINUTE"),to_timestamp_seconds(hits.EventTime)), SearchPhrase@2 as SearchPhrase], aggr=[count() as count(Int64(1))] 07)------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[EventTime, UserID, SearchPhrase], file_type=parquet query IITI rowsort @@ -565,16 +565,16 @@ EXPLAIN SELECT COUNT(*) FROM hits WHERE "URL" LIKE '%google%'; ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 03)----SubqueryAlias: hits 04)------Projection: 05)--------Filter: hits_raw.URL LIKE Utf8View("%google%") 06)----------TableScan: hits_raw projection=[URL], partial_filters=[hits_raw.URL LIKE Utf8View("%google%")] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] 05)--------FilterExec: URL@0 LIKE %google%, projection=[] 06)----------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 07)------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[URL], file_type=parquet, predicate=URL@13 LIKE %google% @@ -591,7 +591,7 @@ EXPLAIN SELECT "SearchPhrase", MIN("URL"), COUNT(*) AS c FROM hits WHERE "URL" L logical_plan 01)Sort: c DESC NULLS FIRST, fetch=10 02)--Projection: hits.SearchPhrase, min(hits.URL), count(Int64(1)) AS count(*) AS c -03)----Aggregate: groupBy=[[hits.SearchPhrase]], aggr=[[min(hits.URL), count(Int64(1))]] +03)----Aggregate: groupBy=[[hits.SearchPhrase]], aggr=[[min(hits.URL), count() AS count(Int64(1))]] 04)------SubqueryAlias: hits 05)--------Filter: hits_raw.SearchPhrase != Utf8View("") AND hits_raw.URL LIKE Utf8View("%google%") 06)----------TableScan: hits_raw projection=[URL, SearchPhrase], partial_filters=[hits_raw.SearchPhrase != Utf8View(""), hits_raw.URL LIKE Utf8View("%google%")] @@ -599,9 +599,9 @@ physical_plan 01)SortPreservingMergeExec: [c@2 DESC], fetch=10 02)--ProjectionExec: expr=[SearchPhrase@0 as SearchPhrase, min(hits.URL)@1 as min(hits.URL), count(Int64(1))@2 as c] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@2 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[SearchPhrase@0 as SearchPhrase], aggr=[min(hits.URL), count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[SearchPhrase@0 as SearchPhrase], aggr=[min(hits.URL), count() as count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([SearchPhrase@0], 4), input_partitions=4 -06)----------AggregateExec: mode=Partial, gby=[SearchPhrase@1 as SearchPhrase], aggr=[min(hits.URL), count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[SearchPhrase@1 as SearchPhrase], aggr=[min(hits.URL), count() as count(Int64(1))] 07)------------FilterExec: SearchPhrase@1 != AND URL@0 LIKE %google% 08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[URL, SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != AND URL@13 LIKE %google%, pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] @@ -617,7 +617,7 @@ EXPLAIN SELECT "SearchPhrase", MIN("URL"), MIN("Title"), COUNT(*) AS c, COUNT(DI logical_plan 01)Sort: c DESC NULLS FIRST, fetch=10 02)--Projection: hits.SearchPhrase, min(hits.URL), min(hits.Title), count(Int64(1)) AS count(*) AS c, count(DISTINCT hits.UserID) -03)----Aggregate: groupBy=[[hits.SearchPhrase]], aggr=[[min(hits.URL), min(hits.Title), count(Int64(1)), count(DISTINCT hits.UserID)]] +03)----Aggregate: groupBy=[[hits.SearchPhrase]], aggr=[[min(hits.URL), min(hits.Title), count() AS count(Int64(1)), count(DISTINCT hits.UserID)]] 04)------SubqueryAlias: hits 05)--------Filter: hits_raw.SearchPhrase != Utf8View("") AND hits_raw.Title LIKE Utf8View("%Google%") AND hits_raw.URL NOT LIKE Utf8View("%.google.%") 06)----------TableScan: hits_raw projection=[Title, UserID, URL, SearchPhrase], partial_filters=[hits_raw.SearchPhrase != Utf8View(""), hits_raw.Title LIKE Utf8View("%Google%"), hits_raw.URL NOT LIKE Utf8View("%.google.%")] @@ -625,9 +625,9 @@ physical_plan 01)SortPreservingMergeExec: [c@3 DESC], fetch=10 02)--ProjectionExec: expr=[SearchPhrase@0 as SearchPhrase, min(hits.URL)@1 as min(hits.URL), min(hits.Title)@2 as min(hits.Title), count(Int64(1))@3 as c, count(DISTINCT hits.UserID)@4 as count(DISTINCT hits.UserID)] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@3 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[SearchPhrase@0 as SearchPhrase], aggr=[min(hits.URL), min(hits.Title), count(Int64(1)), count(DISTINCT hits.UserID)] +04)------AggregateExec: mode=FinalPartitioned, gby=[SearchPhrase@0 as SearchPhrase], aggr=[min(hits.URL), min(hits.Title), count() as count(Int64(1)), count(DISTINCT hits.UserID)] 05)--------RepartitionExec: partitioning=Hash([SearchPhrase@0], 4), input_partitions=4 -06)----------AggregateExec: mode=Partial, gby=[SearchPhrase@3 as SearchPhrase], aggr=[min(hits.URL), min(hits.Title), count(Int64(1)), count(DISTINCT hits.UserID)] +06)----------AggregateExec: mode=Partial, gby=[SearchPhrase@3 as SearchPhrase], aggr=[min(hits.URL), min(hits.Title), count() as count(Int64(1)), count(DISTINCT hits.UserID)] 07)------------FilterExec: SearchPhrase@3 != AND Title@0 LIKE %Google% AND URL@2 NOT LIKE %.google.% 08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[Title, UserID, URL, SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != AND Title@2 LIKE %Google% AND URL@13 NOT LIKE %.google.%, pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] @@ -734,7 +734,7 @@ logical_plan 01)Sort: l DESC NULLS FIRST, fetch=25 02)--Projection: hits.CounterID, avg(octet_length(hits.URL)) AS l, count(Int64(1)) AS count(*) AS c 03)----Filter: count(Int64(1)) > Int64(100000) -04)------Aggregate: groupBy=[[hits.CounterID]], aggr=[[avg(CAST(octet_length(hits.URL) AS Float64)), count(Int64(1))]] +04)------Aggregate: groupBy=[[hits.CounterID]], aggr=[[avg(CAST(octet_length(hits.URL) AS Float64)), count() AS count(Int64(1))]] 05)--------SubqueryAlias: hits 06)----------Filter: hits_raw.URL != Utf8View("") 07)------------TableScan: hits_raw projection=[CounterID, URL], partial_filters=[hits_raw.URL != Utf8View("")] @@ -743,9 +743,9 @@ physical_plan 02)--ProjectionExec: expr=[CounterID@0 as CounterID, avg(octet_length(hits.URL))@1 as l, count(Int64(1))@2 as c] 03)----SortExec: TopK(fetch=25), expr=[avg(octet_length(hits.URL))@1 DESC], preserve_partitioning=[true] 04)------FilterExec: count(Int64(1))@2 > 100000 -05)--------AggregateExec: mode=FinalPartitioned, gby=[CounterID@0 as CounterID], aggr=[avg(octet_length(hits.URL)), count(Int64(1))] +05)--------AggregateExec: mode=FinalPartitioned, gby=[CounterID@0 as CounterID], aggr=[avg(octet_length(hits.URL)), count() as count(Int64(1))] 06)----------RepartitionExec: partitioning=Hash([CounterID@0], 4), input_partitions=4 -07)------------AggregateExec: mode=Partial, gby=[CounterID@0 as CounterID], aggr=[avg(octet_length(hits.URL)), count(Int64(1))] +07)------------AggregateExec: mode=Partial, gby=[CounterID@0 as CounterID], aggr=[avg(octet_length(hits.URL)), count() as count(Int64(1))] 08)--------------FilterExec: URL@1 != 09)----------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 10)------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[CounterID, URL], file_type=parquet, predicate=URL@13 != , pruning_predicate=URL_null_count@2 != row_count@3 AND (URL_min@0 != OR != URL_max@1), required_guarantees=[URL not in ()] @@ -762,7 +762,7 @@ logical_plan 01)Sort: l DESC NULLS FIRST, fetch=25 02)--Projection: regexp_replace(hits.Referer,Utf8("^https?://(?:www\.)?([^/]+)/.*$"),Utf8("\1")) AS k, avg(octet_length(hits.Referer)) AS l, count(Int64(1)) AS count(*) AS c, min(hits.Referer) 03)----Filter: count(Int64(1)) > Int64(100000) -04)------Aggregate: groupBy=[[regexp_replace(hits.Referer, Utf8View("^https?://(?:www\.)?([^/]+)/.*$"), Utf8View("\1")) AS regexp_replace(hits.Referer,Utf8("^https?://(?:www\.)?([^/]+)/.*$"),Utf8("\1"))]], aggr=[[avg(CAST(octet_length(hits.Referer) AS Float64)), count(Int64(1)), min(hits.Referer)]] +04)------Aggregate: groupBy=[[regexp_replace(hits.Referer, Utf8View("^https?://(?:www\.)?([^/]+)/.*$"), Utf8View("\1")) AS regexp_replace(hits.Referer,Utf8("^https?://(?:www\.)?([^/]+)/.*$"),Utf8("\1"))]], aggr=[[avg(CAST(octet_length(hits.Referer) AS Float64)), count() AS count(Int64(1)), min(hits.Referer)]] 05)--------SubqueryAlias: hits 06)----------Filter: hits_raw.Referer != Utf8View("") 07)------------TableScan: hits_raw projection=[Referer], partial_filters=[hits_raw.Referer != Utf8View("")] @@ -771,9 +771,9 @@ physical_plan 02)--ProjectionExec: expr=[regexp_replace(hits.Referer,Utf8("^https?://(?:www\.)?([^/]+)/.*$"),Utf8("\1"))@0 as k, avg(octet_length(hits.Referer))@1 as l, count(Int64(1))@2 as c, min(hits.Referer)@3 as min(hits.Referer)] 03)----SortExec: TopK(fetch=25), expr=[avg(octet_length(hits.Referer))@1 DESC], preserve_partitioning=[true] 04)------FilterExec: count(Int64(1))@2 > 100000 -05)--------AggregateExec: mode=FinalPartitioned, gby=[regexp_replace(hits.Referer,Utf8("^https?://(?:www\.)?([^/]+)/.*$"),Utf8("\1"))@0 as regexp_replace(hits.Referer,Utf8("^https?://(?:www\.)?([^/]+)/.*$"),Utf8("\1"))], aggr=[avg(octet_length(hits.Referer)), count(Int64(1)), min(hits.Referer)] +05)--------AggregateExec: mode=FinalPartitioned, gby=[regexp_replace(hits.Referer,Utf8("^https?://(?:www\.)?([^/]+)/.*$"),Utf8("\1"))@0 as regexp_replace(hits.Referer,Utf8("^https?://(?:www\.)?([^/]+)/.*$"),Utf8("\1"))], aggr=[avg(octet_length(hits.Referer)), count() as count(Int64(1)), min(hits.Referer)] 06)----------RepartitionExec: partitioning=Hash([regexp_replace(hits.Referer,Utf8("^https?://(?:www\.)?([^/]+)/.*$"),Utf8("\1"))@0], 4), input_partitions=4 -07)------------AggregateExec: mode=Partial, gby=[regexp_replace(Referer@0, ^https?://(?:www\.)?([^/]+)/.*$, \1) as regexp_replace(hits.Referer,Utf8("^https?://(?:www\.)?([^/]+)/.*$"),Utf8("\1"))], aggr=[avg(octet_length(hits.Referer)), count(Int64(1)), min(hits.Referer)] +07)------------AggregateExec: mode=Partial, gby=[regexp_replace(Referer@0, ^https?://(?:www\.)?([^/]+)/.*$, \1) as regexp_replace(hits.Referer,Utf8("^https?://(?:www\.)?([^/]+)/.*$"),Utf8("\1"))], aggr=[avg(octet_length(hits.Referer)), count() as count(Int64(1)), min(hits.Referer)] 08)--------------FilterExec: Referer@0 != 09)----------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 10)------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[Referer], file_type=parquet, predicate=Referer@14 != , pruning_predicate=Referer_null_count@2 != row_count@3 AND (Referer_min@0 != OR != Referer_max@1), required_guarantees=[Referer not in ()] @@ -810,7 +810,7 @@ EXPLAIN SELECT "SearchEngineID", "ClientIP", COUNT(*) AS c, SUM("IsRefresh"), AV logical_plan 01)Sort: c DESC NULLS FIRST, fetch=10 02)--Projection: hits.SearchEngineID, hits.ClientIP, count(Int64(1)) AS count(*) AS c, sum(hits.IsRefresh), avg(hits.ResolutionWidth) -03)----Aggregate: groupBy=[[hits.SearchEngineID, hits.ClientIP]], aggr=[[count(Int64(1)), sum(CAST(hits.IsRefresh AS Int64)), avg(CAST(hits.ResolutionWidth AS Float64))]] +03)----Aggregate: groupBy=[[hits.SearchEngineID, hits.ClientIP]], aggr=[[count() AS count(Int64(1)), sum(CAST(hits.IsRefresh AS Int64)), avg(CAST(hits.ResolutionWidth AS Float64))]] 04)------SubqueryAlias: hits 05)--------Projection: hits_raw.ClientIP, hits_raw.IsRefresh, hits_raw.ResolutionWidth, hits_raw.SearchEngineID 06)----------Filter: hits_raw.SearchPhrase != Utf8View("") @@ -819,9 +819,9 @@ physical_plan 01)SortPreservingMergeExec: [c@2 DESC], fetch=10 02)--ProjectionExec: expr=[SearchEngineID@0 as SearchEngineID, ClientIP@1 as ClientIP, count(Int64(1))@2 as c, sum(hits.IsRefresh)@3 as sum(hits.IsRefresh), avg(hits.ResolutionWidth)@4 as avg(hits.ResolutionWidth)] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@2 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[SearchEngineID@0 as SearchEngineID, ClientIP@1 as ClientIP], aggr=[count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] +04)------AggregateExec: mode=FinalPartitioned, gby=[SearchEngineID@0 as SearchEngineID, ClientIP@1 as ClientIP], aggr=[count() as count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] 05)--------RepartitionExec: partitioning=Hash([SearchEngineID@0, ClientIP@1], 4), input_partitions=4 -06)----------AggregateExec: mode=Partial, gby=[SearchEngineID@3 as SearchEngineID, ClientIP@0 as ClientIP], aggr=[count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] +06)----------AggregateExec: mode=Partial, gby=[SearchEngineID@3 as SearchEngineID, ClientIP@0 as ClientIP], aggr=[count() as count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] 07)------------FilterExec: SearchPhrase@4 != , projection=[ClientIP@0, IsRefresh@1, ResolutionWidth@2, SearchEngineID@3] 08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[ClientIP, IsRefresh, ResolutionWidth, SearchEngineID, SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != , pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] @@ -837,7 +837,7 @@ EXPLAIN SELECT "WatchID", "ClientIP", COUNT(*) AS c, SUM("IsRefresh"), AVG("Reso logical_plan 01)Sort: c DESC NULLS FIRST, fetch=10 02)--Projection: hits.WatchID, hits.ClientIP, count(Int64(1)) AS count(*) AS c, sum(hits.IsRefresh), avg(hits.ResolutionWidth) -03)----Aggregate: groupBy=[[hits.WatchID, hits.ClientIP]], aggr=[[count(Int64(1)), sum(CAST(hits.IsRefresh AS Int64)), avg(CAST(hits.ResolutionWidth AS Float64))]] +03)----Aggregate: groupBy=[[hits.WatchID, hits.ClientIP]], aggr=[[count() AS count(Int64(1)), sum(CAST(hits.IsRefresh AS Int64)), avg(CAST(hits.ResolutionWidth AS Float64))]] 04)------SubqueryAlias: hits 05)--------Projection: hits_raw.WatchID, hits_raw.ClientIP, hits_raw.IsRefresh, hits_raw.ResolutionWidth 06)----------Filter: hits_raw.SearchPhrase != Utf8View("") @@ -846,9 +846,9 @@ physical_plan 01)SortPreservingMergeExec: [c@2 DESC], fetch=10 02)--ProjectionExec: expr=[WatchID@0 as WatchID, ClientIP@1 as ClientIP, count(Int64(1))@2 as c, sum(hits.IsRefresh)@3 as sum(hits.IsRefresh), avg(hits.ResolutionWidth)@4 as avg(hits.ResolutionWidth)] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@2 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[WatchID@0 as WatchID, ClientIP@1 as ClientIP], aggr=[count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] +04)------AggregateExec: mode=FinalPartitioned, gby=[WatchID@0 as WatchID, ClientIP@1 as ClientIP], aggr=[count() as count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] 05)--------RepartitionExec: partitioning=Hash([WatchID@0, ClientIP@1], 4), input_partitions=4 -06)----------AggregateExec: mode=Partial, gby=[WatchID@0 as WatchID, ClientIP@1 as ClientIP], aggr=[count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] +06)----------AggregateExec: mode=Partial, gby=[WatchID@0 as WatchID, ClientIP@1 as ClientIP], aggr=[count() as count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] 07)------------FilterExec: SearchPhrase@4 != , projection=[WatchID@0, ClientIP@1, IsRefresh@2, ResolutionWidth@3] 08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[WatchID, ClientIP, IsRefresh, ResolutionWidth, SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != , pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] @@ -864,16 +864,16 @@ EXPLAIN SELECT "WatchID", "ClientIP", COUNT(*) AS c, SUM("IsRefresh"), AVG("Reso logical_plan 01)Sort: c DESC NULLS FIRST, fetch=10 02)--Projection: hits.WatchID, hits.ClientIP, count(Int64(1)) AS count(*) AS c, sum(hits.IsRefresh), avg(hits.ResolutionWidth) -03)----Aggregate: groupBy=[[hits.WatchID, hits.ClientIP]], aggr=[[count(Int64(1)), sum(CAST(hits.IsRefresh AS Int64)), avg(CAST(hits.ResolutionWidth AS Float64))]] +03)----Aggregate: groupBy=[[hits.WatchID, hits.ClientIP]], aggr=[[count() AS count(Int64(1)), sum(CAST(hits.IsRefresh AS Int64)), avg(CAST(hits.ResolutionWidth AS Float64))]] 04)------SubqueryAlias: hits 05)--------TableScan: hits_raw projection=[WatchID, ClientIP, IsRefresh, ResolutionWidth] physical_plan 01)SortPreservingMergeExec: [c@2 DESC], fetch=10 02)--ProjectionExec: expr=[WatchID@0 as WatchID, ClientIP@1 as ClientIP, count(Int64(1))@2 as c, sum(hits.IsRefresh)@3 as sum(hits.IsRefresh), avg(hits.ResolutionWidth)@4 as avg(hits.ResolutionWidth)] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@2 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[WatchID@0 as WatchID, ClientIP@1 as ClientIP], aggr=[count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] +04)------AggregateExec: mode=FinalPartitioned, gby=[WatchID@0 as WatchID, ClientIP@1 as ClientIP], aggr=[count() as count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] 05)--------RepartitionExec: partitioning=Hash([WatchID@0, ClientIP@1], 4), input_partitions=1 -06)----------AggregateExec: mode=Partial, gby=[WatchID@0 as WatchID, ClientIP@1 as ClientIP], aggr=[count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] +06)----------AggregateExec: mode=Partial, gby=[WatchID@0 as WatchID, ClientIP@1 as ClientIP], aggr=[count() as count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] 07)------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[WatchID, ClientIP, IsRefresh, ResolutionWidth], file_type=parquet query IIIIR rowsort @@ -897,16 +897,16 @@ EXPLAIN SELECT "URL", COUNT(*) AS c FROM hits GROUP BY "URL" ORDER BY c DESC LIM logical_plan 01)Sort: c DESC NULLS FIRST, fetch=10 02)--Projection: hits.URL, count(Int64(1)) AS count(*) AS c -03)----Aggregate: groupBy=[[hits.URL]], aggr=[[count(Int64(1))]] +03)----Aggregate: groupBy=[[hits.URL]], aggr=[[count() AS count(Int64(1))]] 04)------SubqueryAlias: hits 05)--------TableScan: hits_raw projection=[URL] physical_plan 01)SortPreservingMergeExec: [c@1 DESC], fetch=10 02)--ProjectionExec: expr=[URL@0 as URL, count(Int64(1))@1 as c] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@1 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[URL@0 as URL], aggr=[count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[URL@0 as URL], aggr=[count() as count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([URL@0], 4), input_partitions=1 -06)----------AggregateExec: mode=Partial, gby=[URL@0 as URL], aggr=[count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[URL@0 as URL], aggr=[count() as count(Int64(1))] 07)------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[URL], file_type=parquet query TI rowsort @@ -926,16 +926,16 @@ EXPLAIN SELECT 1, "URL", COUNT(*) AS c FROM hits GROUP BY 1, "URL" ORDER BY c DE logical_plan 01)Sort: c DESC NULLS FIRST, fetch=10 02)--Projection: Int64(1), hits.URL, count(Int64(1)) AS c -03)----Aggregate: groupBy=[[hits.URL]], aggr=[[count(Int64(1))]] +03)----Aggregate: groupBy=[[hits.URL]], aggr=[[count() AS count(Int64(1))]] 04)------SubqueryAlias: hits 05)--------TableScan: hits_raw projection=[URL] physical_plan 01)SortPreservingMergeExec: [c@2 DESC], fetch=10 02)--SortExec: TopK(fetch=10), expr=[c@2 DESC], preserve_partitioning=[true] 03)----ProjectionExec: expr=[1 as Int64(1), URL@0 as URL, count(Int64(1))@1 as c] -04)------AggregateExec: mode=FinalPartitioned, gby=[URL@0 as URL], aggr=[count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[URL@0 as URL], aggr=[count() as count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([URL@0], 4), input_partitions=1 -06)----------AggregateExec: mode=Partial, gby=[URL@0 as URL], aggr=[count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[URL@0 as URL], aggr=[count() as count(Int64(1))] 07)------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[URL], file_type=parquet query ITI rowsort @@ -956,7 +956,7 @@ logical_plan 01)Sort: c DESC NULLS FIRST, fetch=10 02)--Projection: hits.ClientIP, __common_expr_1 - Int64(1) AS hits.ClientIP - Int64(1), __common_expr_1 - Int64(2) AS hits.ClientIP - Int64(2), __common_expr_1 - Int64(3) AS hits.ClientIP - Int64(3), count(Int64(1)) AS c 03)----Projection: CAST(hits.ClientIP AS Int64) AS __common_expr_1, hits.ClientIP, count(Int64(1)) -04)------Aggregate: groupBy=[[hits.ClientIP]], aggr=[[count(Int64(1))]] +04)------Aggregate: groupBy=[[hits.ClientIP]], aggr=[[count() AS count(Int64(1))]] 05)--------SubqueryAlias: hits 06)----------TableScan: hits_raw projection=[ClientIP] physical_plan @@ -964,9 +964,9 @@ physical_plan 02)--SortExec: TopK(fetch=10), expr=[c@4 DESC], preserve_partitioning=[true] 03)----ProjectionExec: expr=[ClientIP@1 as ClientIP, __common_expr_1@0 - 1 as hits.ClientIP - Int64(1), __common_expr_1@0 - 2 as hits.ClientIP - Int64(2), __common_expr_1@0 - 3 as hits.ClientIP - Int64(3), count(Int64(1))@2 as c] 04)------ProjectionExec: expr=[CAST(ClientIP@0 AS Int64) as __common_expr_1, ClientIP@0 as ClientIP, count(Int64(1))@1 as count(Int64(1))] -05)--------AggregateExec: mode=FinalPartitioned, gby=[ClientIP@0 as ClientIP], aggr=[count(Int64(1))] +05)--------AggregateExec: mode=FinalPartitioned, gby=[ClientIP@0 as ClientIP], aggr=[count() as count(Int64(1))] 06)----------RepartitionExec: partitioning=Hash([ClientIP@0], 4), input_partitions=1 -07)------------AggregateExec: mode=Partial, gby=[ClientIP@0 as ClientIP], aggr=[count(Int64(1))] +07)------------AggregateExec: mode=Partial, gby=[ClientIP@0 as ClientIP], aggr=[count() as count(Int64(1))] 08)--------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[ClientIP], file_type=parquet query IIIII rowsort @@ -984,7 +984,7 @@ EXPLAIN SELECT "URL", COUNT(*) AS PageViews FROM hits WHERE "CounterID" = 62 AND logical_plan 01)Sort: pageviews DESC NULLS FIRST, fetch=10 02)--Projection: hits.URL, count(Int64(1)) AS count(*) AS pageviews -03)----Aggregate: groupBy=[[hits.URL]], aggr=[[count(Int64(1))]] +03)----Aggregate: groupBy=[[hits.URL]], aggr=[[count() AS count(Int64(1))]] 04)------SubqueryAlias: hits 05)--------Projection: hits_raw.URL 06)----------Filter: hits_raw.CounterID = Int32(62) AND hits_raw.EventDate >= UInt16(15887) AND hits_raw.EventDate <= UInt16(15917) AND hits_raw.DontCountHits = Int16(0) AND hits_raw.IsRefresh = Int16(0) AND hits_raw.URL != Utf8View("") @@ -993,9 +993,9 @@ physical_plan 01)SortPreservingMergeExec: [pageviews@1 DESC], fetch=10 02)--ProjectionExec: expr=[URL@0 as URL, count(Int64(1))@1 as pageviews] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@1 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[URL@0 as URL], aggr=[count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[URL@0 as URL], aggr=[count() as count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([URL@0], 4), input_partitions=4 -06)----------AggregateExec: mode=Partial, gby=[URL@0 as URL], aggr=[count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[URL@0 as URL], aggr=[count() as count(Int64(1))] 07)------------FilterExec: CounterID@1 = 62 AND EventDate@0 >= 15887 AND EventDate@0 <= 15917 AND DontCountHits@4 = 0 AND IsRefresh@3 = 0 AND URL@2 != , projection=[URL@2] 08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[EventDate, CounterID, URL, IsRefresh, DontCountHits], file_type=parquet, predicate=CounterID@6 = 62 AND EventDate@5 >= 15887 AND EventDate@5 <= 15917 AND DontCountHits@61 = 0 AND IsRefresh@15 = 0 AND URL@13 != , pruning_predicate=CounterID_null_count@2 != row_count@3 AND CounterID_min@0 <= 62 AND 62 <= CounterID_max@1 AND EventDate_null_count@5 != row_count@3 AND EventDate_max@4 >= 15887 AND EventDate_null_count@5 != row_count@3 AND EventDate_min@6 <= 15917 AND DontCountHits_null_count@9 != row_count@3 AND DontCountHits_min@7 <= 0 AND 0 <= DontCountHits_max@8 AND IsRefresh_null_count@12 != row_count@3 AND IsRefresh_min@10 <= 0 AND 0 <= IsRefresh_max@11 AND URL_null_count@15 != row_count@3 AND (URL_min@13 != OR != URL_max@14), required_guarantees=[CounterID in (62), DontCountHits in (0), IsRefresh in (0), URL not in ()] @@ -1011,7 +1011,7 @@ EXPLAIN SELECT "Title", COUNT(*) AS PageViews FROM hits WHERE "CounterID" = 62 A logical_plan 01)Sort: pageviews DESC NULLS FIRST, fetch=10 02)--Projection: hits.Title, count(Int64(1)) AS count(*) AS pageviews -03)----Aggregate: groupBy=[[hits.Title]], aggr=[[count(Int64(1))]] +03)----Aggregate: groupBy=[[hits.Title]], aggr=[[count() AS count(Int64(1))]] 04)------SubqueryAlias: hits 05)--------Projection: hits_raw.Title 06)----------Filter: hits_raw.CounterID = Int32(62) AND hits_raw.EventDate >= UInt16(15887) AND hits_raw.EventDate <= UInt16(15917) AND hits_raw.DontCountHits = Int16(0) AND hits_raw.IsRefresh = Int16(0) AND hits_raw.Title != Utf8View("") @@ -1020,9 +1020,9 @@ physical_plan 01)SortPreservingMergeExec: [pageviews@1 DESC], fetch=10 02)--ProjectionExec: expr=[Title@0 as Title, count(Int64(1))@1 as pageviews] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@1 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[Title@0 as Title], aggr=[count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[Title@0 as Title], aggr=[count() as count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([Title@0], 4), input_partitions=4 -06)----------AggregateExec: mode=Partial, gby=[Title@0 as Title], aggr=[count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[Title@0 as Title], aggr=[count() as count(Int64(1))] 07)------------FilterExec: CounterID@2 = 62 AND EventDate@1 >= 15887 AND EventDate@1 <= 15917 AND DontCountHits@4 = 0 AND IsRefresh@3 = 0 AND Title@0 != , projection=[Title@0] 08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[Title, EventDate, CounterID, IsRefresh, DontCountHits], file_type=parquet, predicate=CounterID@6 = 62 AND EventDate@5 >= 15887 AND EventDate@5 <= 15917 AND DontCountHits@61 = 0 AND IsRefresh@15 = 0 AND Title@2 != , pruning_predicate=CounterID_null_count@2 != row_count@3 AND CounterID_min@0 <= 62 AND 62 <= CounterID_max@1 AND EventDate_null_count@5 != row_count@3 AND EventDate_max@4 >= 15887 AND EventDate_null_count@5 != row_count@3 AND EventDate_min@6 <= 15917 AND DontCountHits_null_count@9 != row_count@3 AND DontCountHits_min@7 <= 0 AND 0 <= DontCountHits_max@8 AND IsRefresh_null_count@12 != row_count@3 AND IsRefresh_min@10 <= 0 AND 0 <= IsRefresh_max@11 AND Title_null_count@15 != row_count@3 AND (Title_min@13 != OR != Title_max@14), required_guarantees=[CounterID in (62), DontCountHits in (0), IsRefresh in (0), Title not in ()] @@ -1039,7 +1039,7 @@ logical_plan 01)Limit: skip=1000, fetch=10 02)--Sort: pageviews DESC NULLS FIRST, fetch=1010 03)----Projection: hits.URL, count(Int64(1)) AS count(*) AS pageviews -04)------Aggregate: groupBy=[[hits.URL]], aggr=[[count(Int64(1))]] +04)------Aggregate: groupBy=[[hits.URL]], aggr=[[count() AS count(Int64(1))]] 05)--------SubqueryAlias: hits 06)----------Projection: hits_raw.URL 07)------------Filter: hits_raw.CounterID = Int32(62) AND hits_raw.EventDate >= UInt16(15887) AND hits_raw.EventDate <= UInt16(15917) AND hits_raw.IsRefresh = Int16(0) AND hits_raw.IsLink != Int16(0) AND hits_raw.IsDownload = Int16(0) @@ -1049,9 +1049,9 @@ physical_plan 02)--SortPreservingMergeExec: [pageviews@1 DESC], fetch=1010 03)----ProjectionExec: expr=[URL@0 as URL, count(Int64(1))@1 as pageviews] 04)------SortExec: TopK(fetch=1010), expr=[count(Int64(1))@1 DESC], preserve_partitioning=[true] -05)--------AggregateExec: mode=FinalPartitioned, gby=[URL@0 as URL], aggr=[count(Int64(1))] +05)--------AggregateExec: mode=FinalPartitioned, gby=[URL@0 as URL], aggr=[count() as count(Int64(1))] 06)----------RepartitionExec: partitioning=Hash([URL@0], 4), input_partitions=4 -07)------------AggregateExec: mode=Partial, gby=[URL@0 as URL], aggr=[count(Int64(1))] +07)------------AggregateExec: mode=Partial, gby=[URL@0 as URL], aggr=[count() as count(Int64(1))] 08)--------------FilterExec: CounterID@1 = 62 AND EventDate@0 >= 15887 AND EventDate@0 <= 15917 AND IsRefresh@3 = 0 AND IsLink@4 != 0 AND IsDownload@5 = 0, projection=[URL@2] 09)----------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 10)------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[EventDate, CounterID, URL, IsRefresh, IsLink, IsDownload], file_type=parquet, predicate=CounterID@6 = 62 AND EventDate@5 >= 15887 AND EventDate@5 <= 15917 AND IsRefresh@15 = 0 AND IsLink@52 != 0 AND IsDownload@53 = 0, pruning_predicate=CounterID_null_count@2 != row_count@3 AND CounterID_min@0 <= 62 AND 62 <= CounterID_max@1 AND EventDate_null_count@5 != row_count@3 AND EventDate_max@4 >= 15887 AND EventDate_null_count@5 != row_count@3 AND EventDate_min@6 <= 15917 AND IsRefresh_null_count@9 != row_count@3 AND IsRefresh_min@7 <= 0 AND 0 <= IsRefresh_max@8 AND IsLink_null_count@12 != row_count@3 AND (IsLink_min@10 != 0 OR 0 != IsLink_max@11) AND IsDownload_null_count@15 != row_count@3 AND IsDownload_min@13 <= 0 AND 0 <= IsDownload_max@14, required_guarantees=[CounterID in (62), IsDownload in (0), IsLink not in (0), IsRefresh in (0)] @@ -1068,7 +1068,7 @@ logical_plan 01)Limit: skip=1000, fetch=10 02)--Sort: pageviews DESC NULLS FIRST, fetch=1010 03)----Projection: hits.TraficSourceID, hits.SearchEngineID, hits.AdvEngineID, CASE WHEN hits.SearchEngineID = Int64(0) AND hits.AdvEngineID = Int64(0) THEN hits.Referer ELSE Utf8("") END AS src, hits.URL AS dst, count(Int64(1)) AS count(*) AS pageviews -04)------Aggregate: groupBy=[[hits.TraficSourceID, hits.SearchEngineID, hits.AdvEngineID, CASE WHEN hits.SearchEngineID = Int16(0) AND hits.AdvEngineID = Int16(0) THEN hits.Referer ELSE Utf8View("") END AS CASE WHEN hits.SearchEngineID = Int64(0) AND hits.AdvEngineID = Int64(0) THEN hits.Referer ELSE Utf8("") END, hits.URL]], aggr=[[count(Int64(1))]] +04)------Aggregate: groupBy=[[hits.TraficSourceID, hits.SearchEngineID, hits.AdvEngineID, CASE WHEN hits.SearchEngineID = Int16(0) AND hits.AdvEngineID = Int16(0) THEN hits.Referer ELSE Utf8View("") END AS CASE WHEN hits.SearchEngineID = Int64(0) AND hits.AdvEngineID = Int64(0) THEN hits.Referer ELSE Utf8("") END, hits.URL]], aggr=[[count() AS count(Int64(1))]] 05)--------SubqueryAlias: hits 06)----------Projection: hits_raw.URL, hits_raw.Referer, hits_raw.TraficSourceID, hits_raw.SearchEngineID, hits_raw.AdvEngineID 07)------------Filter: hits_raw.CounterID = Int32(62) AND hits_raw.EventDate >= UInt16(15887) AND hits_raw.EventDate <= UInt16(15917) AND hits_raw.IsRefresh = Int16(0) @@ -1078,9 +1078,9 @@ physical_plan 02)--SortPreservingMergeExec: [pageviews@5 DESC], fetch=1010 03)----ProjectionExec: expr=[TraficSourceID@0 as TraficSourceID, SearchEngineID@1 as SearchEngineID, AdvEngineID@2 as AdvEngineID, CASE WHEN hits.SearchEngineID = Int64(0) AND hits.AdvEngineID = Int64(0) THEN hits.Referer ELSE Utf8("") END@3 as src, URL@4 as dst, count(Int64(1))@5 as pageviews] 04)------SortExec: TopK(fetch=1010), expr=[count(Int64(1))@5 DESC], preserve_partitioning=[true] -05)--------AggregateExec: mode=FinalPartitioned, gby=[TraficSourceID@0 as TraficSourceID, SearchEngineID@1 as SearchEngineID, AdvEngineID@2 as AdvEngineID, CASE WHEN hits.SearchEngineID = Int64(0) AND hits.AdvEngineID = Int64(0) THEN hits.Referer ELSE Utf8("") END@3 as CASE WHEN hits.SearchEngineID = Int64(0) AND hits.AdvEngineID = Int64(0) THEN hits.Referer ELSE Utf8("") END, URL@4 as URL], aggr=[count(Int64(1))] +05)--------AggregateExec: mode=FinalPartitioned, gby=[TraficSourceID@0 as TraficSourceID, SearchEngineID@1 as SearchEngineID, AdvEngineID@2 as AdvEngineID, CASE WHEN hits.SearchEngineID = Int64(0) AND hits.AdvEngineID = Int64(0) THEN hits.Referer ELSE Utf8("") END@3 as CASE WHEN hits.SearchEngineID = Int64(0) AND hits.AdvEngineID = Int64(0) THEN hits.Referer ELSE Utf8("") END, URL@4 as URL], aggr=[count() as count(Int64(1))] 06)----------RepartitionExec: partitioning=Hash([TraficSourceID@0, SearchEngineID@1, AdvEngineID@2, CASE WHEN hits.SearchEngineID = Int64(0) AND hits.AdvEngineID = Int64(0) THEN hits.Referer ELSE Utf8("") END@3, URL@4], 4), input_partitions=4 -07)------------AggregateExec: mode=Partial, gby=[TraficSourceID@2 as TraficSourceID, SearchEngineID@3 as SearchEngineID, AdvEngineID@4 as AdvEngineID, CASE WHEN SearchEngineID@3 = 0 AND AdvEngineID@4 = 0 THEN Referer@1 ELSE END as CASE WHEN hits.SearchEngineID = Int64(0) AND hits.AdvEngineID = Int64(0) THEN hits.Referer ELSE Utf8("") END, URL@0 as URL], aggr=[count(Int64(1))] +07)------------AggregateExec: mode=Partial, gby=[TraficSourceID@2 as TraficSourceID, SearchEngineID@3 as SearchEngineID, AdvEngineID@4 as AdvEngineID, CASE WHEN SearchEngineID@3 = 0 AND AdvEngineID@4 = 0 THEN Referer@1 ELSE END as CASE WHEN hits.SearchEngineID = Int64(0) AND hits.AdvEngineID = Int64(0) THEN hits.Referer ELSE Utf8("") END, URL@0 as URL], aggr=[count() as count(Int64(1))] 08)--------------FilterExec: CounterID@1 = 62 AND EventDate@0 >= 15887 AND EventDate@0 <= 15917 AND IsRefresh@4 = 0, projection=[URL@2, Referer@3, TraficSourceID@5, SearchEngineID@6, AdvEngineID@7] 09)----------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 10)------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[EventDate, CounterID, URL, Referer, IsRefresh, TraficSourceID, SearchEngineID, AdvEngineID], file_type=parquet, predicate=CounterID@6 = 62 AND EventDate@5 >= 15887 AND EventDate@5 <= 15917 AND IsRefresh@15 = 0, pruning_predicate=CounterID_null_count@2 != row_count@3 AND CounterID_min@0 <= 62 AND 62 <= CounterID_max@1 AND EventDate_null_count@5 != row_count@3 AND EventDate_max@4 >= 15887 AND EventDate_null_count@5 != row_count@3 AND EventDate_min@6 <= 15917 AND IsRefresh_null_count@9 != row_count@3 AND IsRefresh_min@7 <= 0 AND 0 <= IsRefresh_max@8, required_guarantees=[CounterID in (62), IsRefresh in (0)] @@ -1097,7 +1097,7 @@ logical_plan 01)Limit: skip=100, fetch=10 02)--Sort: pageviews DESC NULLS FIRST, fetch=110 03)----Projection: hits.URLHash, hits.EventDate, count(Int64(1)) AS count(*) AS pageviews -04)------Aggregate: groupBy=[[hits.URLHash, hits.EventDate]], aggr=[[count(Int64(1))]] +04)------Aggregate: groupBy=[[hits.URLHash, hits.EventDate]], aggr=[[count() AS count(Int64(1))]] 05)--------SubqueryAlias: hits 06)----------Projection: hits_raw.URLHash, CAST(CAST(hits_raw.EventDate AS Int32) AS Date32) AS EventDate 07)------------Filter: hits_raw.CounterID = Int32(62) AND hits_raw.EventDate >= UInt16(15887) AND hits_raw.EventDate <= UInt16(15917) AND hits_raw.IsRefresh = Int16(0) AND (hits_raw.TraficSourceID = Int16(-1) OR hits_raw.TraficSourceID = Int16(6)) AND hits_raw.RefererHash = Int64(3594120000172545465) @@ -1107,9 +1107,9 @@ physical_plan 02)--SortPreservingMergeExec: [pageviews@2 DESC], fetch=110 03)----ProjectionExec: expr=[URLHash@0 as URLHash, EventDate@1 as EventDate, count(Int64(1))@2 as pageviews] 04)------SortExec: TopK(fetch=110), expr=[count(Int64(1))@2 DESC], preserve_partitioning=[true] -05)--------AggregateExec: mode=FinalPartitioned, gby=[URLHash@0 as URLHash, EventDate@1 as EventDate], aggr=[count(Int64(1))] +05)--------AggregateExec: mode=FinalPartitioned, gby=[URLHash@0 as URLHash, EventDate@1 as EventDate], aggr=[count() as count(Int64(1))] 06)----------RepartitionExec: partitioning=Hash([URLHash@0, EventDate@1], 4), input_partitions=4 -07)------------AggregateExec: mode=Partial, gby=[URLHash@0 as URLHash, EventDate@1 as EventDate], aggr=[count(Int64(1))] +07)------------AggregateExec: mode=Partial, gby=[URLHash@0 as URLHash, EventDate@1 as EventDate], aggr=[count() as count(Int64(1))] 08)--------------ProjectionExec: expr=[URLHash@0 as URLHash, CAST(CAST(EventDate@1 AS Int32) AS Date32) as EventDate] 09)----------------FilterExec: CounterID@1 = 62 AND EventDate@0 >= 15887 AND EventDate@0 <= 15917 AND IsRefresh@2 = 0 AND (TraficSourceID@3 = -1 OR TraficSourceID@3 = 6) AND RefererHash@4 = 3594120000172545465, projection=[URLHash@5, EventDate@0] 10)------------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 @@ -1127,7 +1127,7 @@ logical_plan 01)Limit: skip=10000, fetch=10 02)--Sort: pageviews DESC NULLS FIRST, fetch=10010 03)----Projection: hits.WindowClientWidth, hits.WindowClientHeight, count(Int64(1)) AS count(*) AS pageviews -04)------Aggregate: groupBy=[[hits.WindowClientWidth, hits.WindowClientHeight]], aggr=[[count(Int64(1))]] +04)------Aggregate: groupBy=[[hits.WindowClientWidth, hits.WindowClientHeight]], aggr=[[count() AS count(Int64(1))]] 05)--------SubqueryAlias: hits 06)----------Projection: hits_raw.WindowClientWidth, hits_raw.WindowClientHeight 07)------------Filter: hits_raw.CounterID = Int32(62) AND hits_raw.EventDate >= UInt16(15887) AND hits_raw.EventDate <= UInt16(15917) AND hits_raw.IsRefresh = Int16(0) AND hits_raw.DontCountHits = Int16(0) AND hits_raw.URLHash = Int64(2868770270353813622) @@ -1137,9 +1137,9 @@ physical_plan 02)--SortPreservingMergeExec: [pageviews@2 DESC], fetch=10010 03)----ProjectionExec: expr=[WindowClientWidth@0 as WindowClientWidth, WindowClientHeight@1 as WindowClientHeight, count(Int64(1))@2 as pageviews] 04)------SortExec: TopK(fetch=10010), expr=[count(Int64(1))@2 DESC], preserve_partitioning=[true] -05)--------AggregateExec: mode=FinalPartitioned, gby=[WindowClientWidth@0 as WindowClientWidth, WindowClientHeight@1 as WindowClientHeight], aggr=[count(Int64(1))] +05)--------AggregateExec: mode=FinalPartitioned, gby=[WindowClientWidth@0 as WindowClientWidth, WindowClientHeight@1 as WindowClientHeight], aggr=[count() as count(Int64(1))] 06)----------RepartitionExec: partitioning=Hash([WindowClientWidth@0, WindowClientHeight@1], 4), input_partitions=4 -07)------------AggregateExec: mode=Partial, gby=[WindowClientWidth@0 as WindowClientWidth, WindowClientHeight@1 as WindowClientHeight], aggr=[count(Int64(1))] +07)------------AggregateExec: mode=Partial, gby=[WindowClientWidth@0 as WindowClientWidth, WindowClientHeight@1 as WindowClientHeight], aggr=[count() as count(Int64(1))] 08)--------------FilterExec: CounterID@1 = 62 AND EventDate@0 >= 15887 AND EventDate@0 <= 15917 AND IsRefresh@2 = 0 AND DontCountHits@5 = 0 AND URLHash@6 = 2868770270353813622, projection=[WindowClientWidth@3, WindowClientHeight@4] 09)----------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 10)------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[EventDate, CounterID, IsRefresh, WindowClientWidth, WindowClientHeight, DontCountHits, URLHash], file_type=parquet, predicate=CounterID@6 = 62 AND EventDate@5 >= 15887 AND EventDate@5 <= 15917 AND IsRefresh@15 = 0 AND DontCountHits@61 = 0 AND URLHash@103 = 2868770270353813622, pruning_predicate=CounterID_null_count@2 != row_count@3 AND CounterID_min@0 <= 62 AND 62 <= CounterID_max@1 AND EventDate_null_count@5 != row_count@3 AND EventDate_max@4 >= 15887 AND EventDate_null_count@5 != row_count@3 AND EventDate_min@6 <= 15917 AND IsRefresh_null_count@9 != row_count@3 AND IsRefresh_min@7 <= 0 AND 0 <= IsRefresh_max@8 AND DontCountHits_null_count@12 != row_count@3 AND DontCountHits_min@10 <= 0 AND 0 <= DontCountHits_max@11 AND URLHash_null_count@15 != row_count@3 AND URLHash_min@13 <= 2868770270353813622 AND 2868770270353813622 <= URLHash_max@14, required_guarantees=[CounterID in (62), DontCountHits in (0), IsRefresh in (0), URLHash in (2868770270353813622)] @@ -1156,7 +1156,7 @@ logical_plan 01)Limit: skip=1000, fetch=10 02)--Sort: date_trunc(Utf8("minute"), m) ASC NULLS LAST, fetch=1010 03)----Projection: date_trunc(Utf8("minute"),to_timestamp_seconds(hits.EventTime)) AS m, count(Int64(1)) AS count(*) AS pageviews -04)------Aggregate: groupBy=[[date_trunc(Utf8("minute"), to_timestamp_seconds(hits.EventTime))]], aggr=[[count(Int64(1))]] +04)------Aggregate: groupBy=[[date_trunc(Utf8("minute"), to_timestamp_seconds(hits.EventTime))]], aggr=[[count() AS count(Int64(1))]] 05)--------SubqueryAlias: hits 06)----------Projection: hits_raw.EventTime 07)------------Filter: hits_raw.CounterID = Int32(62) AND hits_raw.EventDate >= UInt16(15900) AND hits_raw.EventDate <= UInt16(15901) AND hits_raw.IsRefresh = Int16(0) AND hits_raw.DontCountHits = Int16(0) @@ -1166,9 +1166,9 @@ physical_plan 02)--SortPreservingMergeExec: [date_trunc(minute, m@0) ASC NULLS LAST], fetch=1010 03)----SortExec: TopK(fetch=1010), expr=[date_trunc(minute, m@0) ASC NULLS LAST], preserve_partitioning=[true] 04)------ProjectionExec: expr=[date_trunc(Utf8("minute"),to_timestamp_seconds(hits.EventTime))@0 as m, count(Int64(1))@1 as pageviews] -05)--------AggregateExec: mode=FinalPartitioned, gby=[date_trunc(Utf8("minute"),to_timestamp_seconds(hits.EventTime))@0 as date_trunc(Utf8("minute"),to_timestamp_seconds(hits.EventTime))], aggr=[count(Int64(1))] +05)--------AggregateExec: mode=FinalPartitioned, gby=[date_trunc(Utf8("minute"),to_timestamp_seconds(hits.EventTime))@0 as date_trunc(Utf8("minute"),to_timestamp_seconds(hits.EventTime))], aggr=[count() as count(Int64(1))] 06)----------RepartitionExec: partitioning=Hash([date_trunc(Utf8("minute"),to_timestamp_seconds(hits.EventTime))@0], 4), input_partitions=4 -07)------------AggregateExec: mode=Partial, gby=[date_trunc(minute, to_timestamp_seconds(EventTime@0)) as date_trunc(Utf8("minute"),to_timestamp_seconds(hits.EventTime))], aggr=[count(Int64(1))] +07)------------AggregateExec: mode=Partial, gby=[date_trunc(minute, to_timestamp_seconds(EventTime@0)) as date_trunc(Utf8("minute"),to_timestamp_seconds(hits.EventTime))], aggr=[count() as count(Int64(1))] 08)--------------FilterExec: CounterID@2 = 62 AND EventDate@1 >= 15900 AND EventDate@1 <= 15901 AND IsRefresh@3 = 0 AND DontCountHits@4 = 0, projection=[EventTime@0] 09)----------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 10)------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[EventTime, EventDate, CounterID, IsRefresh, DontCountHits], file_type=parquet, predicate=CounterID@6 = 62 AND EventDate@5 >= 15900 AND EventDate@5 <= 15901 AND IsRefresh@15 = 0 AND DontCountHits@61 = 0, pruning_predicate=CounterID_null_count@2 != row_count@3 AND CounterID_min@0 <= 62 AND 62 <= CounterID_max@1 AND EventDate_null_count@5 != row_count@3 AND EventDate_max@4 >= 15900 AND EventDate_null_count@5 != row_count@3 AND EventDate_min@6 <= 15901 AND IsRefresh_null_count@9 != row_count@3 AND IsRefresh_min@7 <= 0 AND 0 <= IsRefresh_max@8 AND DontCountHits_null_count@12 != row_count@3 AND DontCountHits_min@10 <= 0 AND 0 <= DontCountHits_max@11, required_guarantees=[CounterID in (62), DontCountHits in (0), IsRefresh in (0)] diff --git a/datafusion/sqllogictest/test_files/count_star_rule.slt b/datafusion/sqllogictest/test_files/count_star_rule.slt index 31d11e114e893..9814fd1347589 100644 --- a/datafusion/sqllogictest/test_files/count_star_rule.slt +++ b/datafusion/sqllogictest/test_files/count_star_rule.slt @@ -32,7 +32,7 @@ EXPLAIN SELECT COUNT() FROM (SELECT 1 AS a, 2 AS b) AS t; ---- logical_plan 01)Projection: count(Int64(1)) AS count() -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 03)----SubqueryAlias: t 04)------EmptyRelation: rows=1 physical_plan @@ -44,13 +44,13 @@ EXPLAIN SELECT t1.a, COUNT() FROM t1 GROUP BY t1.a; ---- logical_plan 01)Projection: t1.a, count(Int64(1)) AS count() -02)--Aggregate: groupBy=[[t1.a]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[t1.a]], aggr=[[count() AS count(Int64(1))]] 03)----TableScan: t1 projection=[a] physical_plan 01)ProjectionExec: expr=[a@0 as a, count(Int64(1))@1 as count()] -02)--AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[count(Int64(1))] +02)--AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[count() as count(Int64(1))] 03)----RepartitionExec: partitioning=Hash([a@0], 4), input_partitions=1 -04)------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[count() as count(Int64(1))] 05)--------DataSourceExec: partitions=1, partition_sizes=[1] query TT @@ -59,14 +59,14 @@ EXPLAIN SELECT t1.a, COUNT() AS cnt FROM t1 GROUP BY t1.a HAVING COUNT() > 0; logical_plan 01)Projection: t1.a, count(Int64(1)) AS count() AS cnt 02)--Filter: count(Int64(1)) > Int64(0) -03)----Aggregate: groupBy=[[t1.a]], aggr=[[count(Int64(1))]] +03)----Aggregate: groupBy=[[t1.a]], aggr=[[count() AS count(Int64(1))]] 04)------TableScan: t1 projection=[a] physical_plan 01)ProjectionExec: expr=[a@0 as a, count(Int64(1))@1 as cnt] 02)--FilterExec: count(Int64(1))@1 > 0 -03)----AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[count(Int64(1))] +03)----AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[count() as count(Int64(1))] 04)------RepartitionExec: partitioning=Hash([a@0], 4), input_partitions=1 -05)--------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[count(Int64(1))] +05)--------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[count() as count(Int64(1))] 06)----------DataSourceExec: partitions=1, partition_sizes=[1] query II diff --git a/datafusion/sqllogictest/test_files/dynamic_filter_pushdown_config.slt b/datafusion/sqllogictest/test_files/dynamic_filter_pushdown_config.slt index 6a6bad99f0840..f224fbf9066bd 100644 --- a/datafusion/sqllogictest/test_files/dynamic_filter_pushdown_config.slt +++ b/datafusion/sqllogictest/test_files/dynamic_filter_pushdown_config.slt @@ -631,15 +631,15 @@ EXPLAIN SELECT COUNT(*), MAX(score) FROM agg_parquet WHERE category = 'alpha'; ---- logical_plan 01)Projection: count(Int64(1)) AS count(*), max(agg_parquet.score) -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1)), max(agg_parquet.score)]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1)), max(agg_parquet.score)]] 03)----Projection: agg_parquet.score 04)------Filter: agg_parquet.category = Utf8View("alpha") 05)--------TableScan: agg_parquet projection=[category, score], partial_filters=[agg_parquet.category = Utf8View("alpha")] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*), max(agg_parquet.score)@1 as max(agg_parquet.score)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1)), max(agg_parquet.score)] +02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1)), max(agg_parquet.score)] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1)), max(agg_parquet.score)] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1)), max(agg_parquet.score)] 05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/dynamic_filter_pushdown_config/agg_data.parquet]]}, projection=[score], file_type=parquet, predicate=category@0 = alpha, pruning_predicate=category_null_count@2 != row_count@3 AND category_min@0 <= alpha AND alpha <= category_max@1, required_guarantees=[category in (alpha)] diff --git a/datafusion/sqllogictest/test_files/explain_analyze.slt b/datafusion/sqllogictest/test_files/explain_analyze.slt index 511ba88ed3345..d54630b7782d7 100644 --- a/datafusion/sqllogictest/test_files/explain_analyze.slt +++ b/datafusion/sqllogictest/test_files/explain_analyze.slt @@ -500,7 +500,7 @@ GROUP BY k; ---- Plan with Metrics 01)ProjectionExec: expr=[k@0 as k, count(Int64(1))@1 as count(*)], metrics=[output_bytes=1056.0 B] -02)--AggregateExec: mode=Single, gby=[k@0 as k], aggr=[count(Int64(1))], metrics=[output_bytes=1056.0 B, spilled_bytes=0.0 B, peak_mem_used=9.2 KB] +02)--AggregateExec: mode=Single, gby=[k@0 as k], aggr=[count() as count(Int64(1))], metrics=[output_bytes=1056.0 B, spilled_bytes=0.0 B, peak_mem_used=9.2 KB] 03)----ProjectionExec: expr=[column1@0 as k], metrics=[output_bytes=32.0 B] 04)------DataSourceExec: partitions=1, partition_sizes=[1], metrics=[] diff --git a/datafusion/sqllogictest/test_files/explain_tree.slt b/datafusion/sqllogictest/test_files/explain_tree.slt index fcac86aa21a2e..ba2e1d2078993 100644 --- a/datafusion/sqllogictest/test_files/explain_tree.slt +++ b/datafusion/sqllogictest/test_files/explain_tree.slt @@ -1239,45 +1239,47 @@ physical_plan 07)┌─────────────┴─────────────┐ 08)│ AggregateExec │ 09)│ -------------------- │ -10)│ aggr: count(1) │ -11)│ group_by: name │ +10)│ aggr: │ +11)│ count() as count(Int64(1))│ 12)│ │ -13)│ mode: │ -14)│ SinglePartitioned │ -15)└─────────────┬─────────────┘ -16)┌─────────────┴─────────────┐ -17)│ InterleaveExec ├──────────────┐ -18)└─────────────┬─────────────┘ │ -19)┌─────────────┴─────────────┐┌─────────────┴─────────────┐ -20)│ AggregateExec ││ AggregateExec │ -21)│ -------------------- ││ -------------------- │ -22)│ group_by: name ││ group_by: name │ -23)│ ││ │ -24)│ mode: ││ mode: │ -25)│ FinalPartitioned ││ FinalPartitioned │ -26)└─────────────┬─────────────┘└─────────────┬─────────────┘ -27)┌─────────────┴─────────────┐┌─────────────┴─────────────┐ -28)│ RepartitionExec ││ RepartitionExec │ -29)│ -------------------- ││ -------------------- │ -30)│ partition_count(in->out): ││ partition_count(in->out): │ -31)│ 1 -> 4 ││ 1 -> 4 │ -32)│ ││ │ -33)│ partitioning_scheme: ││ partitioning_scheme: │ -34)│ Hash([name@0], 4) ││ Hash([name@0], 4) │ -35)└─────────────┬─────────────┘└─────────────┬─────────────┘ -36)┌─────────────┴─────────────┐┌─────────────┴─────────────┐ -37)│ AggregateExec ││ AggregateExec │ -38)│ -------------------- ││ -------------------- │ -39)│ group_by: name ││ group_by: name │ -40)│ mode: Partial ││ mode: Partial │ -41)└─────────────┬─────────────┘└─────────────┬─────────────┘ -42)┌─────────────┴─────────────┐┌─────────────┴─────────────┐ -43)│ DataSourceExec ││ DataSourceExec │ -44)│ -------------------- ││ -------------------- │ -45)│ bytes: 288 ││ bytes: 280 │ -46)│ format: memory ││ format: memory │ -47)│ rows: 3 ││ rows: 3 │ -48)└───────────────────────────┘└───────────────────────────┘ +13)│ group_by: name │ +14)│ │ +15)│ mode: │ +16)│ SinglePartitioned │ +17)└─────────────┬─────────────┘ +18)┌─────────────┴─────────────┐ +19)│ InterleaveExec ├──────────────┐ +20)└─────────────┬─────────────┘ │ +21)┌─────────────┴─────────────┐┌─────────────┴─────────────┐ +22)│ AggregateExec ││ AggregateExec │ +23)│ -------------------- ││ -------------------- │ +24)│ group_by: name ││ group_by: name │ +25)│ ││ │ +26)│ mode: ││ mode: │ +27)│ FinalPartitioned ││ FinalPartitioned │ +28)└─────────────┬─────────────┘└─────────────┬─────────────┘ +29)┌─────────────┴─────────────┐┌─────────────┴─────────────┐ +30)│ RepartitionExec ││ RepartitionExec │ +31)│ -------------------- ││ -------------------- │ +32)│ partition_count(in->out): ││ partition_count(in->out): │ +33)│ 1 -> 4 ││ 1 -> 4 │ +34)│ ││ │ +35)│ partitioning_scheme: ││ partitioning_scheme: │ +36)│ Hash([name@0], 4) ││ Hash([name@0], 4) │ +37)└─────────────┬─────────────┘└─────────────┬─────────────┘ +38)┌─────────────┴─────────────┐┌─────────────┴─────────────┐ +39)│ AggregateExec ││ AggregateExec │ +40)│ -------------------- ││ -------------------- │ +41)│ group_by: name ││ group_by: name │ +42)│ mode: Partial ││ mode: Partial │ +43)└─────────────┬─────────────┘└─────────────┬─────────────┘ +44)┌─────────────┴─────────────┐┌─────────────┴─────────────┐ +45)│ DataSourceExec ││ DataSourceExec │ +46)│ -------------------- ││ -------------------- │ +47)│ bytes: 288 ││ bytes: 280 │ +48)│ format: memory ││ format: memory │ +49)│ rows: 3 ││ rows: 3 │ +50)└───────────────────────────┘└───────────────────────────┘ # Test explain tree for UnionExec query TT @@ -1713,49 +1715,53 @@ physical_plan 07)┌─────────────┴─────────────┐ 08)│ AggregateExec │ 09)│ -------------------- │ -10)│ aggr: count(1) │ -11)│ mode: Final │ -12)└─────────────┬─────────────┘ -13)┌─────────────┴─────────────┐ -14)│ CoalescePartitionsExec │ -15)└─────────────┬─────────────┘ -16)┌─────────────┴─────────────┐ -17)│ AggregateExec │ -18)│ -------------------- │ -19)│ aggr: count(1) │ -20)│ mode: Partial │ -21)└─────────────┬─────────────┘ -22)┌─────────────┴─────────────┐ -23)│ RepartitionExec │ -24)│ -------------------- │ -25)│ partition_count(in->out): │ -26)│ 1 -> 4 │ -27)│ │ -28)│ partitioning_scheme: │ -29)│ RoundRobinBatch(4) │ -30)└─────────────┬─────────────┘ -31)┌─────────────┴─────────────┐ -32)│ ProjectionExec │ -33)└─────────────┬─────────────┘ -34)┌─────────────┴─────────────┐ -35)│ GlobalLimitExec │ -36)│ -------------------- │ -37)│ limit: 3 │ -38)│ skip: 6 │ -39)└─────────────┬─────────────┘ -40)┌─────────────┴─────────────┐ -41)│ FilterExec │ -42)│ -------------------- │ -43)│ fetch: 9 │ -44)│ predicate: a > 3 │ -45)└─────────────┬─────────────┘ -46)┌─────────────┴─────────────┐ -47)│ DataSourceExec │ -48)│ -------------------- │ -49)│ bytes: 160 │ -50)│ format: memory │ -51)│ rows: 10 │ -52)└───────────────────────────┘ +10)│ aggr: │ +11)│ count() as count(Int64(1))│ +12)│ │ +13)│ mode: Final │ +14)└─────────────┬─────────────┘ +15)┌─────────────┴─────────────┐ +16)│ CoalescePartitionsExec │ +17)└─────────────┬─────────────┘ +18)┌─────────────┴─────────────┐ +19)│ AggregateExec │ +20)│ -------------------- │ +21)│ aggr: │ +22)│ count() as count(Int64(1))│ +23)│ │ +24)│ mode: Partial │ +25)└─────────────┬─────────────┘ +26)┌─────────────┴─────────────┐ +27)│ RepartitionExec │ +28)│ -------------------- │ +29)│ partition_count(in->out): │ +30)│ 1 -> 4 │ +31)│ │ +32)│ partitioning_scheme: │ +33)│ RoundRobinBatch(4) │ +34)└─────────────┬─────────────┘ +35)┌─────────────┴─────────────┐ +36)│ ProjectionExec │ +37)└─────────────┬─────────────┘ +38)┌─────────────┴─────────────┐ +39)│ GlobalLimitExec │ +40)│ -------------------- │ +41)│ limit: 3 │ +42)│ skip: 6 │ +43)└─────────────┬─────────────┘ +44)┌─────────────┴─────────────┐ +45)│ FilterExec │ +46)│ -------------------- │ +47)│ fetch: 9 │ +48)│ predicate: a > 3 │ +49)└─────────────┬─────────────┘ +50)┌─────────────┴─────────────┐ +51)│ DataSourceExec │ +52)│ -------------------- │ +53)│ bytes: 160 │ +54)│ format: memory │ +55)│ rows: 10 │ +56)└───────────────────────────┘ # clean up statement ok diff --git a/datafusion/sqllogictest/test_files/functional_dependencies.slt b/datafusion/sqllogictest/test_files/functional_dependencies.slt index c49004190dc60..db74d9404511c 100644 --- a/datafusion/sqllogictest/test_files/functional_dependencies.slt +++ b/datafusion/sqllogictest/test_files/functional_dependencies.slt @@ -160,7 +160,7 @@ EXPLAIN SELECT x, cnt FROM (SELECT x, count(*) AS cnt FROM t_uniq GROUP BY x) OR logical_plan 01)Sort: t_uniq.x ASC NULLS LAST 02)--Projection: t_uniq.x, count(Int64(1)) AS cnt -03)----Aggregate: groupBy=[[t_uniq.x]], aggr=[[count(Int64(1))]] +03)----Aggregate: groupBy=[[t_uniq.x]], aggr=[[count() AS count(Int64(1))]] 04)------TableScan: t_uniq projection=[x] @@ -268,14 +268,14 @@ EXPLAIN SELECT g.x, count(*) AS c ---- logical_plan 01)Projection: g.x, count(Int64(1)) AS count(*) AS c -02)--Aggregate: groupBy=[[g.x, g.cnt]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[g.x, g.cnt]], aggr=[[count() AS count(Int64(1))]] 03)----Projection: g.x, g.cnt 04)------Left Join: CAST(a.z AS Int64) = g.cnt 05)--------SubqueryAlias: a 06)----------TableScan: t_probe projection=[z] 07)--------SubqueryAlias: g 08)----------Projection: t_null.x, count(Int64(1)) AS count(*) AS cnt -09)------------Aggregate: groupBy=[[t_null.x]], aggr=[[count(Int64(1))]] +09)------------Aggregate: groupBy=[[t_null.x]], aggr=[[count() AS count(Int64(1))]] 10)--------------TableScan: t_null projection=[x] # 5.2 The ORDER BY variant: `g.x` is NULL for both rows, so the `g.cnt` diff --git a/datafusion/sqllogictest/test_files/group_by.slt b/datafusion/sqllogictest/test_files/group_by.slt index 97539ea479d40..e339204a63fc2 100644 --- a/datafusion/sqllogictest/test_files/group_by.slt +++ b/datafusion/sqllogictest/test_files/group_by.slt @@ -4532,16 +4532,16 @@ EXPLAIN SELECT c1, count(distinct c2), min(distinct c2), sum(c3), max(c4) FROM a logical_plan 01)Sort: aggregate_test_100.c1 ASC NULLS LAST 02)--Projection: aggregate_test_100.c1, count(alias1) AS count(DISTINCT aggregate_test_100.c2), min(alias1) AS min(DISTINCT aggregate_test_100.c2), sum(alias2) AS sum(aggregate_test_100.c3), max(alias3) AS max(aggregate_test_100.c4) -03)----Aggregate: groupBy=[[aggregate_test_100.c1]], aggr=[[count(alias1), min(alias1), sum(alias2), max(alias3)]] +03)----Aggregate: groupBy=[[aggregate_test_100.c1]], aggr=[[count() AS count(alias1), min(alias1), sum(alias2), max(alias3)]] 04)------Aggregate: groupBy=[[aggregate_test_100.c1, aggregate_test_100.c2 AS alias1]], aggr=[[sum(CAST(aggregate_test_100.c3 AS Int64)) AS alias2, max(aggregate_test_100.c4) AS alias3]] 05)--------TableScan: aggregate_test_100 projection=[c1, c2, c3, c4] physical_plan 01)SortPreservingMergeExec: [c1@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[c1@0 as c1, count(alias1)@1 as count(DISTINCT aggregate_test_100.c2), min(alias1)@2 as min(DISTINCT aggregate_test_100.c2), sum(alias2)@3 as sum(aggregate_test_100.c3), max(alias3)@4 as max(aggregate_test_100.c4)] 03)----SortExec: expr=[c1@0 ASC NULLS LAST], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[c1@0 as c1], aggr=[count(alias1), min(alias1), sum(alias2), max(alias3)] +04)------AggregateExec: mode=FinalPartitioned, gby=[c1@0 as c1], aggr=[count() as count(alias1), min(alias1), sum(alias2), max(alias3)] 05)--------RepartitionExec: partitioning=Hash([c1@0], 8), input_partitions=8 -06)----------AggregateExec: mode=Partial, gby=[c1@0 as c1], aggr=[count(alias1), min(alias1), sum(alias2), max(alias3)] +06)----------AggregateExec: mode=Partial, gby=[c1@0 as c1], aggr=[count() as count(alias1), min(alias1), sum(alias2), max(alias3)] 07)------------AggregateExec: mode=FinalPartitioned, gby=[c1@0 as c1, alias1@1 as alias1], aggr=[sum(aggregate_test_100.c3) as alias2, max(aggregate_test_100.c4) as alias3] 08)--------------RepartitionExec: partitioning=Hash([c1@0, alias1@1], 8), input_partitions=8 09)----------------AggregateExec: mode=Partial, gby=[c1@0 as c1, c2@1 as alias1], aggr=[sum(aggregate_test_100.c3) as alias2, max(aggregate_test_100.c4) as alias3] @@ -5235,15 +5235,16 @@ GROUP BY ---- logical_plan 01)Projection: date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 120000000000 }"),keywords_stream.ts,Utf8("2000-01-01")) AS ts_chunk, count(keywords_stream.keyword) AS alert_keyword_count -02)--Aggregate: groupBy=[[date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 120000000000 }"), keywords_stream.ts, TimestampNanosecond(946684800000000000, None)) AS date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 120000000000 }"),keywords_stream.ts,Utf8("2000-01-01"))]], aggr=[[count(keywords_stream.keyword)]] -03)----LeftSemi Join: keywords_stream.keyword = __correlated_sq_1.keyword -04)------TableScan: keywords_stream projection=[ts, keyword] -05)------SubqueryAlias: __correlated_sq_1 -06)--------TableScan: alert_keywords projection=[keyword] +02)--Aggregate: groupBy=[[date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 120000000000 }"), keywords_stream.ts, TimestampNanosecond(946684800000000000, None)) AS date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 120000000000 }"),keywords_stream.ts,Utf8("2000-01-01"))]], aggr=[[count() AS count(keywords_stream.keyword)]] +03)----Projection: keywords_stream.ts +04)------LeftSemi Join: keywords_stream.keyword = __correlated_sq_1.keyword +05)--------TableScan: keywords_stream projection=[ts, keyword] +06)--------SubqueryAlias: __correlated_sq_1 +07)----------TableScan: alert_keywords projection=[keyword] physical_plan 01)ProjectionExec: expr=[date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 120000000000 }"),keywords_stream.ts,Utf8("2000-01-01"))@0 as ts_chunk, count(keywords_stream.keyword)@1 as alert_keyword_count] -02)--AggregateExec: mode=Single, gby=[date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 120000000000 }, ts@0, 946684800000000000) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 120000000000 }"),keywords_stream.ts,Utf8("2000-01-01"))], aggr=[count(keywords_stream.keyword)] -03)----HashJoinExec: mode=CollectLeft, join_type=RightSemi, on=[(keyword@0, keyword@1)] +02)--AggregateExec: mode=Single, gby=[date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 120000000000 }, ts@0, 946684800000000000) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 120000000000 }"),keywords_stream.ts,Utf8("2000-01-01"))], aggr=[count() as count(keywords_stream.keyword)] +03)----HashJoinExec: mode=CollectLeft, join_type=RightSemi, on=[(keyword@0, keyword@1)], projection=[ts@0] 04)------DataSourceExec: partitions=1, partition_sizes=[3] 05)------DataSourceExec: partitions=1, partition_sizes=[3] diff --git a/datafusion/sqllogictest/test_files/joins.slt b/datafusion/sqllogictest/test_files/joins.slt index b594c68ce17d6..36f19eaada0d7 100644 --- a/datafusion/sqllogictest/test_files/joins.slt +++ b/datafusion/sqllogictest/test_files/joins.slt @@ -1423,16 +1423,16 @@ group by t1_id ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[join_t1.t1_id]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[join_t1.t1_id]], aggr=[[count() AS count(Int64(1))]] 03)----Projection: join_t1.t1_id 04)------Inner Join: join_t1.t1_id = join_t2.t2_id 05)--------TableScan: join_t1 projection=[t1_id] 06)--------TableScan: join_t2 projection=[t2_id] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@1 as count(*)] -02)--AggregateExec: mode=FinalPartitioned, gby=[t1_id@0 as t1_id], aggr=[count(Int64(1))] +02)--AggregateExec: mode=FinalPartitioned, gby=[t1_id@0 as t1_id], aggr=[count() as count(Int64(1))] 03)----RepartitionExec: partitioning=Hash([t1_id@0], 2), input_partitions=2 -04)------AggregateExec: mode=Partial, gby=[t1_id@0 as t1_id], aggr=[count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[t1_id@0 as t1_id], aggr=[count() as count(Int64(1))] 05)--------HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(t1_id@0, t2_id@0)], projection=[t1_id@0] 06)----------DataSourceExec: partitions=1, partition_sizes=[1] 07)----------RepartitionExec: partitioning=RoundRobinBatch(2), input_partitions=1 @@ -4524,7 +4524,7 @@ JOIN my_catalog.my_schema.table_with_many_types AS r ON l.binary_col = r.binary_ ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 03)----Projection: 04)------Inner Join: l.binary_col = r.binary_col 05)--------SubqueryAlias: l @@ -4533,7 +4533,7 @@ logical_plan 08)----------TableScan: my_catalog.my_schema.table_with_many_types projection=[binary_col] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Single, gby=[], aggr=[count(Int64(1))] +02)--AggregateExec: mode=Single, gby=[], aggr=[count() as count(Int64(1))] 03)----HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(binary_col@0, binary_col@0)], projection=[] 04)------DataSourceExec: partitions=1, partition_sizes=[1] 05)------DataSourceExec: partitions=1, partition_sizes=[1] @@ -5766,7 +5766,7 @@ EXPLAIN SELECT count(*) FROM elim_orders LEFT JOIN elim_users ON user_id = id; ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 03)----TableScan: elim_orders projection=[] physical_plan 01)ProjectionExec: expr=[3 as count(*)] @@ -5862,16 +5862,16 @@ EXPLAIN SELECT count(*) FROM elim_users LEFT JOIN elim_orders ON elim_users.id = ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 03)----Projection: 04)------Left Join: elim_users.id = elim_orders.user_id 05)--------TableScan: elim_users projection=[id] 06)--------TableScan: elim_orders projection=[user_id] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] 05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 06)----------HashJoinExec: mode=CollectLeft, join_type=Left, on=[(id@0, user_id@0)], projection=[] 07)------------DataSourceExec: partitions=1, partition_sizes=[1] @@ -6083,7 +6083,7 @@ EXPLAIN SELECT count(*) FROM elim_users RIGHT JOIN elim_orders ON id = user_id; ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 03)----TableScan: elim_orders projection=[] physical_plan 01)ProjectionExec: expr=[3 as count(*)] diff --git a/datafusion/sqllogictest/test_files/json.slt b/datafusion/sqllogictest/test_files/json.slt index e3b3d6b4e11ed..e9ff1bfafc2ec 100644 --- a/datafusion/sqllogictest/test_files/json.slt +++ b/datafusion/sqllogictest/test_files/json.slt @@ -55,13 +55,13 @@ EXPLAIN SELECT count(*) from json_test ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 03)----TableScan: json_test projection=[] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] 05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/2.json]]}, file_type=json diff --git a/datafusion/sqllogictest/test_files/lateral_join.slt b/datafusion/sqllogictest/test_files/lateral_join.slt index cae3e67153246..5779c14f08e42 100644 --- a/datafusion/sqllogictest/test_files/lateral_join.slt +++ b/datafusion/sqllogictest/test_files/lateral_join.slt @@ -764,7 +764,7 @@ logical_plan 04)------TableScan: t1 projection=[id] 05)------SubqueryAlias: sub 06)--------Projection: count(Int64(1)) AS cnt, t2.t1_id, Boolean(true) AS __always_true -07)----------Aggregate: groupBy=[[t2.t1_id]], aggr=[[count(Int64(1))]] +07)----------Aggregate: groupBy=[[t2.t1_id]], aggr=[[count() AS count(Int64(1))]] 08)------------TableScan: t2 projection=[t1_id] physical_plan 01)SortPreservingMergeExec: [id@0 ASC NULLS LAST] @@ -773,9 +773,9 @@ physical_plan 04)------HashJoinExec: mode=CollectLeft, join_type=Left, on=[(id@0, t1_id@1)], projection=[id@0, __always_true@3, cnt@1] 05)--------DataSourceExec: partitions=1, partition_sizes=[1] 06)--------ProjectionExec: expr=[count(Int64(1))@1 as cnt, t1_id@0 as t1_id, true as __always_true] -07)----------AggregateExec: mode=FinalPartitioned, gby=[t1_id@0 as t1_id], aggr=[count(Int64(1))] +07)----------AggregateExec: mode=FinalPartitioned, gby=[t1_id@0 as t1_id], aggr=[count() as count(Int64(1))] 08)------------RepartitionExec: partitioning=Hash([t1_id@0], 4), input_partitions=1 -09)--------------AggregateExec: mode=Partial, gby=[t1_id@0 as t1_id], aggr=[count(Int64(1))] +09)--------------AggregateExec: mode=Partial, gby=[t1_id@0 as t1_id], aggr=[count() as count(Int64(1))] 10)----------------DataSourceExec: partitions=1, partition_sizes=[1] # Verify LEFT lateral without aggregate decorrelates to left join diff --git a/datafusion/sqllogictest/test_files/limit.slt b/datafusion/sqllogictest/test_files/limit.slt index 1ff6ca4fb0253..c8ea8ca15ad78 100644 --- a/datafusion/sqllogictest/test_files/limit.slt +++ b/datafusion/sqllogictest/test_files/limit.slt @@ -308,7 +308,7 @@ EXPLAIN SELECT COUNT(*) FROM (SELECT a FROM t1 LIMIT 3 OFFSET 11); ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 03)----Limit: skip=11, fetch=3 04)------TableScan: t1 projection=[], fetch=14 physical_plan @@ -327,7 +327,7 @@ EXPLAIN SELECT COUNT(*) FROM (SELECT a FROM t1 LIMIT 3 OFFSET 8); ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 03)----Limit: skip=8, fetch=3 04)------TableScan: t1 projection=[], fetch=11 physical_plan @@ -369,7 +369,7 @@ EXPLAIN SELECT COUNT(*) FROM (SELECT a FROM t1 OFFSET 8); ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 03)----Limit: skip=8, fetch=None 04)------TableScan: t1 projection=[] physical_plan @@ -387,16 +387,16 @@ EXPLAIN SELECT COUNT(*) FROM (SELECT a FROM t1 WHERE a > 3 LIMIT 3 OFFSET 6); ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 03)----Projection: 04)------Limit: skip=6, fetch=3 05)--------Filter: t1.a > Int32(3) 06)----------TableScan: t1 projection=[a] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] 05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 06)----------ProjectionExec: expr=[] 07)------------GlobalLimitExec: skip=6, fetch=3 diff --git a/datafusion/sqllogictest/test_files/nested_loop_join_spill.slt b/datafusion/sqllogictest/test_files/nested_loop_join_spill.slt index 6eaf2b38a5bcf..4e12314e3cb00 100644 --- a/datafusion/sqllogictest/test_files/nested_loop_join_spill.slt +++ b/datafusion/sqllogictest/test_files/nested_loop_join_spill.slt @@ -50,7 +50,7 @@ INNER JOIN generate_series(1, 1) AS t2(v2) ---- Plan with Metrics 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)], metrics=[] -02)--AggregateExec: mode=Single, gby=[], aggr=[count(Int64(1))], metrics=[] +02)--AggregateExec: mode=Single, gby=[], aggr=[count() as count(Int64(1))], metrics=[] 03)----NestedLoopJoinExec: join_type=Inner, filter=v1@0 + v2@1 > 0, projection=[], metrics=[output_rows=100.0 K, spill_count=2, ] 04)------ProjectionExec: expr=[value@0 as v1], metrics=[] 05)--------LazyMemoryExec: partitions=1, batch_generators=[generate_series: start=1, end=100000, batch_size=8192], metrics=[] @@ -104,7 +104,7 @@ RIGHT JOIN generate_series(1, 200) AS t2(v2) ---- Plan with Metrics 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)], metrics=[] -02)--AggregateExec: mode=Single, gby=[], aggr=[count(Int64(1))], metrics=[] +02)--AggregateExec: mode=Single, gby=[], aggr=[count() as count(Int64(1))], metrics=[] 03)----ProjectionExec: expr=[], metrics=[] 04)------NestedLoopJoinExec: join_type=Right, filter=v1@0 + v2@1 = 2 AND join_proj_push_down_1@2, projection=[v1@0, v2@1], metrics=[output_rows=200, spill_count=2, ] 05)--------ProjectionExec: expr=[value@0 as v1], metrics=[] @@ -175,9 +175,9 @@ LEFT JOIN generate_series(1, 100) AS t2(v2) ---- Plan with Metrics 01)ProjectionExec: expr=[count(Int64(1))@0 as cnt], metrics=[] -02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))], metrics=[] +02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))], metrics=[] 03)----CoalescePartitionsExec, metrics=[] -04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))], metrics=[] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))], metrics=[] 05)--------NestedLoopJoinExec: join_type=Left, filter=v1@0 + v2@1 = 101, projection=[], metrics=[output_rows=5.00 K, spill_count=2, ] 06)----------ProjectionExec: expr=[value@0 as v1], metrics=[] 07)------------LazyMemoryExec: partitions=1, batch_generators=[generate_series: start=1, end=5000, batch_size=8192], metrics=[] diff --git a/datafusion/sqllogictest/test_files/optimizer_group_by_constant.slt b/datafusion/sqllogictest/test_files/optimizer_group_by_constant.slt index 8c0556547496a..fd2020e8f7383 100644 --- a/datafusion/sqllogictest/test_files/optimizer_group_by_constant.slt +++ b/datafusion/sqllogictest/test_files/optimizer_group_by_constant.slt @@ -49,7 +49,7 @@ GROUP BY 1, 2, 3, 4 ---- logical_plan 01)Projection: t.c1, Int64(99999), t.c5 + t.c8, Utf8("test"), count(Int64(1)) -02)--Aggregate: groupBy=[[t.c1, t.c5 + t.c8]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[t.c1, t.c5 + t.c8]], aggr=[[count() AS count(Int64(1))]] 03)----SubqueryAlias: t 04)------TableScan: test_table projection=[c1, c5, c8] @@ -60,7 +60,7 @@ FROM test_table t group by 1, 2, 3 ---- logical_plan -01)Aggregate: groupBy=[[Int64(123), Int64(456), Int64(789)]], aggr=[[count(Int64(1)), avg(t.c12)]] +01)Aggregate: groupBy=[[Int64(123), Int64(456), Int64(789)]], aggr=[[count() AS count(Int64(1)), avg(t.c12)]] 02)--SubqueryAlias: t 03)----TableScan: test_table projection=[c12] @@ -72,7 +72,7 @@ GROUP BY 1, 2 ---- logical_plan 01)Projection: to_date(Utf8("2023-05-04")) AS dt, date_part(Utf8("DAY"),now()) < Int64(1000) AS today_filter, count(Int64(1)) -02)--Aggregate: groupBy=[[Date32("2023-05-04") AS to_date(Utf8("2023-05-04")), Boolean(true) AS date_part(Utf8("DAY"),now()) < Int64(1000)]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[Date32("2023-05-04") AS to_date(Utf8("2023-05-04")), Boolean(true) AS date_part(Utf8("DAY"),now()) < Int64(1000)]], aggr=[[count() AS count(Int64(1))]] 03)----SubqueryAlias: t 04)------TableScan: test_table projection=[] @@ -89,7 +89,7 @@ FROM test_table t GROUP BY 1 ---- logical_plan -01)Aggregate: groupBy=[[Boolean(true) AS NOT date_part(Utf8("MONTH"),now()) BETWEEN Int64(50) AND Int64(60)]], aggr=[[count(Int64(1))]] +01)Aggregate: groupBy=[[Boolean(true) AS NOT date_part(Utf8("MONTH"),now()) BETWEEN Int64(50) AND Int64(60)]], aggr=[[count() AS count(Int64(1))]] 02)--SubqueryAlias: t 03)----TableScan: test_table projection=[] diff --git a/datafusion/sqllogictest/test_files/piecewise_merge_join_batches.slt b/datafusion/sqllogictest/test_files/piecewise_merge_join_batches.slt index d8b3c16163aff..95d642c08c04b 100644 --- a/datafusion/sqllogictest/test_files/piecewise_merge_join_batches.slt +++ b/datafusion/sqllogictest/test_files/piecewise_merge_join_batches.slt @@ -38,7 +38,7 @@ EXPLAIN SELECT count(*) FROM pb_l l WHERE EXISTS (SELECT 1 FROM range(3, 8) r WH ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 03)----Projection: 04)------LeftSemi Join: Filter: CAST(l.v AS Int64) > __correlated_sq_1.value 05)--------SubqueryAlias: l @@ -48,9 +48,9 @@ logical_plan 09)------------TableScan: range() projection=[value] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] 05)--------ProjectionExec: expr=[] 06)----------PiecewiseMergeJoin: operator=Gt, join_type=LeftSemi, on=(CAST(v AS Int64) > value) 07)------------SortPreservingMergeExec: [CAST(v@0 AS Int64) ASC] @@ -71,7 +71,7 @@ EXPLAIN SELECT count(*) FROM pb_l l JOIN range(3, 8) r ON l.v < r.value; ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 03)----Projection: 04)------Inner Join: Filter: CAST(l.v AS Int64) < r.value 05)--------SubqueryAlias: l @@ -80,9 +80,9 @@ logical_plan 08)----------TableScan: range() projection=[value] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] 05)--------ProjectionExec: expr=[] 06)----------PiecewiseMergeJoin: operator=Lt, join_type=Inner, on=(CAST(v AS Int64) < value) 07)------------SortPreservingMergeExec: [CAST(v@0 AS Int64) DESC] diff --git a/datafusion/sqllogictest/test_files/preserve_file_partitioning.slt b/datafusion/sqllogictest/test_files/preserve_file_partitioning.slt index e2dd22cc82bba..38654f36adc46 100644 --- a/datafusion/sqllogictest/test_files/preserve_file_partitioning.slt +++ b/datafusion/sqllogictest/test_files/preserve_file_partitioning.slt @@ -223,13 +223,13 @@ EXPLAIN SELECT f_dkey, count(*), sum(value) FROM fact_table GROUP BY f_dkey; ---- logical_plan 01)Projection: fact_table.f_dkey, count(Int64(1)) AS count(*), sum(fact_table.value) -02)--Aggregate: groupBy=[[fact_table.f_dkey]], aggr=[[count(Int64(1)), sum(fact_table.value)]] +02)--Aggregate: groupBy=[[fact_table.f_dkey]], aggr=[[count() AS count(Int64(1)), sum(fact_table.value)]] 03)----TableScan: fact_table projection=[value, f_dkey] physical_plan 01)ProjectionExec: expr=[f_dkey@0 as f_dkey, count(Int64(1))@1 as count(*), sum(fact_table.value)@2 as sum(fact_table.value)] -02)--AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[count(Int64(1)), sum(fact_table.value)] +02)--AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[count() as count(Int64(1)), sum(fact_table.value)] 03)----RepartitionExec: partitioning=Hash([f_dkey@0], 3), input_partitions=3 -04)------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[count(Int64(1)), sum(fact_table.value)] +04)------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[count() as count(Int64(1)), sum(fact_table.value)] 05)--------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=C/data.parquet]]}, projection=[value, f_dkey], file_type=parquet # Verify results without optimization @@ -253,11 +253,11 @@ EXPLAIN SELECT f_dkey, count(*), sum(value) FROM fact_table GROUP BY f_dkey; ---- logical_plan 01)Projection: fact_table.f_dkey, count(Int64(1)) AS count(*), sum(fact_table.value) -02)--Aggregate: groupBy=[[fact_table.f_dkey]], aggr=[[count(Int64(1)), sum(fact_table.value)]] +02)--Aggregate: groupBy=[[fact_table.f_dkey]], aggr=[[count() AS count(Int64(1)), sum(fact_table.value)]] 03)----TableScan: fact_table projection=[value, f_dkey] physical_plan 01)ProjectionExec: expr=[f_dkey@0 as f_dkey, count(Int64(1))@1 as count(*), sum(fact_table.value)@2 as sum(fact_table.value)] -02)--AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[count(Int64(1)), sum(fact_table.value)] +02)--AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[count() as count(Int64(1)), sum(fact_table.value)] 03)----DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=C/data.parquet]]}, projection=[value, f_dkey], output_partitioning=Hash([f_dkey@1], 3), file_type=parquet # Verify results with optimization match results without optimization @@ -282,14 +282,14 @@ EXPLAIN SELECT f_dkey, count(*), avg(value) FROM fact_table_ordered GROUP BY f_d logical_plan 01)Sort: fact_table_ordered.f_dkey ASC NULLS LAST 02)--Projection: fact_table_ordered.f_dkey, count(Int64(1)) AS count(*), avg(fact_table_ordered.value) -03)----Aggregate: groupBy=[[fact_table_ordered.f_dkey]], aggr=[[count(Int64(1)), avg(fact_table_ordered.value)]] +03)----Aggregate: groupBy=[[fact_table_ordered.f_dkey]], aggr=[[count() AS count(Int64(1)), avg(fact_table_ordered.value)]] 04)------TableScan: fact_table_ordered projection=[value, f_dkey] physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, count(Int64(1))@1 as count(*), avg(fact_table_ordered.value)@2 as avg(fact_table_ordered.value)] -03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted +03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[count() as count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted 04)------RepartitionExec: partitioning=Hash([f_dkey@0], 3), input_partitions=3, preserve_order=true, sort_exprs=f_dkey@0 ASC NULLS LAST -05)--------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted +05)--------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[count() as count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted 06)----------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=C/data.parquet]]}, projection=[value, f_dkey], output_ordering=[f_dkey@1 ASC NULLS LAST], file_type=parquet # Verify results without optimization @@ -314,12 +314,12 @@ EXPLAIN SELECT f_dkey, count(*), avg(value) FROM fact_table_ordered GROUP BY f_d logical_plan 01)Sort: fact_table_ordered.f_dkey ASC NULLS LAST 02)--Projection: fact_table_ordered.f_dkey, count(Int64(1)) AS count(*), avg(fact_table_ordered.value) -03)----Aggregate: groupBy=[[fact_table_ordered.f_dkey]], aggr=[[count(Int64(1)), avg(fact_table_ordered.value)]] +03)----Aggregate: groupBy=[[fact_table_ordered.f_dkey]], aggr=[[count() AS count(Int64(1)), avg(fact_table_ordered.value)]] 04)------TableScan: fact_table_ordered projection=[value, f_dkey] physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, count(Int64(1))@1 as count(*), avg(fact_table_ordered.value)@2 as avg(fact_table_ordered.value)] -03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted +03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[count() as count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted 04)------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=C/data.parquet]]}, projection=[value, f_dkey], output_ordering=[f_dkey@1 ASC NULLS LAST], output_partitioning=Hash([f_dkey@1], 3), file_type=parquet query TIR @@ -348,7 +348,7 @@ ORDER BY f.f_dkey; logical_plan 01)Sort: f.f_dkey ASC NULLS LAST 02)--Projection: f.f_dkey, max(d.env), max(d.service), count(Int64(1)) AS count(*), sum(f.value) -03)----Aggregate: groupBy=[[f.f_dkey]], aggr=[[max(d.env), max(d.service), count(Int64(1)), sum(f.value)]] +03)----Aggregate: groupBy=[[f.f_dkey]], aggr=[[max(d.env), max(d.service), count() AS count(Int64(1)), sum(f.value)]] 04)------Projection: f.value, f.f_dkey, d.env, d.service 05)--------Inner Join: f.f_dkey = d.d_dkey 06)----------SubqueryAlias: f @@ -359,9 +359,9 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, max(d.env)@1 as max(d.env), max(d.service)@2 as max(d.service), count(Int64(1))@3 as count(*), sum(f.value)@4 as sum(f.value)] -03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[max(d.env), max(d.service), count(Int64(1)), sum(f.value)], ordering_mode=Sorted +03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[max(d.env), max(d.service), count() as count(Int64(1)), sum(f.value)], ordering_mode=Sorted 04)------RepartitionExec: partitioning=Hash([f_dkey@0], 3), input_partitions=3, preserve_order=true, sort_exprs=f_dkey@0 ASC NULLS LAST -05)--------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[max(d.env), max(d.service), count(Int64(1)), sum(f.value)], ordering_mode=Sorted +05)--------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[max(d.env), max(d.service), count() as count(Int64(1)), sum(f.value)], ordering_mode=Sorted 06)----------HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(d_dkey@0, f_dkey@1)], projection=[value@3, f_dkey@4, env@1, service@2] 07)------------CoalescePartitionsExec 08)--------------FilterExec: service@2 = log @@ -401,7 +401,7 @@ ORDER BY f.f_dkey; logical_plan 01)Sort: f.f_dkey ASC NULLS LAST 02)--Projection: f.f_dkey, max(d.env), max(d.service), count(Int64(1)) AS count(*), sum(f.value) -03)----Aggregate: groupBy=[[f.f_dkey]], aggr=[[max(d.env), max(d.service), count(Int64(1)), sum(f.value)]] +03)----Aggregate: groupBy=[[f.f_dkey]], aggr=[[max(d.env), max(d.service), count() AS count(Int64(1)), sum(f.value)]] 04)------Projection: f.value, f.f_dkey, d.env, d.service 05)--------Inner Join: f.f_dkey = d.d_dkey 06)----------SubqueryAlias: f @@ -412,7 +412,7 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, max(d.env)@1 as max(d.env), max(d.service)@2 as max(d.service), count(Int64(1))@3 as count(*), sum(f.value)@4 as sum(f.value)] -03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[max(d.env), max(d.service), count(Int64(1)), sum(f.value)], ordering_mode=Sorted +03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[max(d.env), max(d.service), count() as count(Int64(1)), sum(f.value)], ordering_mode=Sorted 04)------HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(d_dkey@0, f_dkey@1)], projection=[value@3, f_dkey@4, env@1, service@2] 05)--------CoalescePartitionsExec 06)----------FilterExec: service@2 = log @@ -543,11 +543,11 @@ EXPLAIN SELECT f_dkey, count(*), sum(value) FROM high_cardinality_table GROUP BY ---- logical_plan 01)Projection: high_cardinality_table.f_dkey, count(Int64(1)) AS count(*), sum(high_cardinality_table.value) -02)--Aggregate: groupBy=[[high_cardinality_table.f_dkey]], aggr=[[count(Int64(1)), sum(high_cardinality_table.value)]] +02)--Aggregate: groupBy=[[high_cardinality_table.f_dkey]], aggr=[[count() AS count(Int64(1)), sum(high_cardinality_table.value)]] 03)----TableScan: high_cardinality_table projection=[value, f_dkey] physical_plan 01)ProjectionExec: expr=[f_dkey@0 as f_dkey, count(Int64(1))@1 as count(*), sum(high_cardinality_table.value)@2 as sum(high_cardinality_table.value)] -02)--AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[count(Int64(1)), sum(high_cardinality_table.value)] +02)--AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[count() as count(Int64(1)), sum(high_cardinality_table.value)] 03)----DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/high_cardinality/f_dkey=A/data.parquet, WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/high_cardinality/f_dkey=D/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/high_cardinality/f_dkey=B/data.parquet, WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/high_cardinality/f_dkey=E/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/high_cardinality/f_dkey=C/data.parquet]]}, projection=[value, f_dkey], output_partitioning=Hash([f_dkey@1], 3), file_type=parquet # Verify results with optimization match results without optimization @@ -585,13 +585,13 @@ EXPLAIN SELECT f_dkey, count(*), sum(value) FROM high_cardinality_table GROUP BY ---- logical_plan 01)Projection: high_cardinality_table.f_dkey, count(Int64(1)) AS count(*), sum(high_cardinality_table.value) -02)--Aggregate: groupBy=[[high_cardinality_table.f_dkey]], aggr=[[count(Int64(1)), sum(high_cardinality_table.value)]] +02)--Aggregate: groupBy=[[high_cardinality_table.f_dkey]], aggr=[[count() AS count(Int64(1)), sum(high_cardinality_table.value)]] 03)----TableScan: high_cardinality_table projection=[value, f_dkey] physical_plan 01)ProjectionExec: expr=[f_dkey@0 as f_dkey, count(Int64(1))@1 as count(*), sum(high_cardinality_table.value)@2 as sum(high_cardinality_table.value)] -02)--AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[count(Int64(1)), sum(high_cardinality_table.value)] +02)--AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[count() as count(Int64(1)), sum(high_cardinality_table.value)] 03)----RepartitionExec: partitioning=Hash([f_dkey@0], 3), input_partitions=3 -04)------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[count(Int64(1)), sum(high_cardinality_table.value)] +04)------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[count() as count(Int64(1)), sum(high_cardinality_table.value)] 05)--------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/high_cardinality/f_dkey=A/data.parquet, WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/high_cardinality/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/high_cardinality/f_dkey=C/data.parquet, WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/high_cardinality/f_dkey=D/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/high_cardinality/f_dkey=E/data.parquet]]}, projection=[value, f_dkey], file_type=parquet query TIR rowsort @@ -717,11 +717,11 @@ GROUP BY f_dkey, timestamp; ---- logical_plan 01)Projection: fact_table.f_dkey, fact_table.timestamp, count(Int64(1)) AS count(*), avg(fact_table.value) -02)--Aggregate: groupBy=[[fact_table.f_dkey, fact_table.timestamp]], aggr=[[count(Int64(1)), avg(fact_table.value)]] +02)--Aggregate: groupBy=[[fact_table.f_dkey, fact_table.timestamp]], aggr=[[count() AS count(Int64(1)), avg(fact_table.value)]] 03)----TableScan: fact_table projection=[timestamp, value, f_dkey] physical_plan 01)ProjectionExec: expr=[f_dkey@0 as f_dkey, timestamp@1 as timestamp, count(Int64(1))@2 as count(*), avg(fact_table.value)@3 as avg(fact_table.value)] -02)--AggregateExec: mode=SinglePartitioned, gby=[f_dkey@2 as f_dkey, timestamp@0 as timestamp], aggr=[count(Int64(1)), avg(fact_table.value)] +02)--AggregateExec: mode=SinglePartitioned, gby=[f_dkey@2 as f_dkey, timestamp@0 as timestamp], aggr=[count() as count(Int64(1)), avg(fact_table.value)] 03)----DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=C/data.parquet]]}, projection=[timestamp, value, f_dkey], output_partitioning=Hash([f_dkey@2], 3), file_type=parquet query TPIR rowsort diff --git a/datafusion/sqllogictest/test_files/projection_pushdown.slt b/datafusion/sqllogictest/test_files/projection_pushdown.slt index c92c95fdfbc59..1020741e21aa4 100644 --- a/datafusion/sqllogictest/test_files/projection_pushdown.slt +++ b/datafusion/sqllogictest/test_files/projection_pushdown.slt @@ -1858,11 +1858,11 @@ FROM simple_struct GROUP BY s; ---- logical_plan 01)Projection: get_field(simple_struct.s, Utf8("label")) IS NOT NULL AS has_label, count(Int64(1)) -02)--Aggregate: groupBy=[[simple_struct.s]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[simple_struct.s]], aggr=[[count() AS count(Int64(1))]] 03)----TableScan: simple_struct projection=[s] physical_plan 01)ProjectionExec: expr=[get_field(s@0, label) IS NOT NULL as has_label, count(Int64(1))@1 as count(Int64(1))] -02)--AggregateExec: mode=Single, gby=[s@0 as s], aggr=[count(Int64(1))] +02)--AggregateExec: mode=Single, gby=[s@0 as s], aggr=[count() as count(Int64(1))] 03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[s], file_type=parquet # Verify correctness - all labels are non-null diff --git a/datafusion/sqllogictest/test_files/push_down_filter_regression.slt b/datafusion/sqllogictest/test_files/push_down_filter_regression.slt index c260efae95334..4bb6e293a6d1f 100644 --- a/datafusion/sqllogictest/test_files/push_down_filter_regression.slt +++ b/datafusion/sqllogictest/test_files/push_down_filter_regression.slt @@ -649,14 +649,14 @@ EXPLAIN SELECT k, c FROM (SELECT random() < 0.5 AS k, count(*) AS c FROM generat logical_plan 01)Projection: random() < Float64(0.5) AS k, count(Int64(1)) AS c 02)--Filter: random() < Float64(0.5) -03)----Aggregate: groupBy=[[random() < Float64(0.5)]], aggr=[[count(Int64(1))]] +03)----Aggregate: groupBy=[[random() < Float64(0.5)]], aggr=[[count() AS count(Int64(1))]] 04)------TableScan: generate_series() projection=[] physical_plan 01)ProjectionExec: expr=[random() < Float64(0.5)@0 as k, count(Int64(1))@1 as c] 02)--FilterExec: random() < Float64(0.5)@0 -03)----AggregateExec: mode=FinalPartitioned, gby=[random() < Float64(0.5)@0 as random() < Float64(0.5)], aggr=[count(Int64(1))] +03)----AggregateExec: mode=FinalPartitioned, gby=[random() < Float64(0.5)@0 as random() < Float64(0.5)], aggr=[count() as count(Int64(1))] 04)------RepartitionExec: partitioning=Hash([random() < Float64(0.5)@0], 4), input_partitions=4 -05)--------AggregateExec: mode=Partial, gby=[random() < 0.5 as random() < Float64(0.5)], aggr=[count(Int64(1))] +05)--------AggregateExec: mode=Partial, gby=[random() < 0.5 as random() < Float64(0.5)], aggr=[count() as count(Int64(1))] 06)----------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 07)------------LazyMemoryExec: partitions=1, batch_generators=[generate_series: start=1, end=10000, batch_size=8192] diff --git a/datafusion/sqllogictest/test_files/repartition_subset_satisfaction.slt b/datafusion/sqllogictest/test_files/repartition_subset_satisfaction.slt index 5371ca59beea1..d6399f2419c18 100644 --- a/datafusion/sqllogictest/test_files/repartition_subset_satisfaction.slt +++ b/datafusion/sqllogictest/test_files/repartition_subset_satisfaction.slt @@ -156,14 +156,14 @@ ORDER BY f_dkey, time_bin; logical_plan 01)Sort: fact_table_ordered.f_dkey ASC NULLS LAST, time_bin ASC NULLS LAST 02)--Projection: fact_table_ordered.f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp) AS time_bin, count(Int64(1)) AS count(*), avg(fact_table_ordered.value) -03)----Aggregate: groupBy=[[fact_table_ordered.f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"), fact_table_ordered.timestamp)]], aggr=[[count(Int64(1)), avg(fact_table_ordered.value)]] +03)----Aggregate: groupBy=[[fact_table_ordered.f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"), fact_table_ordered.timestamp)]], aggr=[[count() AS count(Int64(1)), avg(fact_table_ordered.value)]] 04)------TableScan: fact_table_ordered projection=[timestamp, value, f_dkey] physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST, time_bin@1 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1 as time_bin, count(Int64(1))@2 as count(*), avg(fact_table_ordered.value)@3 as avg(fact_table_ordered.value)] -03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted +03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count() as count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted 04)------RepartitionExec: partitioning=Hash([f_dkey@0, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1], 3), input_partitions=3, preserve_order=true, sort_exprs=f_dkey@0 ASC NULLS LAST, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1 ASC NULLS LAST -05)--------AggregateExec: mode=Partial, gby=[f_dkey@2 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@0) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted +05)--------AggregateExec: mode=Partial, gby=[f_dkey@2 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@0) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count() as count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted 06)----------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=C/data.parquet]]}, projection=[timestamp, value, f_dkey], output_ordering=[f_dkey@2 ASC NULLS LAST, timestamp@0 ASC NULLS LAST], output_partitioning=Hash([f_dkey@2], 3), file_type=parquet # Verify results without subset satisfaction @@ -198,12 +198,12 @@ ORDER BY f_dkey, time_bin; logical_plan 01)Sort: fact_table_ordered.f_dkey ASC NULLS LAST, time_bin ASC NULLS LAST 02)--Projection: fact_table_ordered.f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp) AS time_bin, count(Int64(1)) AS count(*), avg(fact_table_ordered.value) -03)----Aggregate: groupBy=[[fact_table_ordered.f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"), fact_table_ordered.timestamp)]], aggr=[[count(Int64(1)), avg(fact_table_ordered.value)]] +03)----Aggregate: groupBy=[[fact_table_ordered.f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"), fact_table_ordered.timestamp)]], aggr=[[count() AS count(Int64(1)), avg(fact_table_ordered.value)]] 04)------TableScan: fact_table_ordered projection=[timestamp, value, f_dkey] physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST, time_bin@1 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1 as time_bin, count(Int64(1))@2 as count(*), avg(fact_table_ordered.value)@3 as avg(fact_table_ordered.value)] -03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@2 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@0) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted +03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@2 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@0) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count() as count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted 04)------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=C/data.parquet]]}, projection=[timestamp, value, f_dkey], output_ordering=[f_dkey@2 ASC NULLS LAST, timestamp@0 ASC NULLS LAST], output_partitioning=Hash([f_dkey@2], 3), file_type=parquet # Verify results match with subset satisfaction diff --git a/datafusion/sqllogictest/test_files/select.slt b/datafusion/sqllogictest/test_files/select.slt index d65cfa900c249..1ac0943b17edd 100644 --- a/datafusion/sqllogictest/test_files/select.slt +++ b/datafusion/sqllogictest/test_files/select.slt @@ -1565,16 +1565,16 @@ GROUP BY c2; ---- logical_plan 01)Projection: aggregate_test_100.c2, count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[aggregate_test_100.c2]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[aggregate_test_100.c2]], aggr=[[count() AS count(Int64(1))]] 03)----Projection: aggregate_test_100.c2 04)------Sort: aggregate_test_100.c1 ASC NULLS LAST, aggregate_test_100.c2 ASC NULLS LAST, fetch=4 05)--------Projection: aggregate_test_100.c2, aggregate_test_100.c1 06)----------TableScan: aggregate_test_100 projection=[c1, c2] physical_plan 01)ProjectionExec: expr=[c2@0 as c2, count(Int64(1))@1 as count(*)] -02)--AggregateExec: mode=FinalPartitioned, gby=[c2@0 as c2], aggr=[count(Int64(1))] +02)--AggregateExec: mode=FinalPartitioned, gby=[c2@0 as c2], aggr=[count() as count(Int64(1))] 03)----RepartitionExec: partitioning=Hash([c2@0], 2), input_partitions=2 -04)------AggregateExec: mode=Partial, gby=[c2@0 as c2], aggr=[count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[c2@0 as c2], aggr=[count() as count(Int64(1))] 05)--------RepartitionExec: partitioning=RoundRobinBatch(2), input_partitions=1 06)----------ProjectionExec: expr=[c2@1 as c2] 07)------------SortExec: TopK(fetch=4), expr=[c1@0 ASC NULLS LAST, c2@1 ASC NULLS LAST], preserve_partitioning=[false] diff --git a/datafusion/sqllogictest/test_files/single_distinct_to_groupby.slt b/datafusion/sqllogictest/test_files/single_distinct_to_groupby.slt index fddfc661b8cc5..cc414d025ef13 100644 --- a/datafusion/sqllogictest/test_files/single_distinct_to_groupby.slt +++ b/datafusion/sqllogictest/test_files/single_distinct_to_groupby.slt @@ -60,7 +60,7 @@ EXPLAIN SELECT g, count(*) AS records, count(DISTINCT x) AS distinct_x FROM t GR logical_plan 01)Projection: t.g, CASE WHEN sum(alias2) IS NOT NULL THEN sum(alias2) ELSE Int64(0) END AS records, count(alias1) AS distinct_x 02)--Aggregate: groupBy=[[t.g]], aggr=[[sum(alias2), count(alias1)]] -03)----Aggregate: groupBy=[[t.g, t.x AS alias1]], aggr=[[count(Int64(1)) AS alias2]] +03)----Aggregate: groupBy=[[t.g, t.x AS alias1]], aggr=[[count() AS alias2]] 04)------TableScan: t projection=[g, x] # A count next to min, max and sum, all beside the distinct count @@ -73,7 +73,7 @@ FROM t GROUP BY g; logical_plan 01)Projection: t.g, CASE WHEN sum(alias2) IS NOT NULL THEN sum(alias2) ELSE Int64(0) END AS records, count(alias1) AS distinct_x, min(alias3) AS min_v, max(alias4) AS max_v, sum(alias5) AS sum_v 02)--Aggregate: groupBy=[[t.g]], aggr=[[sum(alias2), count(alias1), min(alias3), max(alias4), sum(alias5)]] -03)----Aggregate: groupBy=[[t.g, t.x AS alias1]], aggr=[[count(Int64(1)) AS alias2, min(t.v) AS alias3, max(t.v) AS alias4, sum(CAST(t.v AS Int64)) AS alias5]] +03)----Aggregate: groupBy=[[t.g, t.x AS alias1]], aggr=[[count() AS alias2, min(t.v) AS alias3, max(t.v) AS alias4, sum(CAST(t.v AS Int64)) AS alias5]] 04)------TableScan: t projection=[g, x, v] # A count with a FILTER still blocks the rewrite: the filter is per input row, @@ -83,7 +83,7 @@ EXPLAIN SELECT g, count(*) FILTER (WHERE v > 1) AS records, count(DISTINCT x) AS ---- logical_plan 01)Projection: t.g, count(Int64(1)) FILTER (WHERE t.v > Int64(1)) AS count(*) FILTER (WHERE t.v > Int64(1)) AS records, count(DISTINCT t.x) AS distinct_x -02)--Aggregate: groupBy=[[t.g]], aggr=[[count(Int64(1)) FILTER (WHERE t.v > Int32(1)) AS count(Int64(1)) FILTER (WHERE t.v > Int64(1)), count(DISTINCT t.x)]] +02)--Aggregate: groupBy=[[t.g]], aggr=[[count() FILTER (WHERE t.v > Int32(1)) AS count(Int64(1)) FILTER (WHERE t.v > Int64(1)), count(DISTINCT t.x)]] 03)----TableScan: t projection=[g, x, v] # An unsupported non-distinct aggregate still blocks the rewrite @@ -108,7 +108,7 @@ EXPLAIN SELECT g, count(*) AS records, count(DISTINCT xi) AS distinct_xi FROM t ---- logical_plan 01)Projection: t.g, count(Int64(1)) AS count(*) AS records, count(DISTINCT t.xi) AS distinct_xi -02)--Aggregate: groupBy=[[t.g]], aggr=[[count(Int64(1)), count(DISTINCT t.xi)]] +02)--Aggregate: groupBy=[[t.g]], aggr=[[count() AS count(Int64(1)), count(DISTINCT t.xi)]] 03)----TableScan: t projection=[g, xi] # `sum` reports nothing about `GroupsAccumulator` support from the argument @@ -121,7 +121,7 @@ EXPLAIN SELECT g, count(*) AS records, sum(DISTINCT v) AS distinct_v FROM t GROU ---- logical_plan 01)Projection: t.g, count(Int64(1)) AS count(*) AS records, sum(DISTINCT t.v) AS distinct_v -02)--Aggregate: groupBy=[[t.g]], aggr=[[count(Int64(1)), sum(DISTINCT CAST(t.v AS Int64))]] +02)--Aggregate: groupBy=[[t.g]], aggr=[[count() AS count(Int64(1)), sum(DISTINCT CAST(t.v AS Int64))]] 03)----TableScan: t projection=[g, v] # `min(DISTINCT v)` is the same value as `min(v)`. EliminateAggregateDistinct @@ -133,7 +133,7 @@ EXPLAIN SELECT g, count(*) AS records, min(DISTINCT v) AS distinct_min_v FROM t ---- logical_plan 01)Projection: t.g, count(Int64(1)) AS count(*) AS records, min(DISTINCT t.v) AS distinct_min_v -02)--Aggregate: groupBy=[[t.g]], aggr=[[count(Int64(1)), min(t.v) AS min(DISTINCT t.v)]] +02)--Aggregate: groupBy=[[t.g]], aggr=[[count() AS count(Int64(1)), min(t.v) AS min(DISTINCT t.v)]] 03)----TableScan: t projection=[g, v] # The gate covers only the count. A plan that already qualified through sum, @@ -293,7 +293,7 @@ logical_plan 02)--Projection: t.g, CASE WHEN sum(alias2) IS NOT NULL THEN sum(alias2) ELSE Int64(0) END AS records, count(alias1) AS distinct_x 03)----Filter: CASE WHEN sum(alias2) IS NOT NULL THEN sum(alias2) ELSE Int64(0) END > Int64(1) AND count(alias1) > Int64(0) 04)------Aggregate: groupBy=[[t.g]], aggr=[[sum(alias2), count(alias1)]] -05)--------Aggregate: groupBy=[[t.g, t.x AS alias1]], aggr=[[count(Int64(1)) AS alias2]] +05)--------Aggregate: groupBy=[[t.g, t.x AS alias1]], aggr=[[count() AS alias2]] 06)----------TableScan: t projection=[g, x] statement ok diff --git a/datafusion/sqllogictest/test_files/subquery.slt b/datafusion/sqllogictest/test_files/subquery.slt index ca0b0b1ff8719..994ec93d8d37f 100644 --- a/datafusion/sqllogictest/test_files/subquery.slt +++ b/datafusion/sqllogictest/test_files/subquery.slt @@ -541,7 +541,7 @@ logical_plan 03)----Subquery: 04)------Projection: count(Int64(1)) AS count(*) 05)--------Filter: sum(outer_ref(t1.t1_int) + t2.t2_id) > Int64(0) -06)----------Aggregate: groupBy=[[]], aggr=[[count(Int64(1)), sum(CAST(outer_ref(t1.t1_int) + t2.t2_id AS Int64))]] +06)----------Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1)), sum(CAST(outer_ref(t1.t1_int) + t2.t2_id AS Int64))]] 07)------------Filter: outer_ref(t1.t1_name) = t2.t2_name 08)--------------TableScan: t2 09)----TableScan: t1 projection=[t1_id, t1_name, t1_int] @@ -725,7 +725,7 @@ logical_plan 01)Projection: () AS b 02)--Subquery: 03)----Projection: count(Int64(1)) AS count(*) -04)------Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +04)------Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 05)--------TableScan: t1 projection=[] 06)--EmptyRelation: rows=1 @@ -739,10 +739,10 @@ logical_plan 01)Projection: () AS b, () 02)--Subquery: 03)----Projection: count(Int64(1)) AS count(*) -04)------Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +04)------Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 05)--------TableScan: t1 projection=[] 06)--Subquery: -07)----Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +07)----Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 08)------TableScan: t2 projection=[] 09)--EmptyRelation: rows=1 @@ -808,10 +808,10 @@ logical_plan 01)Projection: () AS b, () 02)--Subquery: 03)----Projection: count(Int64(1)) AS count(*) -04)------Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +04)------Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 05)--------TableScan: t1 projection=[] 06)--Subquery: -07)----Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +07)----Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 08)------TableScan: t2 projection=[] 09)--EmptyRelation: rows=1 physical_plan @@ -841,7 +841,7 @@ logical_plan 03)----TableScan: t1 projection=[t1_id, t1_int] 04)----SubqueryAlias: __scalar_sq_1 05)------Projection: count(Int64(1)) AS count(*), t2.t2_int, Boolean(true) AS __always_true -06)--------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count(Int64(1))]] +06)--------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count() AS count(Int64(1))]] 07)----------TableScan: t2 projection=[t2_int] query II rowsort @@ -879,7 +879,7 @@ logical_plan 03)----TableScan: t1 projection=[t1_id, t1_int] 04)----SubqueryAlias: __scalar_sq_1 05)------Projection: count(Int64(1)) AS count(*), t2.t2_int, Boolean(true) AS __always_true -06)--------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count(Int64(1))]] +06)--------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count() AS count(Int64(1))]] 07)----------TableScan: t2 projection=[t2_int] query II rowsort @@ -962,7 +962,7 @@ logical_plan 03)----TableScan: t1 projection=[t1_id, t1_int] 04)----SubqueryAlias: __scalar_sq_1 05)------Projection: count(Int64(1)) AS _cnt, t2.t2_int, Boolean(true) AS __always_true -06)--------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count(Int64(1))]] +06)--------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count() AS count(Int64(1))]] 07)----------TableScan: t2 projection=[t2_int] query II rowsort @@ -983,7 +983,7 @@ logical_plan 03)----TableScan: t1 projection=[t1_id, t1_int] 04)----SubqueryAlias: __scalar_sq_1 05)------Projection: count(Int64(1)) + Int64(2) AS _cnt, t2.t2_int, Boolean(true) AS __always_true -06)--------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count(Int64(1))]] +06)--------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count() AS count(Int64(1))]] 07)----------TableScan: t2 projection=[t2_int] query II rowsort @@ -1006,7 +1006,7 @@ logical_plan 05)--------TableScan: t1 projection=[t1_id, t1_int] 06)--------SubqueryAlias: __scalar_sq_1 07)----------Projection: count(Int64(1)) AS count(*), t2.t2_id, Boolean(true) AS __always_true -08)------------Aggregate: groupBy=[[t2.t2_id]], aggr=[[count(Int64(1))]] +08)------------Aggregate: groupBy=[[t2.t2_id]], aggr=[[count() AS count(Int64(1))]] 09)--------------TableScan: t2 projection=[t2_id] query I rowsort @@ -1028,7 +1028,7 @@ logical_plan 04)----SubqueryAlias: __scalar_sq_1 05)------Projection: count(Int64(1)) AS count(*) + Int64(2) AS cnt_plus_2, t2.t2_int 06)--------Filter: count(Int64(1)) > Int64(1) -07)----------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count(Int64(1))]] +07)----------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count() AS count(Int64(1))]] 08)------------TableScan: t2 projection=[t2_int] query II rowsort @@ -1050,7 +1050,7 @@ logical_plan 03)----TableScan: t1 projection=[t1_id, t1_int] 04)----SubqueryAlias: __scalar_sq_1 05)------Projection: count(Int64(1)) + Int64(2) AS cnt_plus_2, t2.t2_int, count(Int64(1)), Boolean(true) AS __always_true -06)--------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count(Int64(1))]] +06)--------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count() AS count(Int64(1))]] 07)----------TableScan: t2 projection=[t2_int] query II rowsort @@ -1074,7 +1074,7 @@ logical_plan 06)----------TableScan: t1 projection=[t1_int] 07)--------SubqueryAlias: __scalar_sq_1 08)----------Projection: count(Int64(1)) AS count(*), t2.t2_int, Boolean(true) AS __always_true -09)------------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count(Int64(1))]] +09)------------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count() AS count(Int64(1))]] 10)--------------TableScan: t2 projection=[t2_int] query I rowsort @@ -1095,7 +1095,7 @@ logical_plan 05)--------TableScan: t1 projection=[t1_int] 06)--------SubqueryAlias: __scalar_sq_1 07)----------Projection: count(Int64(1)) AS cnt, t2.t2_int, Boolean(true) AS __always_true -08)------------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count(Int64(1))]] +08)------------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count() AS count(Int64(1))]] 09)--------------TableScan: t2 projection=[t2_int] @@ -1125,7 +1125,7 @@ logical_plan 05)--------TableScan: t1 projection=[t1_int] 06)--------SubqueryAlias: __scalar_sq_1 07)----------Projection: count(Int64(1)) + Int64(1) + Int64(1) AS cnt_plus_two, t2.t2_int, count(Int64(1)), Boolean(true) AS __always_true -08)------------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count(Int64(1))]] +08)------------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count() AS count(Int64(1))]] 09)--------------TableScan: t2 projection=[t2_int] query I rowsort @@ -1154,7 +1154,7 @@ logical_plan 05)--------TableScan: t1 projection=[t1_int] 06)--------SubqueryAlias: __scalar_sq_1 07)----------Projection: CASE WHEN count(Int64(1)) = Int64(1) THEN Int64(NULL) ELSE count(Int64(1)) END AS cnt, t2.t2_int, Boolean(true) AS __always_true -08)------------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count(Int64(1))]] +08)------------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count() AS count(Int64(1))]] 09)--------------TableScan: t2 projection=[t2_int] diff --git a/datafusion/sqllogictest/test_files/tpch/plans/q1.slt.part b/datafusion/sqllogictest/test_files/tpch/plans/q1.slt.part index b227f94553e2f..5d7670090c744 100644 --- a/datafusion/sqllogictest/test_files/tpch/plans/q1.slt.part +++ b/datafusion/sqllogictest/test_files/tpch/plans/q1.slt.part @@ -42,7 +42,7 @@ explain select logical_plan 01)Sort: lineitem.l_returnflag ASC NULLS LAST, lineitem.l_linestatus ASC NULLS LAST 02)--Projection: lineitem.l_returnflag, lineitem.l_linestatus, sum(lineitem.l_quantity) AS sum_qty, sum(lineitem.l_extendedprice) AS sum_base_price, sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount) AS sum_disc_price, sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount * Int64(1) + lineitem.l_tax) AS sum_charge, avg(lineitem.l_quantity) AS avg_qty, avg(lineitem.l_extendedprice) AS avg_price, avg(lineitem.l_discount) AS avg_disc, count(Int64(1)) AS count(*) AS count_order -03)----Aggregate: groupBy=[[lineitem.l_returnflag, lineitem.l_linestatus]], aggr=[[sum(lineitem.l_quantity), sum(lineitem.l_extendedprice), sum(__common_expr_1) AS sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount), sum(__common_expr_1 * (Decimal128(1,20,0) + lineitem.l_tax)) AS sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount * Int64(1) + lineitem.l_tax), avg(lineitem.l_quantity), avg(lineitem.l_extendedprice), avg(lineitem.l_discount), count(Int64(1))]] +03)----Aggregate: groupBy=[[lineitem.l_returnflag, lineitem.l_linestatus]], aggr=[[sum(lineitem.l_quantity), sum(lineitem.l_extendedprice), sum(__common_expr_1) AS sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount), sum(__common_expr_1 * (Decimal128(1,20,0) + lineitem.l_tax)) AS sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount * Int64(1) + lineitem.l_tax), avg(lineitem.l_quantity), avg(lineitem.l_extendedprice), avg(lineitem.l_discount), count() AS count(Int64(1))]] 04)------Projection: lineitem.l_extendedprice * (Decimal128(1,20,0) - lineitem.l_discount) AS __common_expr_1, lineitem.l_quantity, lineitem.l_extendedprice, lineitem.l_discount, lineitem.l_tax, lineitem.l_returnflag, lineitem.l_linestatus 05)--------Filter: lineitem.l_shipdate <= Date32("1998-09-02") 06)----------TableScan: lineitem projection=[l_quantity, l_extendedprice, l_discount, l_tax, l_returnflag, l_linestatus, l_shipdate], partial_filters=[lineitem.l_shipdate <= Date32("1998-09-02")] @@ -50,9 +50,9 @@ physical_plan 01)SortPreservingMergeExec: [l_returnflag@0 ASC NULLS LAST, l_linestatus@1 ASC NULLS LAST] 02)--ProjectionExec: expr=[l_returnflag@0 as l_returnflag, l_linestatus@1 as l_linestatus, sum(lineitem.l_quantity)@2 as sum_qty, sum(lineitem.l_extendedprice)@3 as sum_base_price, sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount)@4 as sum_disc_price, sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount * Int64(1) + lineitem.l_tax)@5 as sum_charge, avg(lineitem.l_quantity)@6 as avg_qty, avg(lineitem.l_extendedprice)@7 as avg_price, avg(lineitem.l_discount)@8 as avg_disc, count(Int64(1))@9 as count_order] 03)----SortExec: expr=[l_returnflag@0 ASC NULLS LAST, l_linestatus@1 ASC NULLS LAST], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[l_returnflag@0 as l_returnflag, l_linestatus@1 as l_linestatus], aggr=[sum(lineitem.l_quantity), sum(lineitem.l_extendedprice), sum(__common_expr_1) as sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount), sum(__common_expr_1 * 1 + lineitem.l_tax) as sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount * Int64(1) + lineitem.l_tax), avg(lineitem.l_quantity), avg(lineitem.l_extendedprice), avg(lineitem.l_discount), count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[l_returnflag@0 as l_returnflag, l_linestatus@1 as l_linestatus], aggr=[sum(lineitem.l_quantity), sum(lineitem.l_extendedprice), sum(__common_expr_1) as sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount), sum(__common_expr_1 * 1 + lineitem.l_tax) as sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount * Int64(1) + lineitem.l_tax), avg(lineitem.l_quantity), avg(lineitem.l_extendedprice), avg(lineitem.l_discount), count() as count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([l_returnflag@0, l_linestatus@1], 4), input_partitions=4 -06)----------AggregateExec: mode=Partial, gby=[l_returnflag@5 as l_returnflag, l_linestatus@6 as l_linestatus], aggr=[sum(lineitem.l_quantity), sum(lineitem.l_extendedprice), sum(__common_expr_1) as sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount), sum(__common_expr_1 * 1 + lineitem.l_tax) as sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount * Int64(1) + lineitem.l_tax), avg(lineitem.l_quantity), avg(lineitem.l_extendedprice), avg(lineitem.l_discount), count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[l_returnflag@5 as l_returnflag, l_linestatus@6 as l_linestatus], aggr=[sum(lineitem.l_quantity), sum(lineitem.l_extendedprice), sum(__common_expr_1) as sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount), sum(__common_expr_1 * 1 + lineitem.l_tax) as sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount * Int64(1) + lineitem.l_tax), avg(lineitem.l_quantity), avg(lineitem.l_extendedprice), avg(lineitem.l_discount), count() as count(Int64(1))] 07)------------ProjectionExec: expr=[l_extendedprice@0 * (1 - l_discount@1) as __common_expr_1, l_quantity@2 as l_quantity, l_extendedprice@0 as l_extendedprice, l_discount@1 as l_discount, l_tax@3 as l_tax, l_returnflag@4 as l_returnflag, l_linestatus@5 as l_linestatus] 08)--------------FilterExec: l_shipdate@6 <= 1998-09-02, projection=[l_extendedprice@1, l_discount@2, l_quantity@0, l_tax@3, l_returnflag@4, l_linestatus@5] 09)----------------DataSourceExec: file_groups={4 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/tpch/data/lineitem.tbl:0..18561749], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/tpch/data/lineitem.tbl:18561749..37123498], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/tpch/data/lineitem.tbl:37123498..55685247], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/tpch/data/lineitem.tbl:55685247..74246996]]}, projection=[l_quantity, l_extendedprice, l_discount, l_tax, l_returnflag, l_linestatus, l_shipdate], constraints=[PrimaryKey([0, 3])], file_type=csv, has_header=false diff --git a/datafusion/sqllogictest/test_files/tpch/plans/q13.slt.part b/datafusion/sqllogictest/test_files/tpch/plans/q13.slt.part index 9f9cbb3b6af68..d5b36f1398809 100644 --- a/datafusion/sqllogictest/test_files/tpch/plans/q13.slt.part +++ b/datafusion/sqllogictest/test_files/tpch/plans/q13.slt.part @@ -42,7 +42,7 @@ limit 10; logical_plan 01)Sort: custdist DESC NULLS FIRST, c_orders.c_count DESC NULLS FIRST, fetch=10 02)--Projection: c_orders.c_count, count(Int64(1)) AS count(*) AS custdist -03)----Aggregate: groupBy=[[c_orders.c_count]], aggr=[[count(Int64(1))]] +03)----Aggregate: groupBy=[[c_orders.c_count]], aggr=[[count() AS count(Int64(1))]] 04)------SubqueryAlias: c_orders 05)--------Projection: count(orders.o_orderkey) AS c_count 06)----------Aggregate: groupBy=[[customer.c_custkey]], aggr=[[count(orders.o_orderkey)]] @@ -56,9 +56,9 @@ physical_plan 01)SortPreservingMergeExec: [custdist@1 DESC, c_count@0 DESC], fetch=10 02)--ProjectionExec: expr=[c_count@0 as c_count, count(Int64(1))@1 as custdist] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@1 DESC, c_count@0 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[c_count@0 as c_count], aggr=[count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[c_count@0 as c_count], aggr=[count() as count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([c_count@0], 4), input_partitions=4 -06)----------AggregateExec: mode=Partial, gby=[c_count@0 as c_count], aggr=[count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[c_count@0 as c_count], aggr=[count() as count(Int64(1))] 07)------------ProjectionExec: expr=[count(orders.o_orderkey)@1 as c_count] 08)--------------AggregateExec: mode=SinglePartitioned, gby=[c_custkey@0 as c_custkey], aggr=[count(orders.o_orderkey)] 09)----------------HashJoinExec: mode=Partitioned, join_type=Left, on=[(c_custkey@0, o_custkey@1)], projection=[c_custkey@0, o_orderkey@1] diff --git a/datafusion/sqllogictest/test_files/tpch/plans/q21.slt.part b/datafusion/sqllogictest/test_files/tpch/plans/q21.slt.part index 47e5d6d888dc5..c4794d150eea3 100644 --- a/datafusion/sqllogictest/test_files/tpch/plans/q21.slt.part +++ b/datafusion/sqllogictest/test_files/tpch/plans/q21.slt.part @@ -60,7 +60,7 @@ order by logical_plan 01)Sort: numwait DESC NULLS FIRST, supplier.s_name ASC NULLS LAST 02)--Projection: supplier.s_name, count(Int64(1)) AS count(*) AS numwait -03)----Aggregate: groupBy=[[supplier.s_name]], aggr=[[count(Int64(1))]] +03)----Aggregate: groupBy=[[supplier.s_name]], aggr=[[count() AS count(Int64(1))]] 04)------Projection: supplier.s_name 05)--------LeftAnti Join: l1.l_orderkey = __correlated_sq_2.l_orderkey Filter: __correlated_sq_2.l_suppkey != l1.l_suppkey 06)----------LeftSemi Join: l1.l_orderkey = __correlated_sq_1.l_orderkey Filter: __correlated_sq_1.l_suppkey != l1.l_suppkey @@ -92,9 +92,9 @@ physical_plan 01)SortPreservingMergeExec: [numwait@1 DESC, s_name@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[s_name@0 as s_name, count(Int64(1))@1 as numwait] 03)----SortExec: expr=[count(Int64(1))@1 DESC, s_name@0 ASC NULLS LAST], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[s_name@0 as s_name], aggr=[count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[s_name@0 as s_name], aggr=[count() as count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([s_name@0], 4), input_partitions=4 -06)----------AggregateExec: mode=Partial, gby=[s_name@0 as s_name], aggr=[count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[s_name@0 as s_name], aggr=[count() as count(Int64(1))] 07)------------HashJoinExec: mode=Partitioned, join_type=LeftAnti, on=[(l_orderkey@1, l_orderkey@0)], filter=l_suppkey@1 != l_suppkey@0, projection=[s_name@0] 08)--------------HashJoinExec: mode=Partitioned, join_type=LeftSemi, on=[(l_orderkey@1, l_orderkey@0)], filter=l_suppkey@1 != l_suppkey@0 09)----------------RepartitionExec: partitioning=Hash([l_orderkey@1], 4), input_partitions=4 diff --git a/datafusion/sqllogictest/test_files/tpch/plans/q22.slt.part b/datafusion/sqllogictest/test_files/tpch/plans/q22.slt.part index d3f27021f1781..44933c238cc7d 100644 --- a/datafusion/sqllogictest/test_files/tpch/plans/q22.slt.part +++ b/datafusion/sqllogictest/test_files/tpch/plans/q22.slt.part @@ -58,7 +58,7 @@ order by logical_plan 01)Sort: custsale.cntrycode ASC NULLS LAST 02)--Projection: custsale.cntrycode, count(Int64(1)) AS count(*) AS numcust, sum(custsale.c_acctbal) AS totacctbal -03)----Aggregate: groupBy=[[custsale.cntrycode]], aggr=[[count(Int64(1)), sum(custsale.c_acctbal)]] +03)----Aggregate: groupBy=[[custsale.cntrycode]], aggr=[[count() AS count(Int64(1)), sum(custsale.c_acctbal)]] 04)------SubqueryAlias: custsale 05)--------Projection: substr(customer.c_phone, Int64(1), Int64(2)) AS cntrycode, customer.c_acctbal 06)----------LeftAnti Join: customer.c_custkey = __correlated_sq_1.o_custkey @@ -76,9 +76,9 @@ physical_plan 02)--SortPreservingMergeExec: [cntrycode@0 ASC NULLS LAST] 03)----ProjectionExec: expr=[cntrycode@0 as cntrycode, count(Int64(1))@1 as numcust, sum(custsale.c_acctbal)@2 as totacctbal] 04)------SortExec: expr=[cntrycode@0 ASC NULLS LAST], preserve_partitioning=[true] -05)--------AggregateExec: mode=FinalPartitioned, gby=[cntrycode@0 as cntrycode], aggr=[count(Int64(1)), sum(custsale.c_acctbal)] +05)--------AggregateExec: mode=FinalPartitioned, gby=[cntrycode@0 as cntrycode], aggr=[count() as count(Int64(1)), sum(custsale.c_acctbal)] 06)----------RepartitionExec: partitioning=Hash([cntrycode@0], 4), input_partitions=4 -07)------------AggregateExec: mode=Partial, gby=[cntrycode@0 as cntrycode], aggr=[count(Int64(1)), sum(custsale.c_acctbal)] +07)------------AggregateExec: mode=Partial, gby=[cntrycode@0 as cntrycode], aggr=[count() as count(Int64(1)), sum(custsale.c_acctbal)] 08)--------------ProjectionExec: expr=[substr(c_phone@0, 1, 2) as cntrycode, c_acctbal@1 as c_acctbal] 09)----------------HashJoinExec: mode=Partitioned, join_type=LeftAnti, on=[(c_custkey@0, o_custkey@0)], projection=[c_phone@1, c_acctbal@2] 10)------------------RepartitionExec: partitioning=Hash([c_custkey@0], 4), input_partitions=4 diff --git a/datafusion/sqllogictest/test_files/tpch/plans/q4.slt.part b/datafusion/sqllogictest/test_files/tpch/plans/q4.slt.part index 470d7a6527a52..6c48abe9fb888 100644 --- a/datafusion/sqllogictest/test_files/tpch/plans/q4.slt.part +++ b/datafusion/sqllogictest/test_files/tpch/plans/q4.slt.part @@ -42,7 +42,7 @@ order by logical_plan 01)Sort: orders.o_orderpriority ASC NULLS LAST 02)--Projection: orders.o_orderpriority, count(Int64(1)) AS count(*) AS order_count -03)----Aggregate: groupBy=[[orders.o_orderpriority]], aggr=[[count(Int64(1))]] +03)----Aggregate: groupBy=[[orders.o_orderpriority]], aggr=[[count() AS count(Int64(1))]] 04)------Projection: orders.o_orderpriority 05)--------LeftSemi Join: orders.o_orderkey = __correlated_sq_1.l_orderkey 06)----------Projection: orders.o_orderkey, orders.o_orderpriority @@ -56,9 +56,9 @@ physical_plan 01)SortPreservingMergeExec: [o_orderpriority@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[o_orderpriority@0 as o_orderpriority, count(Int64(1))@1 as order_count] 03)----SortExec: expr=[o_orderpriority@0 ASC NULLS LAST], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[o_orderpriority@0 as o_orderpriority], aggr=[count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[o_orderpriority@0 as o_orderpriority], aggr=[count() as count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([o_orderpriority@0], 4), input_partitions=4 -06)----------AggregateExec: mode=Partial, gby=[o_orderpriority@0 as o_orderpriority], aggr=[count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[o_orderpriority@0 as o_orderpriority], aggr=[count() as count(Int64(1))] 07)------------HashJoinExec: mode=Partitioned, join_type=LeftSemi, on=[(o_orderkey@0, l_orderkey@0)], projection=[o_orderpriority@1] 08)--------------RepartitionExec: partitioning=Hash([o_orderkey@0], 4), input_partitions=4 09)----------------FilterExec: o_orderdate@1 >= 1993-07-01 AND o_orderdate@1 < 1993-10-01, projection=[o_orderkey@0, o_orderpriority@2] diff --git a/datafusion/sqllogictest/test_files/union.slt b/datafusion/sqllogictest/test_files/union.slt index 115a010330103..1a8e758262d33 100644 --- a/datafusion/sqllogictest/test_files/union.slt +++ b/datafusion/sqllogictest/test_files/union.slt @@ -600,7 +600,7 @@ SELECT count(*) FROM ( ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[name]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[name]], aggr=[[count() AS count(Int64(1))]] 03)----Union 04)------Aggregate: groupBy=[[t1.name]], aggr=[[]] 05)--------TableScan: t1 projection=[name] @@ -608,7 +608,7 @@ logical_plan 07)--------TableScan: t2 projection=[name] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@1 as count(*)] -02)--AggregateExec: mode=SinglePartitioned, gby=[name@0 as name], aggr=[count(Int64(1))] +02)--AggregateExec: mode=SinglePartitioned, gby=[name@0 as name], aggr=[count() as count(Int64(1))] 03)----InterleaveExec 04)------AggregateExec: mode=FinalPartitioned, gby=[name@0 as name], aggr=[] 05)--------RepartitionExec: partitioning=Hash([name@0], 4), input_partitions=4 @@ -642,7 +642,7 @@ logical_plan 02)--Union 03)----Projection: count(Int64(1)) AS count(*) AS cnt 04)------Limit: skip=0, fetch=3 -05)--------Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +05)--------Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 06)----------SubqueryAlias: a 07)------------Projection: 08)--------------Aggregate: groupBy=[[aggregate_test_100.c1]], aggr=[[]] @@ -663,9 +663,9 @@ physical_plan 02)--UnionExec 03)----ProjectionExec: expr=[CAST(count(Int64(1))@0 AS Int64) as cnt] 04)------GlobalLimitExec: skip=0, fetch=3 -05)--------AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] +05)--------AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] 06)----------CoalescePartitionsExec -07)------------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] +07)------------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] 08)--------------ProjectionExec: expr=[] 09)----------------AggregateExec: mode=FinalPartitioned, gby=[c1@0 as c1], aggr=[] 10)------------------RepartitionExec: partitioning=Hash([c1@0], 4), input_partitions=4 @@ -799,7 +799,7 @@ select x, y from (select 1 as x , max(10) as y) b logical_plan 01)Union 02)--Projection: count(Int64(1)) AS count(*) AS count, a.n -03)----Aggregate: groupBy=[[a.n]], aggr=[[count(Int64(1))]] +03)----Aggregate: groupBy=[[a.n]], aggr=[[count() AS count(Int64(1))]] 04)------SubqueryAlias: a 05)--------Projection: Int64(5) AS n 06)----------EmptyRelation: rows=1 @@ -811,7 +811,7 @@ logical_plan physical_plan 01)UnionExec 02)--ProjectionExec: expr=[count(Int64(1))@1 as count, CAST(n@0 AS Int64) as n] -03)----AggregateExec: mode=SinglePartitioned, gby=[n@0 as n], aggr=[count(Int64(1))] +03)----AggregateExec: mode=SinglePartitioned, gby=[n@0 as n], aggr=[count() as count(Int64(1))] 04)------ProjectionExec: expr=[5 as n] 05)--------PlaceholderRowExec 06)--ProjectionExec: expr=[1 as count, max(Int64(10))@0 as n] diff --git a/datafusion/sqllogictest/test_files/window.slt b/datafusion/sqllogictest/test_files/window.slt index af12fad670796..fa65f8902709e 100644 --- a/datafusion/sqllogictest/test_files/window.slt +++ b/datafusion/sqllogictest/test_files/window.slt @@ -1868,7 +1868,7 @@ EXPLAIN SELECT count(*) as global_count FROM ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) AS global_count -02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] 03)----SubqueryAlias: a 04)------Projection: 05)--------Aggregate: groupBy=[[aggregate_test_100.c1]], aggr=[[]] @@ -1877,9 +1877,9 @@ logical_plan 08)--------------TableScan: aggregate_test_100 projection=[c1, c13], partial_filters=[aggregate_test_100.c13 != Utf8View("C2GT5KVyOPZpgKVl110TyZO0NcJ434")] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as global_count] -02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] 05)--------ProjectionExec: expr=[] 06)----------AggregateExec: mode=FinalPartitioned, gby=[c1@0 as c1], aggr=[] 07)------------RepartitionExec: partitioning=Hash([c1@0], 2), input_partitions=2 @@ -7277,7 +7277,7 @@ statement error DataFusion error: This feature is not implemented: Unsupported O SELECT a FROM window_in_filter_t OFFSET row_number() OVER (); # ... and the same for an aggregate function -statement error DataFusion error: This feature is not implemented: Unsupported LIMIT expression: count\(Int64\(1\)\) AS count\(\*\) +statement error DataFusion error: This feature is not implemented: Unsupported LIMIT expression: count\(\) AS count\(\*\) SELECT a FROM window_in_filter_t LIMIT count(*); statement ok diff --git a/datafusion/substrait/tests/cases/roundtrip_logical_plan.rs b/datafusion/substrait/tests/cases/roundtrip_logical_plan.rs index 6d783af16e461..38fdef982f015 100644 --- a/datafusion/substrait/tests/cases/roundtrip_logical_plan.rs +++ b/datafusion/substrait/tests/cases/roundtrip_logical_plan.rs @@ -1328,7 +1328,7 @@ async fn simple_intersect() -> Result<()> { async fn check_wildcard(syntax: &str) -> Result<()> { let expected_plan_str = format!( "Projection: count(Int64(1)) AS {syntax}\ - \n Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]]\ + \n Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]]\ \n Projection:\ \n LeftSemi Join: data.a = data2.a\ \n Aggregate: groupBy=[[data.a]], aggr=[[]]\ @@ -1362,13 +1362,9 @@ async fn simple_intersect() -> Result<()> { check_wildcard("count(*)").await?; check_wildcard("count()").await?; - check_constant("count(1)", "count(Int64(1))").await?; - check_constant("count(2)", "count(Int64(2))").await?; - check_constant( - "count(1 + 2)", - "count(Int64(3)) AS count(Int64(1) + Int64(2))", - ) - .await?; + check_constant("count(1)", "count() AS count(Int64(1))").await?; + check_constant("count(2)", "count() AS count(Int64(2))").await?; + check_constant("count(1 + 2)", "count() AS count(Int64(1) + Int64(2))").await?; Ok(()) } @@ -1546,7 +1542,7 @@ async fn simple_intersect_table_reuse() -> Result<()> { async fn check_wildcard(syntax: &str) -> Result<()> { let expected_plan_str = format!( "Projection: count(Int64(1)) AS {syntax}\ - \n Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]]\ + \n Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]]\ \n Projection:\ \n LeftSemi Join: left.a = right.a\ \n SubqueryAlias: left\ @@ -1584,13 +1580,9 @@ async fn simple_intersect_table_reuse() -> Result<()> { check_wildcard("count(*)").await?; check_wildcard("count()").await?; - check_constant("count(1)", "count(Int64(1))").await?; - check_constant("count(2)", "count(Int64(2))").await?; - check_constant( - "count(1 + 2)", - "count(Int64(3)) AS count(Int64(1) + Int64(2))", - ) - .await?; + check_constant("count(1)", "count() AS count(Int64(1))").await?; + check_constant("count(2)", "count() AS count(Int64(2))").await?; + check_constant("count(1 + 2)", "count() AS count(Int64(1) + Int64(2))").await?; Ok(()) } From f1fd6ee2edc7ad0effbe5e8ad461365dda82cb5d Mon Sep 17 00:00:00 2001 From: wudidapaopao <664920313@qq.com> Date: Tue, 29 Sep 2026 04:28:21 +0800 Subject: [PATCH 2/6] test: cover nullary count statistics --- datafusion/functions-aggregate/src/count.rs | 21 +++++++++++++++++++++ 1 file changed, 21 insertions(+) diff --git a/datafusion/functions-aggregate/src/count.rs b/datafusion/functions-aggregate/src/count.rs index 83abb5f3c0c90..6ecca023e9fbb 100644 --- a/datafusion/functions-aggregate/src/count.rs +++ b/datafusion/functions-aggregate/src/count.rs @@ -1091,6 +1091,27 @@ mod tests { Ok(()) } + #[test] + fn count_nullary_value_from_stats() { + let statistics = datafusion_common::Statistics { + num_rows: Precision::Exact(42), + total_byte_size: Precision::Absent, + column_statistics: vec![], + }; + let return_type = DataType::Int64; + let statistics_args = StatisticsArgs { + statistics: &statistics, + return_type: &return_type, + is_distinct: false, + exprs: &[], + }; + + assert_eq!( + Count::new().value_from_stats(&statistics_args), + Some(ScalarValue::Int64(Some(42))) + ); + } + #[test] fn count_groups_accumulator_nullary() -> Result<()> { let mut accumulator = CountGroupsAccumulator::new(); From 2ebda14e123e7d5177b6d9293b1833660438241e Mon Sep 17 00:00:00 2001 From: wudidapaopao <664920313@qq.com> Date: Sat, 3 Oct 2026 15:33:28 +0800 Subject: [PATCH 3/6] refactor: split non-nullable count rewrite --- .../tests/datasource/object_store_access.rs | 4 +- datafusion/core/tests/sql/explain_analyze.rs | 4 +- datafusion/core/tests/sql/unparser.rs | 5 +- datafusion/functions-aggregate/src/count.rs | 48 +++++-------------- .../simplify_expressions/simplify_exprs.rs | 2 +- .../test_files/aggregate_memory_spill.slt | 2 +- .../sqllogictest/test_files/group_by.slt | 21 ++++---- 7 files changed, 30 insertions(+), 56 deletions(-) diff --git a/datafusion/core/tests/datasource/object_store_access.rs b/datafusion/core/tests/datasource/object_store_access.rs index 9d51834f28719..16d894bde1303 100644 --- a/datafusion/core/tests/datasource/object_store_access.rs +++ b/datafusion/core/tests/datasource/object_store_access.rs @@ -866,8 +866,8 @@ async fn query_single_parquet_file() { RequestCountingObjectStore() Total Requests: 3 - GET (opts) path=parquet_table.parquet head=true - - GET (ranges) path=parquet_table.parquet ranges=4-534 - - GET (ranges) path=parquet_table.parquet ranges=1064-1594 + - GET (ranges) path=parquet_table.parquet ranges=4-534,534-1064 + - GET (ranges) path=parquet_table.parquet ranges=1064-1594,1594-2124 " ); } diff --git a/datafusion/core/tests/sql/explain_analyze.rs b/datafusion/core/tests/sql/explain_analyze.rs index e1dc9badb9c9f..28fcc6eee4b11 100644 --- a/datafusion/core/tests/sql/explain_analyze.rs +++ b/datafusion/core/tests/sql/explain_analyze.rs @@ -1137,7 +1137,7 @@ async fn explain_analyze_aggregate_metrics_map_indices_to_expressions() { .to_string(); assert_contains!( normal.as_str(), - "aggr=[sum(aggregate_test_100.c5), sum(aggregate_test_100.c6), count() as count(aggregate_test_100.c7)]" + "aggr=[sum(aggregate_test_100.c5), sum(aggregate_test_100.c6), count(aggregate_test_100.c7)]" ); assert_contains!(normal.as_str(), "agg_expr_0_arguments_time"); assert_contains!(normal.as_str(), "agg_expr_1_arguments_time"); @@ -1159,7 +1159,7 @@ async fn explain_analyze_aggregate_metrics_map_indices_to_expressions() { ); assert_contains!( verbose.as_str(), - "agg_expr_2_arguments_time{partition=0, aggregate=count() as count(aggregate_test_100.c7)}" + "agg_expr_2_arguments_time{partition=0, aggregate=count(aggregate_test_100.c7)}" ); } diff --git a/datafusion/core/tests/sql/unparser.rs b/datafusion/core/tests/sql/unparser.rs index 46c81f133dee0..b77f53cb46932 100644 --- a/datafusion/core/tests/sql/unparser.rs +++ b/datafusion/core/tests/sql/unparser.rs @@ -567,8 +567,9 @@ async fn optimized_duckdb_unparse_preserves_nested_aggregate_scope() -> Result<( assert!( sql.contains(concat!( - r#"FROM (SELECT date_part('year', "signup_date") AS "group_alias_0", "#, - r#"sum("total_revenue") AS "alias2" "# + r#"FROM (SELECT sum("total_revenue") AS "alias2", "#, + r#"date_part('year', "signup_date") AS "group_alias_0", "#, + r#""customer_id" AS "alias1" "# )), "inner aggregate should define the aliases before the outer aggregate uses them: {sql}", ); diff --git a/datafusion/functions-aggregate/src/count.rs b/datafusion/functions-aggregate/src/count.rs index 6ecca023e9fbb..10d8eb5da1357 100644 --- a/datafusion/functions-aggregate/src/count.rs +++ b/datafusion/functions-aggregate/src/count.rs @@ -40,7 +40,6 @@ use datafusion_expr::{ TypeSignature, Volatility, WindowFunctionDefinition, expr::WindowFunction, function::{AccumulatorArgs, AggregateFunctionSimplification, StateFieldsArgs}, - simplify::SimplifyContext, utils::{AggregateOrderSensitivity, format_state_name}, }; use datafusion_functions_aggregate_common::aggregate::count_distinct::PrimitiveDistinctCountGroupsAccumulator; @@ -402,15 +401,13 @@ impl AggregateUDFImpl for Count { } fn simplify(&self) -> Option { - Some(Box::new(|mut aggregate_function, info| { + Some(Box::new(|mut aggregate_function, _| { let params = &aggregate_function.params; - // Every row is counted when none of the arguments can be null if !params.distinct - && !params.args.is_empty() - && params - .args - .iter() - .all(|arg| is_safe_non_null_count_arg(arg, info)) + && matches!( + params.args.as_slice(), + [Expr::Literal(value, _)] if value == &COUNT_STAR_EXPANSION + ) { aggregate_function.params.args.clear(); } @@ -493,16 +490,6 @@ impl AggregateUDFImpl for Count { } } -/// Returns true if `expr` is a non-null literal or non-nullable column that is -/// safe to elide from `COUNT`. -fn is_safe_non_null_count_arg(expr: &Expr, info: &SimplifyContext) -> bool { - match expr { - Expr::Literal(value, _) => !value.is_null(), - Expr::Column(_) => matches!(info.nullable(expr), Ok(false)), - _ => false, - } -} - #[cold] fn create_distinct_count_groups_accumulator( args: &AccumulatorArgs, @@ -1039,7 +1026,6 @@ mod tests { }, datatypes::{DataType, Field, Int32Type, Schema}, }; - use datafusion_common::DFSchema; use datafusion_expr::function::AccumulatorArgs; use datafusion_expr::{col, lit}; use datafusion_physical_expr::{PhysicalExpr, expressions::Column}; @@ -1139,14 +1125,8 @@ mod tests { } #[test] - fn simplify_count_safe_non_null_args() -> Result<()> { - let schema = DFSchema::try_from(Schema::new(vec![ - Field::new("a", DataType::Int32, false), - Field::new("b", DataType::Int32, true), - ]))?; - let info = SimplifyContext::builder() - .with_schema(Arc::new(schema)) - .build(); + fn simplify_count_star() -> Result<()> { + let info = datafusion_expr::simplify::SimplifyContext::default(); let simplify = Count::new().simplify().unwrap(); let simplified_args = |args: Vec, distinct: bool| -> Result> { let aggregate_function = datafusion_expr::expr::AggregateFunction::new_udf( @@ -1163,24 +1143,18 @@ mod tests { } }; + assert!(simplified_args(vec![lit(1_i64)], false)?.is_empty()); for args in [ - vec![lit(1i64)], + vec![lit(ScalarValue::Null)], + vec![lit(2_i64)], vec![lit("x")], vec![col("a")], - vec![col("a"), lit(2)], - ] { - assert!(simplified_args(args, false)?.is_empty()); - } - for args in [ - vec![lit(ScalarValue::Null)], - vec![col("b")], - vec![col("a"), col("b")], vec![col("a") + lit(1)], ] { assert_eq!(simplified_args(args.clone(), false)?, args); } assert!(simplified_args(vec![], false)?.is_empty()); - assert_eq!(simplified_args(vec![lit(1i64)], true)?, vec![lit(1i64)]); + assert_eq!(simplified_args(vec![lit(1_i64)], true)?, vec![lit(1_i64)]); Ok(()) } diff --git a/datafusion/optimizer/src/simplify_expressions/simplify_exprs.rs b/datafusion/optimizer/src/simplify_expressions/simplify_exprs.rs index 65dccc7126e53..41dd24b6d857f 100644 --- a/datafusion/optimizer/src/simplify_expressions/simplify_exprs.rs +++ b/datafusion/optimizer/src/simplify_expressions/simplify_exprs.rs @@ -396,7 +396,7 @@ mod tests { plan, @r" Projection: sum(test.a) + Int64(2) * CAST(count(test.a) AS Int64) AS sum(test.a + Int64(2)), sum(test.a) + Int64(3) * CAST(count(test.a) AS Int64) AS sum(test.a + Int64(3)) - Aggregate: groupBy=[[]], aggr=[[sum(test.a), count() AS count(test.a)]] + Aggregate: groupBy=[[]], aggr=[[sum(test.a), count(test.a)]] TableScan: test " )?; diff --git a/datafusion/sqllogictest/test_files/aggregate_memory_spill.slt b/datafusion/sqllogictest/test_files/aggregate_memory_spill.slt index f8b05506800e8..48622127bce99 100644 --- a/datafusion/sqllogictest/test_files/aggregate_memory_spill.slt +++ b/datafusion/sqllogictest/test_files/aggregate_memory_spill.slt @@ -113,7 +113,7 @@ FROM ( ) ---- -04)------AggregateExec: mode=Single, gby=[group_alias_0@0 as group_alias_0], aggr=[count() as count(alias1)], metrics=[spill_count=7,] +04)------AggregateExec: mode=Single, gby=[group_alias_0@0 as group_alias_0], aggr=[count(alias1)], metrics=[spill_count=7,] # --- Case D: multiple aggregates (sum/min/max) under memory limit --- diff --git a/datafusion/sqllogictest/test_files/group_by.slt b/datafusion/sqllogictest/test_files/group_by.slt index e339204a63fc2..97539ea479d40 100644 --- a/datafusion/sqllogictest/test_files/group_by.slt +++ b/datafusion/sqllogictest/test_files/group_by.slt @@ -4532,16 +4532,16 @@ EXPLAIN SELECT c1, count(distinct c2), min(distinct c2), sum(c3), max(c4) FROM a logical_plan 01)Sort: aggregate_test_100.c1 ASC NULLS LAST 02)--Projection: aggregate_test_100.c1, count(alias1) AS count(DISTINCT aggregate_test_100.c2), min(alias1) AS min(DISTINCT aggregate_test_100.c2), sum(alias2) AS sum(aggregate_test_100.c3), max(alias3) AS max(aggregate_test_100.c4) -03)----Aggregate: groupBy=[[aggregate_test_100.c1]], aggr=[[count() AS count(alias1), min(alias1), sum(alias2), max(alias3)]] +03)----Aggregate: groupBy=[[aggregate_test_100.c1]], aggr=[[count(alias1), min(alias1), sum(alias2), max(alias3)]] 04)------Aggregate: groupBy=[[aggregate_test_100.c1, aggregate_test_100.c2 AS alias1]], aggr=[[sum(CAST(aggregate_test_100.c3 AS Int64)) AS alias2, max(aggregate_test_100.c4) AS alias3]] 05)--------TableScan: aggregate_test_100 projection=[c1, c2, c3, c4] physical_plan 01)SortPreservingMergeExec: [c1@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[c1@0 as c1, count(alias1)@1 as count(DISTINCT aggregate_test_100.c2), min(alias1)@2 as min(DISTINCT aggregate_test_100.c2), sum(alias2)@3 as sum(aggregate_test_100.c3), max(alias3)@4 as max(aggregate_test_100.c4)] 03)----SortExec: expr=[c1@0 ASC NULLS LAST], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[c1@0 as c1], aggr=[count() as count(alias1), min(alias1), sum(alias2), max(alias3)] +04)------AggregateExec: mode=FinalPartitioned, gby=[c1@0 as c1], aggr=[count(alias1), min(alias1), sum(alias2), max(alias3)] 05)--------RepartitionExec: partitioning=Hash([c1@0], 8), input_partitions=8 -06)----------AggregateExec: mode=Partial, gby=[c1@0 as c1], aggr=[count() as count(alias1), min(alias1), sum(alias2), max(alias3)] +06)----------AggregateExec: mode=Partial, gby=[c1@0 as c1], aggr=[count(alias1), min(alias1), sum(alias2), max(alias3)] 07)------------AggregateExec: mode=FinalPartitioned, gby=[c1@0 as c1, alias1@1 as alias1], aggr=[sum(aggregate_test_100.c3) as alias2, max(aggregate_test_100.c4) as alias3] 08)--------------RepartitionExec: partitioning=Hash([c1@0, alias1@1], 8), input_partitions=8 09)----------------AggregateExec: mode=Partial, gby=[c1@0 as c1, c2@1 as alias1], aggr=[sum(aggregate_test_100.c3) as alias2, max(aggregate_test_100.c4) as alias3] @@ -5235,16 +5235,15 @@ GROUP BY ---- logical_plan 01)Projection: date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 120000000000 }"),keywords_stream.ts,Utf8("2000-01-01")) AS ts_chunk, count(keywords_stream.keyword) AS alert_keyword_count -02)--Aggregate: groupBy=[[date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 120000000000 }"), keywords_stream.ts, TimestampNanosecond(946684800000000000, None)) AS date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 120000000000 }"),keywords_stream.ts,Utf8("2000-01-01"))]], aggr=[[count() AS count(keywords_stream.keyword)]] -03)----Projection: keywords_stream.ts -04)------LeftSemi Join: keywords_stream.keyword = __correlated_sq_1.keyword -05)--------TableScan: keywords_stream projection=[ts, keyword] -06)--------SubqueryAlias: __correlated_sq_1 -07)----------TableScan: alert_keywords projection=[keyword] +02)--Aggregate: groupBy=[[date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 120000000000 }"), keywords_stream.ts, TimestampNanosecond(946684800000000000, None)) AS date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 120000000000 }"),keywords_stream.ts,Utf8("2000-01-01"))]], aggr=[[count(keywords_stream.keyword)]] +03)----LeftSemi Join: keywords_stream.keyword = __correlated_sq_1.keyword +04)------TableScan: keywords_stream projection=[ts, keyword] +05)------SubqueryAlias: __correlated_sq_1 +06)--------TableScan: alert_keywords projection=[keyword] physical_plan 01)ProjectionExec: expr=[date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 120000000000 }"),keywords_stream.ts,Utf8("2000-01-01"))@0 as ts_chunk, count(keywords_stream.keyword)@1 as alert_keyword_count] -02)--AggregateExec: mode=Single, gby=[date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 120000000000 }, ts@0, 946684800000000000) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 120000000000 }"),keywords_stream.ts,Utf8("2000-01-01"))], aggr=[count() as count(keywords_stream.keyword)] -03)----HashJoinExec: mode=CollectLeft, join_type=RightSemi, on=[(keyword@0, keyword@1)], projection=[ts@0] +02)--AggregateExec: mode=Single, gby=[date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 120000000000 }, ts@0, 946684800000000000) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 120000000000 }"),keywords_stream.ts,Utf8("2000-01-01"))], aggr=[count(keywords_stream.keyword)] +03)----HashJoinExec: mode=CollectLeft, join_type=RightSemi, on=[(keyword@0, keyword@1)] 04)------DataSourceExec: partitions=1, partition_sizes=[3] 05)------DataSourceExec: partitions=1, partition_sizes=[3] From dcf212bbd7b2adc704ab362fa7a7b7b4eec82f2a Mon Sep 17 00:00:00 2001 From: wudidapaopao <664920313@qq.com> Date: Mon, 5 Oct 2026 02:38:51 +0800 Subject: [PATCH 4/6] feat: cache constant aggregate arguments --- datafusion/core/src/physical_planner.rs | 4 +- datafusion/core/tests/dataframe/mod.rs | 114 +++++------ datafusion/core/tests/sql/explain_analyze.rs | 2 +- datafusion/expr-common/src/accumulator.rs | 13 -- .../expr-common/src/groups_accumulator.rs | 14 -- .../src/type_coercion/aggregates.rs | 8 - datafusion/functions-aggregate/src/count.rs | 183 ++---------------- .../simplify_expressions/expr_simplifier.rs | 48 +---- .../src/single_distinct_to_groupby.rs | 66 +------ .../optimizer/tests/optimizer_integration.rs | 8 +- datafusion/physical-expr-common/src/utils.rs | 117 ++++++++++- datafusion/physical-expr/src/aggregate.rs | 6 +- .../aggregates/aggregate_hash_table/common.rs | 49 ++--- .../aggregate_hash_table/common_ordered.rs | 4 +- .../aggregate_hash_table/partial_table.rs | 4 +- .../src/aggregates/aggregate_stream.rs | 27 ++- .../src/aggregates/grouped_hash_stream.rs | 8 +- datafusion/physical-plan/src/windows/mod.rs | 31 +-- datafusion/sql/src/unparser/expr.rs | 6 - .../sqllogictest/test_files/aggregate.slt | 30 ++- .../test_files/aggregate_repartition.slt | 16 +- .../test_files/array/array_has.slt | 36 ++-- datafusion/sqllogictest/test_files/avro.slt | 6 +- .../sqllogictest/test_files/clickbench.slt | 168 ++++++++-------- .../test_files/count_star_rule.slt | 14 +- .../dynamic_filter_pushdown_config.slt | 6 +- .../test_files/explain_analyze.slt | 2 +- .../sqllogictest/test_files/explain_tree.slt | 168 ++++++++-------- .../test_files/functional_dependencies.slt | 6 +- datafusion/sqllogictest/test_files/joins.slt | 20 +- datafusion/sqllogictest/test_files/json.slt | 6 +- .../sqllogictest/test_files/lateral_join.slt | 6 +- datafusion/sqllogictest/test_files/limit.slt | 12 +- .../test_files/nested_loop_join_spill.slt | 8 +- .../optimizer_group_by_constant.slt | 8 +- .../piecewise_merge_join_batches.slt | 12 +- .../test_files/preserve_file_partitioning.slt | 44 ++--- .../test_files/projection_pushdown.slt | 4 +- .../push_down_filter_regression.slt | 6 +- .../repartition_subset_satisfaction.slt | 10 +- datafusion/sqllogictest/test_files/select.slt | 6 +- .../test_files/single_distinct_to_groupby.slt | 14 +- .../sqllogictest/test_files/subquery.slt | 34 ++-- .../test_files/tpch/plans/q1.slt.part | 6 +- .../test_files/tpch/plans/q13.slt.part | 6 +- .../test_files/tpch/plans/q21.slt.part | 6 +- .../test_files/tpch/plans/q22.slt.part | 6 +- .../test_files/tpch/plans/q4.slt.part | 6 +- datafusion/sqllogictest/test_files/union.slt | 14 +- datafusion/sqllogictest/test_files/window.slt | 8 +- .../tests/cases/roundtrip_logical_plan.rs | 24 ++- 51 files changed, 618 insertions(+), 812 deletions(-) diff --git a/datafusion/core/src/physical_planner.rs b/datafusion/core/src/physical_planner.rs index ce17dd902b5c7..c88c34645b6de 100644 --- a/datafusion/core/src/physical_planner.rs +++ b/datafusion/core/src/physical_planner.rs @@ -4800,7 +4800,7 @@ mod tests { assert_contains!( aggregate_explain(&logical_plan).await?, - "aggr=[count() as count(*)]" + "aggr=[count(1) as count(*)]" ); Ok(()) @@ -4815,7 +4815,7 @@ mod tests { assert_contains!( aggregate_explain(&logical_plan).await?, - "aggr=[count() as total_rows]" + "aggr=[count(1) as total_rows]" ); Ok(()) diff --git a/datafusion/core/tests/dataframe/mod.rs b/datafusion/core/tests/dataframe/mod.rs index 6328c2bf8eb14..43ceee3444ced 100644 --- a/datafusion/core/tests/dataframe/mod.rs +++ b/datafusion/core/tests/dataframe/mod.rs @@ -3160,42 +3160,42 @@ async fn test_count_wildcard_on_sort() -> Result<()> { assert_snapshot!( pretty_format_batches(&sql_results).unwrap(), @r" - +---------------+-----------------------------------------------------------------------------------------------+ - | plan_type | plan | - +---------------+-----------------------------------------------------------------------------------------------+ - | logical_plan | Sort: count(*) ASC NULLS LAST | - | | Projection: t1.b, count(Int64(1)) AS count(*) | - | | Aggregate: groupBy=[[t1.b]], aggr=[[count() AS count(Int64(1))]] | - | | TableScan: t1 projection=[b] | - | physical_plan | SortPreservingMergeExec: [count(*)@1 ASC NULLS LAST] | - | | ProjectionExec: expr=[b@0 as b, count(Int64(1))@1 as count(*)] | - | | SortExec: expr=[count(Int64(1))@1 ASC NULLS LAST], preserve_partitioning=[true] | - | | AggregateExec: mode=FinalPartitioned, gby=[b@0 as b], aggr=[count() as count(Int64(1))] | - | | RepartitionExec: partitioning=Hash([b@0], 4), input_partitions=1 | - | | AggregateExec: mode=Partial, gby=[b@0 as b], aggr=[count() as count(Int64(1))] | - | | DataSourceExec: partitions=1, partition_sizes=[1] | - | | | - +---------------+-----------------------------------------------------------------------------------------------+ + +---------------+-------------------------------------------------------------------------------------+ + | plan_type | plan | + +---------------+-------------------------------------------------------------------------------------+ + | logical_plan | Sort: count(*) ASC NULLS LAST | + | | Projection: t1.b, count(Int64(1)) AS count(*) | + | | Aggregate: groupBy=[[t1.b]], aggr=[[count(Int64(1))]] | + | | TableScan: t1 projection=[b] | + | physical_plan | SortPreservingMergeExec: [count(*)@1 ASC NULLS LAST] | + | | ProjectionExec: expr=[b@0 as b, count(Int64(1))@1 as count(*)] | + | | SortExec: expr=[count(Int64(1))@1 ASC NULLS LAST], preserve_partitioning=[true] | + | | AggregateExec: mode=FinalPartitioned, gby=[b@0 as b], aggr=[count(Int64(1))] | + | | RepartitionExec: partitioning=Hash([b@0], 4), input_partitions=1 | + | | AggregateExec: mode=Partial, gby=[b@0 as b], aggr=[count(Int64(1))] | + | | DataSourceExec: partitions=1, partition_sizes=[1] | + | | | + +---------------+-------------------------------------------------------------------------------------+ " ); assert_snapshot!( pretty_format_batches(&df_results).unwrap(), @r" - +---------------+--------------------------------------------------------------------------------------+ - | plan_type | plan | - +---------------+--------------------------------------------------------------------------------------+ - | logical_plan | Sort: count(*) AS count(*) ASC NULLS LAST | - | | Aggregate: groupBy=[[t1.b]], aggr=[[count() AS count(*)]] | - | | TableScan: t1 projection=[b] | - | physical_plan | SortPreservingMergeExec: [count(*)@1 ASC NULLS LAST] | - | | SortExec: expr=[count(*)@1 ASC NULLS LAST], preserve_partitioning=[true] | - | | AggregateExec: mode=FinalPartitioned, gby=[b@0 as b], aggr=[count() as count(*)] | - | | RepartitionExec: partitioning=Hash([b@0], 4), input_partitions=1 | - | | AggregateExec: mode=Partial, gby=[b@0 as b], aggr=[count() as count(*)] | - | | DataSourceExec: partitions=1, partition_sizes=[1] | - | | | - +---------------+--------------------------------------------------------------------------------------+ + +---------------+---------------------------------------------------------------------------------------+ + | plan_type | plan | + +---------------+---------------------------------------------------------------------------------------+ + | logical_plan | Sort: count(*) AS count(*) ASC NULLS LAST | + | | Aggregate: groupBy=[[t1.b]], aggr=[[count(Int64(1)) AS count(*)]] | + | | TableScan: t1 projection=[b] | + | physical_plan | SortPreservingMergeExec: [count(*)@1 ASC NULLS LAST] | + | | SortExec: expr=[count(*)@1 ASC NULLS LAST], preserve_partitioning=[true] | + | | AggregateExec: mode=FinalPartitioned, gby=[b@0 as b], aggr=[count(1) as count(*)] | + | | RepartitionExec: partitioning=Hash([b@0], 4), input_partitions=1 | + | | AggregateExec: mode=Partial, gby=[b@0 as b], aggr=[count(1) as count(*)] | + | | DataSourceExec: partitions=1, partition_sizes=[1] | + | | | + +---------------+---------------------------------------------------------------------------------------+ " ); Ok(()) @@ -3221,7 +3221,7 @@ async fn test_count_wildcard_on_where_in() -> Result<()> { | | TableScan: t1 projection=[a, b] | | | SubqueryAlias: __correlated_sq_1 | | | Projection: count(Int64(1)) AS count(*) | - | | Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] | + | | Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] | | | TableScan: t2 projection=[] | | physical_plan | HashJoinExec: mode=CollectLeft, join_type=RightSemi, on=[(count(*)@0, CAST(t1.a AS Int64)@2)], projection=[a@0, b@1] | | | ProjectionExec: expr=[4 as count(*)] | @@ -3265,7 +3265,7 @@ async fn test_count_wildcard_on_where_in() -> Result<()> { | logical_plan | LeftSemi Join: CAST(t1.a AS Int64) = __correlated_sq_1.count(*) | | | TableScan: t1 projection=[a, b] | | | SubqueryAlias: __correlated_sq_1 | - | | Aggregate: groupBy=[[]], aggr=[[count() AS count(*)]] | + | | Aggregate: groupBy=[[]], aggr=[[count(Int64(1)) AS count(*)]] | | | TableScan: t2 projection=[] | | physical_plan | HashJoinExec: mode=CollectLeft, join_type=RightSemi, on=[(count(*)@0, CAST(t1.a AS Int64)@2)], projection=[a@0, b@1] | | | ProjectionExec: expr=[4 as count(*)] | @@ -3541,16 +3541,16 @@ async fn test_count_wildcard_on_aggregate() -> Result<()> { assert_snapshot!( pretty_format_batches(&sql_results).unwrap(), @r" - +---------------+----------------------------------------------------------------+ - | plan_type | plan | - +---------------+----------------------------------------------------------------+ - | logical_plan | Projection: count(Int64(1)) AS count(*) | - | | Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] | - | | TableScan: t1 projection=[] | - | physical_plan | ProjectionExec: expr=[4 as count(*)] | - | | PlaceholderRowExec | - | | | - +---------------+----------------------------------------------------------------+ + +---------------+-----------------------------------------------------+ + | plan_type | plan | + +---------------+-----------------------------------------------------+ + | logical_plan | Projection: count(Int64(1)) AS count(*) | + | | Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] | + | | TableScan: t1 projection=[] | + | physical_plan | ProjectionExec: expr=[4 as count(*)] | + | | PlaceholderRowExec | + | | | + +---------------+-----------------------------------------------------+ " ); @@ -3567,15 +3567,15 @@ async fn test_count_wildcard_on_aggregate() -> Result<()> { assert_snapshot!( pretty_format_batches(&df_results).unwrap(), @r" - +---------------+-------------------------------------------------------+ - | plan_type | plan | - +---------------+-------------------------------------------------------+ - | logical_plan | Aggregate: groupBy=[[]], aggr=[[count() AS count(*)]] | - | | TableScan: t1 projection=[] | - | physical_plan | ProjectionExec: expr=[4 as count(*)] | - | | PlaceholderRowExec | - | | | - +---------------+-------------------------------------------------------+ + +---------------+---------------------------------------------------------------+ + | plan_type | plan | + +---------------+---------------------------------------------------------------+ + | logical_plan | Aggregate: groupBy=[[]], aggr=[[count(Int64(1)) AS count(*)]] | + | | TableScan: t1 projection=[] | + | physical_plan | ProjectionExec: expr=[4 as count(*)] | + | | PlaceholderRowExec | + | | | + +---------------+---------------------------------------------------------------+ " ); @@ -3606,16 +3606,16 @@ async fn test_count_wildcard_on_where_scalar_subquery() -> Result<()> { | | TableScan: t1 projection=[a, b] | | | SubqueryAlias: __scalar_sq_1 | | | Projection: count(Int64(1)) AS count(*), t2.a, Boolean(true) AS __always_true | - | | Aggregate: groupBy=[[t2.a]], aggr=[[count() AS count(Int64(1))]] | + | | Aggregate: groupBy=[[t2.a]], aggr=[[count(Int64(1))]] | | | TableScan: t2 projection=[a] | | physical_plan | FilterExec: CASE WHEN __always_true@3 IS NULL THEN 0 ELSE count(*)@2 END > 0, projection=[a@0, b@1] | | | RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 | | | HashJoinExec: mode=CollectLeft, join_type=Right, on=[(a@1, a@0)], projection=[a@3, b@4, count(*)@0, __always_true@2] | | | CoalescePartitionsExec | | | ProjectionExec: expr=[count(Int64(1))@1 as count(*), a@0 as a, true as __always_true] | - | | AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[count() as count(Int64(1))] | + | | AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[count(Int64(1))] | | | RepartitionExec: partitioning=Hash([a@0], 4), input_partitions=1 | - | | AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[count() as count(Int64(1))] | + | | AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[count(Int64(1))] | | | DataSourceExec: partitions=1, partition_sizes=[1] | | | DataSourceExec: partitions=1, partition_sizes=[1] | | | | @@ -3661,16 +3661,16 @@ async fn test_count_wildcard_on_where_scalar_subquery() -> Result<()> { | | TableScan: t1 projection=[a, b] | | | SubqueryAlias: __scalar_sq_1 | | | Projection: count(*), t2.a, Boolean(true) AS __always_true | - | | Aggregate: groupBy=[[t2.a]], aggr=[[count() AS count(*)]] | + | | Aggregate: groupBy=[[t2.a]], aggr=[[count(Int64(1)) AS count(*)]] | | | TableScan: t2 projection=[a] | | physical_plan | FilterExec: CASE WHEN __always_true@3 IS NULL THEN 0 ELSE count(*)@2 END > 0, projection=[a@0, b@1] | | | RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 | | | HashJoinExec: mode=CollectLeft, join_type=Right, on=[(a@1, a@0)], projection=[a@3, b@4, count(*)@0, __always_true@2] | | | CoalescePartitionsExec | | | ProjectionExec: expr=[count(*)@1 as count(*), a@0 as a, true as __always_true] | - | | AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[count() as count(*)] | + | | AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[count(1) as count(*)] | | | RepartitionExec: partitioning=Hash([a@0], 4), input_partitions=1 | - | | AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[count() as count(*)] | + | | AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[count(1) as count(*)] | | | DataSourceExec: partitions=1, partition_sizes=[1] | | | DataSourceExec: partitions=1, partition_sizes=[1] | | | | diff --git a/datafusion/core/tests/sql/explain_analyze.rs b/datafusion/core/tests/sql/explain_analyze.rs index 28fcc6eee4b11..67614cfbdea9a 100644 --- a/datafusion/core/tests/sql/explain_analyze.rs +++ b/datafusion/core/tests/sql/explain_analyze.rs @@ -1178,7 +1178,7 @@ async fn explain_logical_plan_only() { @r#" logical_plan Projection: count(Int64(1)) AS count(*) - Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] + Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] SubqueryAlias: t Projection: Values: (Utf8("a"), Int64(1), Int64(100)), (Utf8("a"), Int64(2), Int64(150)) diff --git a/datafusion/expr-common/src/accumulator.rs b/datafusion/expr-common/src/accumulator.rs index 660d32699ffe7..95cee870e2f0b 100644 --- a/datafusion/expr-common/src/accumulator.rs +++ b/datafusion/expr-common/src/accumulator.rs @@ -138,19 +138,6 @@ pub trait Accumulator: Send + Sync + Debug + std::any::Any { /// running sum. fn update_batch(&mut self, values: &[ArrayRef]) -> Result<()>; - /// Like [`Self::update_batch`], but also receives the input row count. - /// - /// Aggregates without arguments cannot derive this count from an empty - /// `values` slice and must override this method. The default delegates to - /// [`Self::update_batch`]. - fn update_batch_with_num_rows( - &mut self, - values: &[ArrayRef], - _num_rows: usize, - ) -> Result<()> { - self.update_batch(values) - } - /// Returns an optional metric timed once per grouped adapter input batch. /// /// A grouped accumulator adapter uses this for aggregate-owned work it diff --git a/datafusion/expr-common/src/groups_accumulator.rs b/datafusion/expr-common/src/groups_accumulator.rs index 7c2e34d5519e4..1c004cb70b931 100644 --- a/datafusion/expr-common/src/groups_accumulator.rs +++ b/datafusion/expr-common/src/groups_accumulator.rs @@ -373,20 +373,6 @@ pub trait GroupsAccumulator: Send + std::any::Any { opt_filter: Option<&BooleanArray>, ) -> Result>; - /// Like [`Self::convert_to_state`], but also receives the input row count. - /// - /// Aggregates without arguments cannot derive this count from an empty - /// `values` slice and must override this method. The default delegates to - /// [`Self::convert_to_state`]. - fn convert_to_state_with_num_rows( - &self, - values: &[ArrayRef], - opt_filter: Option<&BooleanArray>, - _num_rows: usize, - ) -> Result> { - self.convert_to_state(values, opt_filter) - } - /// Amount of memory used to store the state of this accumulator, /// in bytes. /// diff --git a/datafusion/expr-common/src/type_coercion/aggregates.rs b/datafusion/expr-common/src/type_coercion/aggregates.rs index 623e34bdda0b3..ada0bd26b8d06 100644 --- a/datafusion/expr-common/src/type_coercion/aggregates.rs +++ b/datafusion/expr-common/src/type_coercion/aggregates.rs @@ -75,14 +75,6 @@ pub fn check_arg_count( ); } } - TypeSignature::Nullary => { - if !input_fields.is_empty() { - return plan_err!( - "The function {func_name} expects 0 arguments, but {} were provided", - input_fields.len() - ); - } - } TypeSignature::OneOf(variants) => { let ok = variants .iter() diff --git a/datafusion/functions-aggregate/src/count.rs b/datafusion/functions-aggregate/src/count.rs index 10d8eb5da1357..ed2abe88644ae 100644 --- a/datafusion/functions-aggregate/src/count.rs +++ b/datafusion/functions-aggregate/src/count.rs @@ -39,7 +39,7 @@ use datafusion_expr::{ GroupsAccumulator, ReversedUDAF, SetMonotonicity, Signature, StatisticsArgs, TypeSignature, Volatility, WindowFunctionDefinition, expr::WindowFunction, - function::{AccumulatorArgs, AggregateFunctionSimplification, StateFieldsArgs}, + function::{AccumulatorArgs, StateFieldsArgs}, utils::{AggregateOrderSensitivity, format_state_name}, }; use datafusion_functions_aggregate_common::aggregate::count_distinct::PrimitiveDistinctCountGroupsAccumulator; @@ -362,12 +362,12 @@ impl AggregateUDFImpl for Count { arg_types: &[DataType], is_distinct: bool, ) -> Option { - if !is_distinct { - return Some(arg_types.len() <= 1); - } if arg_types.len() != 1 { return Some(false); } + if !is_distinct { + return Some(true); + } // Keep in step with `create_distinct_count_groups_accumulator`. Some(matches!( arg_types[0], @@ -400,32 +400,17 @@ impl AggregateUDFImpl for Count { AggregateOrderSensitivity::Insensitive } - fn simplify(&self) -> Option { - Some(Box::new(|mut aggregate_function, _| { - let params = &aggregate_function.params; - if !params.distinct - && matches!( - params.args.as_slice(), - [Expr::Literal(value, _)] if value == &COUNT_STAR_EXPANSION - ) - { - aggregate_function.params.args.clear(); - } - Ok(Expr::AggregateFunction(aggregate_function)) - })) - } - fn default_value(&self, _data_type: &DataType) -> Result { Ok(ScalarValue::Int64(Some(0))) } fn value_from_stats(&self, statistics_args: &StatisticsArgs) -> Option { + let [expr] = statistics_args.exprs else { + return None; + }; let col_stats = &statistics_args.statistics.column_statistics; if statistics_args.is_distinct { - let [expr] = statistics_args.exprs else { - return None; - }; // Only column references can be resolved from statistics; // expressions like casts or literals are not supported. let col_expr = expr.downcast_ref::()?; @@ -440,15 +425,6 @@ impl AggregateUDFImpl for Count { return None; }; - if statistics_args.exprs.is_empty() { - let num_rows = i64::try_from(num_rows).ok()?; - return Some(ScalarValue::Int64(Some(num_rows))); - } - - let [expr] = statistics_args.exprs else { - return None; - }; - // TODO optimize with exprs other than Column if let Some(col_expr) = expr.downcast_ref::() { if let Precision::Exact(val) = col_stats[col_expr.index()].null_count { @@ -638,19 +614,6 @@ impl Accumulator for CountAccumulator { Ok(()) } - fn update_batch_with_num_rows( - &mut self, - values: &[ArrayRef], - num_rows: usize, - ) -> Result<()> { - if values.is_empty() { - self.count += num_rows as i64; - Ok(()) - } else { - self.update_batch(values) - } - } - fn retract_batch(&mut self, values: &[ArrayRef]) -> Result<()> { let array = &values[0]; self.count -= (array.len() - null_count_for_multiple_cols(values)) as i64; @@ -710,18 +673,15 @@ impl GroupsAccumulator for CountGroupsAccumulator { opt_filter: Option<&BooleanArray>, total_num_groups: usize, ) -> Result<()> { - assert!( - values.len() <= 1, - "COUNT expects zero or one argument to update_batch" - ); - let logical_nulls = values.first().and_then(|values| values.logical_nulls()); + assert_eq!(values.len(), 1, "single argument to update_batch"); + let values = &values[0]; // Add one to each group's counter for each non null, non // filtered value self.counts.resize(total_num_groups, 0); accumulate_indices( group_indices, - logical_nulls.as_ref(), + values.logical_nulls().as_ref(), opt_filter, |group_index| { // SAFETY: group_index is guaranteed to be in bounds @@ -809,35 +769,12 @@ impl GroupsAccumulator for CountGroupsAccumulator { values: &[ArrayRef], opt_filter: Option<&BooleanArray>, ) -> Result> { - let Some(values) = values.first() else { - return internal_err!( - "Nullary COUNT requires convert_to_state_with_num_rows" - ); - }; - self.convert_to_state_with_num_rows( - std::slice::from_ref(values), - opt_filter, - values.len(), - ) - } - - fn convert_to_state_with_num_rows( - &self, - values: &[ArrayRef], - opt_filter: Option<&BooleanArray>, - num_rows: usize, - ) -> Result> { - if values.len() > 1 { - return internal_err!( - "COUNT expects zero or one argument to convert_to_state" - ); - } - let logical_nulls = values.first().and_then(|values| values.logical_nulls()); + let values = &values[0]; - let state_array = match (logical_nulls, opt_filter) { + let state_array = match (values.logical_nulls(), opt_filter) { (None, None) => { // In case there is no nulls in input and no filter, returning array of 1 - Arc::new(Int64Array::from_value(1, num_rows)) + Arc::new(Int64Array::from_value(1, values.len())) } (Some(nulls), None) => { // If there are any nulls in input values -- casting `nulls` (true for values, false for nulls) @@ -1021,13 +958,10 @@ mod tests { use super::*; use arrow::{ - array::{ - BooleanArray, DictionaryArray, Int32Array, Int64Array, NullArray, StringArray, - }, + array::{DictionaryArray, Int32Array, Int64Array, NullArray, StringArray}, datatypes::{DataType, Field, Int32Type, Schema}, }; use datafusion_expr::function::AccumulatorArgs; - use datafusion_expr::{col, lit}; use datafusion_physical_expr::{PhysicalExpr, expressions::Column}; use std::sync::Arc; /// Helper function to create a dictionary array with non-null keys but some null values @@ -1069,95 +1003,6 @@ mod tests { Ok(()) } - #[test] - fn count_accumulator_nullary() -> Result<()> { - let mut accumulator = CountAccumulator::new(); - accumulator.update_batch_with_num_rows(&[], 10)?; - assert_eq!(accumulator.evaluate()?, ScalarValue::Int64(Some(10))); - Ok(()) - } - - #[test] - fn count_nullary_value_from_stats() { - let statistics = datafusion_common::Statistics { - num_rows: Precision::Exact(42), - total_byte_size: Precision::Absent, - column_statistics: vec![], - }; - let return_type = DataType::Int64; - let statistics_args = StatisticsArgs { - statistics: &statistics, - return_type: &return_type, - is_distinct: false, - exprs: &[], - }; - - assert_eq!( - Count::new().value_from_stats(&statistics_args), - Some(ScalarValue::Int64(Some(42))) - ); - } - - #[test] - fn count_groups_accumulator_nullary() -> Result<()> { - let mut accumulator = CountGroupsAccumulator::new(); - accumulator.update_batch(&[], &[0, 1, 0, 2], None, 3)?; - - let result = accumulator.evaluate(EmitTo::All)?; - let expected = Int64Array::from(vec![2, 1, 1]); - assert_eq!(result.as_primitive::(), &expected); - - let state = accumulator.convert_to_state_with_num_rows(&[], None, 3)?; - let expected = Int64Array::from(vec![1, 1, 1]); - assert_eq!(state[0].as_primitive::(), &expected); - - let filter = BooleanArray::from(vec![Some(true), None, Some(false), Some(true)]); - let mut filtered_accumulator = CountGroupsAccumulator::new(); - filtered_accumulator.update_batch(&[], &[0, 1, 0, 2], Some(&filter), 3)?; - let result = filtered_accumulator.evaluate(EmitTo::All)?; - let expected = Int64Array::from(vec![1, 0, 1]); - assert_eq!(result.as_primitive::(), &expected); - - let state = accumulator.convert_to_state_with_num_rows(&[], Some(&filter), 4)?; - let expected = Int64Array::from(vec![1, 0, 0, 1]); - assert_eq!(state[0].as_primitive::(), &expected); - Ok(()) - } - - #[test] - fn simplify_count_star() -> Result<()> { - let info = datafusion_expr::simplify::SimplifyContext::default(); - let simplify = Count::new().simplify().unwrap(); - let simplified_args = |args: Vec, distinct: bool| -> Result> { - let aggregate_function = datafusion_expr::expr::AggregateFunction::new_udf( - count_udaf(), - args, - distinct, - None, - vec![], - None, - ); - match simplify(aggregate_function, &info)? { - Expr::AggregateFunction(f) => Ok(f.params.args), - other => internal_err!("unexpected expression {other}"), - } - }; - - assert!(simplified_args(vec![lit(1_i64)], false)?.is_empty()); - for args in [ - vec![lit(ScalarValue::Null)], - vec![lit(2_i64)], - vec![lit("x")], - vec![col("a")], - vec![col("a") + lit(1)], - ] { - assert_eq!(simplified_args(args.clone(), false)?, args); - } - assert!(simplified_args(vec![], false)?.is_empty()); - assert_eq!(simplified_args(vec![lit(1_i64)], true)?, vec![lit(1_i64)]); - Ok(()) - } - #[test] fn count_groups_preserving_reads() -> Result<()> { let mut accumulator = CountGroupsAccumulator::new(); diff --git a/datafusion/optimizer/src/simplify_expressions/expr_simplifier.rs b/datafusion/optimizer/src/simplify_expressions/expr_simplifier.rs index b6e219e9426b7..506692f0dfa6d 100644 --- a/datafusion/optimizer/src/simplify_expressions/expr_simplifier.rs +++ b/datafusion/optimizer/src/simplify_expressions/expr_simplifier.rs @@ -1679,13 +1679,7 @@ impl TreeNodeRewriter for Simplifier<'_> { .. }) => match (func.simplify(), expr) { (Some(simplify_function), Expr::AggregateFunction(af)) => { - let original = Expr::AggregateFunction(af.clone()); - let simplified = simplify_function(af, info)?; - if simplified == original { - Transformed::no(simplified) - } else { - Transformed::yes(simplified) - } + Transformed::yes(simplify_function(af, info)?) } (_, expr) => Transformed::no(expr), }, @@ -5634,47 +5628,23 @@ mod tests { let expected = aggregate_function_expr.clone(); assert_eq!(simplify(aggregate_function_expr), expected); - - let udaf = AggregateUDF::new_from_impl(SimplifyMockUdaf::new_with_noop()); - let aggregate_function_expr = - Expr::AggregateFunction(expr::AggregateFunction::new_udf( - udaf.into(), - vec![], - false, - None, - vec![], - None, - )); - - let expected = aggregate_function_expr.clone(); - let (actual, cycles) = simplify_with_cycle_count(aggregate_function_expr); - assert_eq!(actual, expected); - assert_eq!(cycles, 1); } /// A Mock UDAF which defines `simplify` to be used in tests /// related to UDAF simplification #[derive(Debug, Clone, PartialEq, Eq, Hash)] struct SimplifyMockUdaf { - simplify: Option, + simplify: bool, } impl SimplifyMockUdaf { /// make simplify method return new expression fn new_with_simplify() -> Self { - Self { - simplify: Some(true), - } + Self { simplify: true } } /// make simplify method return no change fn new_without_simplify() -> Self { - Self { simplify: None } - } - /// make simplify method return the original expression - fn new_with_noop() -> Self { - Self { - simplify: Some(false), - } + Self { simplify: false } } } @@ -5710,12 +5680,10 @@ mod tests { } fn simplify(&self) -> Option { - match self.simplify { - Some(true) => Some(Box::new(|_, _| Ok(col("result_column")))), - Some(false) => Some(Box::new(|aggregate_function, _| { - Ok(Expr::AggregateFunction(aggregate_function)) - })), - None => None, + if self.simplify { + Some(Box::new(|_, _| Ok(col("result_column")))) + } else { + None } } } diff --git a/datafusion/optimizer/src/single_distinct_to_groupby.rs b/datafusion/optimizer/src/single_distinct_to_groupby.rs index d366532d29e0e..f663003a9e4a4 100644 --- a/datafusion/optimizer/src/single_distinct_to_groupby.rs +++ b/datafusion/optimizer/src/single_distinct_to_groupby.rs @@ -104,32 +104,6 @@ struct CountRollup { sum: Arc, } -fn unalias_top(mut expr: &Expr) -> &Expr { - while let Expr::Alias(alias) = expr - && alias - .metadata - .as_ref() - .is_none_or(|metadata| metadata.is_empty()) - { - expr = &alias.expr; - } - expr -} - -fn into_unaliased_top(expr: Expr) -> Expr { - match expr { - Expr::Alias(alias) - if alias - .metadata - .as_ref() - .is_none_or(|metadata| metadata.is_empty()) => - { - into_unaliased_top(*alias.expr) - } - expr => expr, - } -} - impl CountRollup { fn try_new(config: &dyn OptimizerConfig) -> Option { let registry = config.function_registry()?; @@ -157,7 +131,6 @@ fn is_single_distinct_agg( let mut distinct_aggs = vec![]; let mut has_count_rollup = false; for expr in aggr_expr { - let expr = unalias_top(expr); if let Expr::AggregateFunction(AggregateFunction { func, params: @@ -328,7 +301,7 @@ impl OptimizerRule for SingleDistinctToGroupBy { // zero that `sum` reports as NULL over an empty input. let (outer_aggr_exprs, outer_proj_exprs): (Vec, Vec) = aggr_expr .into_iter() - .map(|aggr_expr| match into_unaliased_top(aggr_expr) { + .map(|aggr_expr| match aggr_expr { Expr::AggregateFunction(AggregateFunction { func, params: @@ -408,7 +381,7 @@ impl OptimizerRule for SingleDistinctToGroupBy { Ok((outer, proj)) } } - aggr_expr => Ok((aggr_expr.clone(), aggr_expr)), + _ => Ok((aggr_expr.clone(), aggr_expr)), }) .collect::>>()? .into_iter() @@ -1184,41 +1157,6 @@ mod tests { ) } - #[test] - fn aliased_count_star_and_distinct_without_groupby() -> Result<()> { - let table_scan = test_table_scan_utf8_b()?; - - // Simplifying `count(1)` to `count()` preserves its old name with an - // alias before this rule runs. - let plan = LogicalPlanBuilder::from(table_scan) - .aggregate( - Vec::::new(), - vec![ - Expr::AggregateFunction(AggregateFunction::new_udf( - count_udaf(), - vec![], - false, - None, - vec![], - None, - )) - .alias("count(Int64(1))"), - count_distinct(col("b")), - ], - )? - .build()?; - - assert_optimized_plan_equal!( - plan, - @r" - Projection: CASE WHEN sum(alias2) IS NOT NULL THEN sum(alias2) ELSE Int64(0) END AS count(Int64(1)), count(alias1) AS count(DISTINCT test.b) [count(Int64(1)):Int64, count(DISTINCT test.b):Int64] - Aggregate: groupBy=[[]], aggr=[[sum(alias2), count(alias1)]] [sum(alias2):Int64;N, count(alias1):Int64] - Aggregate: groupBy=[[test.b AS alias1]], aggr=[[count() AS alias2]] [alias1:Utf8, alias2:Int64] - TableScan: test [a:UInt32, b:Utf8, c:UInt32] - " - ) - } - #[test] fn count_star_min_max_sum_and_distinct_with_groupby() -> Result<()> { let table_scan = test_table_scan_utf8_b()?; diff --git a/datafusion/optimizer/tests/optimizer_integration.rs b/datafusion/optimizer/tests/optimizer_integration.rs index 0db3131c29827..26b48c5e1f352 100644 --- a/datafusion/optimizer/tests/optimizer_integration.rs +++ b/datafusion/optimizer/tests/optimizer_integration.rs @@ -296,7 +296,7 @@ fn between_date32_plus_interval() -> Result<()> { assert_snapshot!( format!("{plan}"), @r#" - Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] + Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] Projection: Filter: test.col_date32 >= Date32("1998-03-18") AND test.col_date32 <= Date32("1998-06-16") TableScan: test projection=[col_date32] @@ -314,7 +314,7 @@ fn between_date64_plus_interval() -> Result<()> { assert_snapshot!( format!("{plan}"), @r#" - Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] + Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] Projection: Filter: test.col_date64 >= Date64("1998-03-18") AND test.col_date64 <= Date64("1998-06-16") TableScan: test projection=[col_date64] @@ -388,7 +388,7 @@ fn push_down_filter_groupby_expr_contains_alias() { format!("{plan}"), @r" Projection: test.col_int32 + test.col_uint32 AS c, count(Int64(1)) AS count(*) - Aggregate: groupBy=[[CAST(test.col_int32 AS Int64) + CAST(test.col_uint32 AS Int64)]], aggr=[[count() AS count(Int64(1))]] + Aggregate: groupBy=[[CAST(test.col_int32 AS Int64) + CAST(test.col_uint32 AS Int64)]], aggr=[[count(Int64(1))]] Filter: CAST(test.col_int32 AS Int64) + CAST(test.col_uint32 AS Int64) > Int64(3) TableScan: test projection=[col_int32, col_uint32] " @@ -445,7 +445,7 @@ fn eliminate_redundant_null_check_on_count() { format!("{plan}"), @r" Projection: test.col_int32, count(Int64(1)) AS count(*) AS c - Aggregate: groupBy=[[test.col_int32]], aggr=[[count() AS count(Int64(1))]] + Aggregate: groupBy=[[test.col_int32]], aggr=[[count(Int64(1))]] TableScan: test projection=[col_int32] " ); diff --git a/datafusion/physical-expr-common/src/utils.rs b/datafusion/physical-expr-common/src/utils.rs index b67e391dd497d..823f4f84ea417 100644 --- a/datafusion/physical-expr-common/src/utils.rs +++ b/datafusion/physical-expr-common/src/utils.rs @@ -36,7 +36,8 @@ use arrow::datatypes::{ }; use arrow::record_batch::RecordBatch; use arrow::{downcast_dictionary_array, downcast_primitive_array}; -use datafusion_common::Result; +use datafusion_common::{Result, ScalarValue, assert_eq_or_internal_err}; +use datafusion_expr_common::columnar_value::ColumnarValue; use datafusion_expr_common::sort_properties::ExprProperties; /// Represents a [`PhysicalExpr`] node with associated properties (order and @@ -425,6 +426,68 @@ pub fn evaluate_expressions_to_arrays_with_metrics<'a>( .collect::>>() } +/// Largest array, in bytes, that a [`ScalarArrayCache`] keeps between calls. +const MAX_CACHED_SCALAR_ARRAY_BYTES: usize = 1024 * 1024; + +/// Reuses the array expanded from a scalar expression result across batches. +#[derive(Debug, Default)] +pub struct ScalarArrayCache { + cached: Option<(ScalarValue, ArrayRef)>, +} + +impl ScalarArrayCache { + /// Like [`ColumnarValue::into_array_of_size`], but reuses the array built + /// for an earlier, equal scalar. + pub fn into_array_of_size( + &mut self, + value: ColumnarValue, + num_rows: usize, + ) -> Result { + let scalar = match value { + ColumnarValue::Scalar(scalar) => scalar, + array @ ColumnarValue::Array(_) => { + return array.into_array_of_size(num_rows); + } + }; + + if let Some((cached_scalar, array)) = &self.cached + && array.len() >= num_rows + && *cached_scalar == scalar + { + return Ok(array.slice(0, num_rows)); + } + + let array = scalar.to_array_of_size(num_rows)?; + if array.get_array_memory_size() <= MAX_CACHED_SCALAR_ARRAY_BYTES { + self.cached = Some((scalar, Arc::clone(&array))); + } + Ok(array) + } +} + +/// Like [`evaluate_expressions_to_arrays`], but expands scalar results through +/// one cache per expression. +pub fn evaluate_expressions_to_arrays_with_cache( + exprs: &[Arc], + caches: &mut [ScalarArrayCache], + batch: &RecordBatch, +) -> Result> { + assert_eq_or_internal_err!( + exprs.len(), + caches.len(), + "expected one scalar array cache per expression" + ); + let num_rows = batch.num_rows(); + exprs + .iter() + .zip(caches) + .map(|(expr, cache)| { + expr.evaluate(batch) + .and_then(|value| cache.into_array_of_size(value, num_rows)) + }) + .collect() +} + #[cfg(test)] mod tests { @@ -656,4 +719,56 @@ mod tests { assert_eq!(scattered.value(4), 50); Ok(()) } + + #[test] + fn scalar_array_cache_reuses_equal_scalars() -> Result<()> { + let mut cache = ScalarArrayCache::default(); + let one = || ColumnarValue::Scalar(ScalarValue::Int32(Some(1))); + + let first = cache.into_array_of_size(one(), 4)?; + let second = cache.into_array_of_size(one(), 3)?; + assert_eq!(as_int32_array(&second)?, &Int32Array::from(vec![1, 1, 1])); + assert_eq!( + as_int32_array(&second)?.values().as_ptr(), + as_int32_array(&first)?.values().as_ptr() + ); + + let larger = cache.into_array_of_size(one(), 6)?; + assert_eq!(as_int32_array(&larger)?, &Int32Array::from(vec![1; 6])); + let smaller = cache.into_array_of_size(one(), 5)?; + assert_eq!( + as_int32_array(&smaller)?.values().as_ptr(), + as_int32_array(&larger)?.values().as_ptr() + ); + + let two = ColumnarValue::Scalar(ScalarValue::Int32(Some(2))); + let two = cache.into_array_of_size(two, 2)?; + assert_eq!(as_int32_array(&two)?, &Int32Array::from(vec![2, 2])); + + let array: ArrayRef = Arc::new(Int32Array::from(vec![7, 8])); + let result = + cache.into_array_of_size(ColumnarValue::Array(Arc::clone(&array)), 2)?; + assert!(Arc::ptr_eq(&result, &array)); + assert!( + cache + .into_array_of_size(ColumnarValue::Array(array), 3) + .is_err() + ); + Ok(()) + } + + #[test] + fn scalar_array_cache_skips_large_arrays() -> Result<()> { + let mut cache = ScalarArrayCache::default(); + let value = || ColumnarValue::Scalar(ScalarValue::from("x".repeat(1024))); + + let first = cache.into_array_of_size(value(), 2048)?; + let second = cache.into_array_of_size(value(), 2048)?; + assert_eq!(first.as_ref(), second.as_ref()); + assert_ne!( + as_string_array(&first).values().as_ptr(), + as_string_array(&second).values().as_ptr() + ); + Ok(()) + } } diff --git a/datafusion/physical-expr/src/aggregate.rs b/datafusion/physical-expr/src/aggregate.rs index 9f1dfaac3a4a1..6d95d8ea12bd8 100644 --- a/datafusion/physical-expr/src/aggregate.rs +++ b/datafusion/physical-expr/src/aggregate.rs @@ -44,7 +44,9 @@ use crate::planner::{create_physical_expr, create_physical_exprs}; use arrow::compute::SortOptions; use arrow::datatypes::{DataType, FieldRef, Schema, SchemaRef}; use datafusion_common::metadata::FieldMetadata; -use datafusion_common::{DFSchema, Result, ScalarValue, internal_err, not_impl_err}; +use datafusion_common::{ + DFSchema, Result, ScalarValue, assert_or_internal_err, internal_err, not_impl_err, +}; use datafusion_expr::execution_props::ExecutionProps; use datafusion_expr::expr::{ AggregateFunction, AggregateFunctionParams, NullTreatment, physical_name, @@ -260,6 +262,8 @@ impl AggregateExprBuilder { is_distinct, is_reversed, } = self; + assert_or_internal_err!(!args.is_empty(), "args should not be empty"); + // An order-insensitive aggregate ignores its ORDER BY, so drop it here. // Everything derived from `order_bys` below, such as the ordering fields // in the aggregate's state, then agrees that there is no ordering. diff --git a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common.rs b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common.rs index 6d3c67078bf73..92f55029e1168 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common.rs @@ -31,6 +31,7 @@ use datafusion_execution::memory_pool::proxy::VecAllocExt; use datafusion_expr::{AggregateMetrics, EmitTo, GroupsAccumulator}; use datafusion_physical_expr::GroupsAccumulatorAdapter; use datafusion_physical_expr::aggregate::AggregateFunctionExpr; +use datafusion_physical_expr_common::utils::ScalarArrayCache; use log::debug; use crate::PhysicalExpr; @@ -203,10 +204,10 @@ impl AggregateHashTable { /// See comments in [`EvaluatedAggregateBatch`] pub(super) fn evaluate_batch( - &self, + &mut self, batch: &RecordBatch, ) -> Result { - let state = self.state.building(); + let state = self.state.building_mut(); // Outer vec: one per grouping set; inner vec: group-by expressions. let grouping_set_args = self .group_by_metrics @@ -216,7 +217,7 @@ impl AggregateHashTable { let accumulator_args = self.group_by_metrics.time_aggregate_arguments(|| { state .accumulators - .iter() + .iter_mut() .enumerate() .map(|(idx, acc)| { self.aggregate_argument_metrics @@ -529,6 +530,9 @@ pub(super) struct HashAggregateAccumulator { /// Example: `CORR(x, y)` stores two expressions here, while `SUM(x)` stores one. arguments: Vec>, + /// One cache per aggregate argument. + argument_caches: Vec, + /// Optional `FILTER` expression for this accumulator. /// /// Example: `SUM(x) FILTER (WHERE x > 10)` stores the `x > 10` predicate. @@ -583,8 +587,6 @@ pub(super) struct RowAlignedAccumulatorArgs { pub(super) arguments: Vec, /// Original row-aligned filter passed through to state conversion. pub(super) filter: Option, - /// Number of rows represented by the arguments. - pub(super) num_rows: usize, } /// Evaluated all group by keys and accumulator args. @@ -720,9 +722,14 @@ impl HashAggregateAccumulator { accumulator: Box, submetrics: Arc, ) -> Self { + let argument_caches = arguments + .iter() + .map(|_| ScalarArrayCache::default()) + .collect(); Self { aggregate_expr, arguments, + argument_caches, filter, accumulator, submetrics, @@ -752,7 +759,7 @@ impl HashAggregateAccumulator { /// Before updating [`GroupsAccumulator`], the retained selection is used to /// compact the matching group IDs and is not passed through. pub(super) fn evaluate_compacted_args( - &self, + &mut self, batch: &RecordBatch, ) -> Result { let selection = self.evaluate_filter(batch)?; @@ -774,10 +781,12 @@ impl HashAggregateAccumulator { let arguments = self .arguments .iter() - .map(|expr| { + .zip(&mut self.argument_caches) + .map(|(expr, cache)| { if let Some(argument_batch) = argument_batch { - expr.evaluate(argument_batch) - .and_then(|value| value.into_array(argument_batch.num_rows())) + expr.evaluate(argument_batch).and_then(|value| { + cache.into_array_of_size(value, argument_batch.num_rows()) + }) } else { let data_type = expr.data_type(batch.schema_ref().as_ref())?; Ok(new_empty_array(&data_type)) @@ -797,7 +806,7 @@ impl HashAggregateAccumulator { /// rows remain as null argument values and the filter is passed to /// [`GroupsAccumulator::convert_to_state`]. pub(super) fn evaluate_row_aligned_args( - &self, + &mut self, batch: &RecordBatch, ) -> Result { let filter = self.evaluate_filter(batch)?; @@ -805,21 +814,18 @@ impl HashAggregateAccumulator { let arguments = self .arguments .iter() - .map(|expr| { + .zip(&mut self.argument_caches) + .map(|(expr, cache)| { selection .map_or_else( || expr.evaluate(batch), |selection| expr.evaluate_selection(batch, selection), ) - .and_then(|value| value.into_array(batch.num_rows())) + .and_then(|value| cache.into_array_of_size(value, batch.num_rows())) }) .collect::>()?; - Ok(RowAlignedAccumulatorArgs { - arguments, - filter, - num_rows: batch.num_rows(), - }) + Ok(RowAlignedAccumulatorArgs { arguments, filter }) } fn evaluate_filter(&self, batch: &RecordBatch) -> Result> { @@ -894,11 +900,8 @@ impl HashAggregateAccumulator { &self, values: &RowAlignedAccumulatorArgs, ) -> Result> { - self.accumulator.convert_to_state_with_num_rows( - &values.arguments, - values.filter.as_ref(), - values.num_rows, - ) + self.accumulator + .convert_to_state(&values.arguments, values.filter.as_ref()) } pub(super) fn null_arguments( @@ -1006,7 +1009,7 @@ mod tests { let submetrics = aggregate_sub_metrics(&metrics, 0, ["SUM(value)"]) .pop() .expect("one aggregate submetric factory"); - let accumulator = sum_accumulator(&schema, "include", 1, submetrics)?; + let mut accumulator = sum_accumulator(&schema, "include", 1, submetrics)?; let group_by_metrics = GroupByMetrics::new(&metrics, 0); let argument_metrics = AggregateArgumentMetrics::new(&metrics, 0, ["SUM(value)"]); let accumulator_metrics = AggregateAccumulatorMetrics::new( diff --git a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common_ordered.rs b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common_ordered.rs index 9476c0991a2b6..39ae9eb0bcf12 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common_ordered.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common_ordered.rs @@ -250,7 +250,7 @@ impl OrderedAggregateTable { /// e.g., `select k+1, sum(v*v) from t group by (k+1)`, this function /// evaluates `k+1`, `v*v`. pub(super) fn evaluate_batch( - &self, + &mut self, batch: &RecordBatch, ) -> Result { let grouping_set_args = @@ -261,7 +261,7 @@ impl OrderedAggregateTable { let accumulator_args = self.group_by_metrics.time_aggregate_arguments(|| { self.buffer .accumulators - .iter() + .iter_mut() .enumerate() .map(|(idx, acc)| { self.aggregate_argument_metrics diff --git a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/partial_table.rs b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/partial_table.rs index 90f81930344f7..1e9c1c8497283 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/partial_table.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/partial_table.rs @@ -131,7 +131,7 @@ impl AggregateHashTable { &mut self, batch: &RecordBatch, ) -> Result { - let state = self.state.building(); + let state = self.state.building_mut(); let grouping_set_args = self .group_by_metrics .time_group_key_preparation(|| evaluate_group_by(&state.group_by, batch))?; @@ -144,7 +144,7 @@ impl AggregateHashTable { let mut output = grouping_set_args.into_iter().next().unwrap_or_default(); let accumulator_metrics = Arc::clone(&self.aggregate_accumulator_metrics); - for (idx, acc) in state.accumulators.iter().enumerate() { + for (idx, acc) in state.accumulators.iter_mut().enumerate() { let values = self.group_by_metrics.time_aggregate_arguments(|| { self.aggregate_argument_metrics .time(idx, || acc.evaluate_row_aligned_args(batch)) diff --git a/datafusion/physical-plan/src/aggregates/aggregate_stream.rs b/datafusion/physical-plan/src/aggregates/aggregate_stream.rs index 49f6c2f0cf216..4b8476b9b38d2 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_stream.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_stream.rs @@ -32,9 +32,7 @@ use crate::{RecordBatchStream, SendableRecordBatchStream}; use arrow::array::ArrayRef; use arrow::datatypes::SchemaRef; use arrow::record_batch::RecordBatch; -use datafusion_common::{ - DataFusionError, Result, ScalarValue, internal_datafusion_err, internal_err, -}; +use datafusion_common::{Result, ScalarValue, internal_datafusion_err, internal_err}; use datafusion_execution::TaskContext; use datafusion_expr::Operator; use datafusion_physical_expr::PhysicalExpr; @@ -49,7 +47,9 @@ use std::task::{Context, Poll}; use super::AggregateExec; use crate::filter::batch_filter; use datafusion_execution::memory_pool::{MemoryConsumer, MemoryReservation}; -use datafusion_physical_expr_common::utils::evaluate_expressions_to_arrays; +use datafusion_physical_expr_common::utils::{ + ScalarArrayCache, evaluate_expressions_to_arrays_with_cache, +}; use futures::stream::{Stream, StreamExt}; /// stream struct for aggregation without grouping columns @@ -71,6 +71,7 @@ struct AggregateStreamInner { mode: AggregateMode, input: SendableRecordBatchStream, aggregate_expressions: Vec>>, + aggregate_argument_caches: Vec>, filter_expressions: Arc<[Option>]>, aggregate_argument_metrics: AggregateArgumentMetrics, aggregate_accumulator_metrics: AggregateAccumulatorMetrics, @@ -303,6 +304,10 @@ impl AggregateStream { let input = agg.input.execute(partition, Arc::clone(context))?; let aggregate_expressions = aggregate_expressions(&agg.aggr_expr, &agg.mode, 0)?; + let aggregate_argument_caches = aggregate_expressions + .iter() + .map(|exprs| exprs.iter().map(|_| ScalarArrayCache::default()).collect()) + .collect(); let filter_expressions = match agg.mode.input_mode() { AggregateInputMode::Raw => agg_filter_expr, AggregateInputMode::Partial => vec![None; agg.aggr_expr.len()].into(), @@ -364,6 +369,7 @@ impl AggregateStream { input, baseline_metrics, aggregate_expressions, + aggregate_argument_caches, filter_expressions, aggregate_argument_metrics, aggregate_accumulator_metrics, @@ -389,6 +395,7 @@ impl AggregateStream { &batch, &mut this.accumulators, &this.aggregate_expressions, + &mut this.aggregate_argument_caches, &this.filter_expressions, &this.aggregate_argument_metrics, &this.aggregate_accumulator_metrics, @@ -474,11 +481,13 @@ impl RecordBatchStream for AggregateStream { /// If successful, this returns the additional number of bytes that were allocated during this process. /// /// TODO: Make this a member function +#[expect(clippy::too_many_arguments)] fn aggregate_batch( mode: &AggregateMode, batch: &RecordBatch, accumulators: &mut [AccumulatorItem], expressions: &[Vec>], + argument_caches: &mut [Vec], filters: &[Option>], aggregate_argument_metrics: &AggregateArgumentMetrics, aggregate_accumulator_metrics: &AggregateAccumulatorMetrics, @@ -494,17 +503,17 @@ fn aggregate_batch( accumulators .iter_mut() .zip(expressions) + .zip(argument_caches) .zip(filters) .enumerate() - .try_for_each(|(index, ((accum, expr), filter))| { + .try_for_each(|(index, (((accum, expr), caches), filter))| { // 1.2 and 1.3 - let (values, num_rows) = aggregate_argument_metrics.time(index, || { + let values = aggregate_argument_metrics.time(index, || { let batch = match filter { Some(filter) => Cow::Owned(batch_filter(batch, filter)?), None => Cow::Borrowed(batch), }; - let values = evaluate_expressions_to_arrays(expr, batch.as_ref())?; - Ok::<_, DataFusionError>((values, batch.num_rows())) + evaluate_expressions_to_arrays_with_cache(expr, caches, batch.as_ref()) })?; // 1.4 @@ -513,7 +522,7 @@ fn aggregate_batch( AggregateInputMode::Raw => aggregate_accumulator_metrics.time( index, AccumulatorPhase::Update, - || accum.update_batch_with_num_rows(&values, num_rows), + || accum.update_batch(&values), ), AggregateInputMode::Partial => aggregate_accumulator_metrics.time( index, diff --git a/datafusion/physical-plan/src/aggregates/grouped_hash_stream.rs b/datafusion/physical-plan/src/aggregates/grouped_hash_stream.rs index e3861a7a70cb9..042f4a90449dc 100644 --- a/datafusion/physical-plan/src/aggregates/grouped_hash_stream.rs +++ b/datafusion/physical-plan/src/aggregates/grouped_hash_stream.rs @@ -1469,13 +1469,7 @@ impl GroupedHashAggregateStream { output.extend(self.aggregate_accumulator_metrics.time( idx, AccumulatorPhase::ConvertToState, - || { - acc.convert_to_state_with_num_rows( - values, - opt_filter, - batch.num_rows(), - ) - }, + || acc.convert_to_state(values, opt_filter), )?); } diff --git a/datafusion/physical-plan/src/windows/mod.rs b/datafusion/physical-plan/src/windows/mod.rs index d000d8e980e63..7c5f55f661f38 100644 --- a/datafusion/physical-plan/src/windows/mod.rs +++ b/datafusion/physical-plan/src/windows/mod.rs @@ -33,7 +33,7 @@ use crate::{ use arrow::datatypes::{Schema, SchemaRef}; use arrow_schema::{FieldRef, SortOptions}; -use datafusion_common::{Result, assert_or_internal_err, exec_err, not_impl_err}; +use datafusion_common::{Result, assert_or_internal_err, exec_err}; use datafusion_expr::{ LimitEffect, PartitionEvaluator, ReversedUDWF, SetMonotonicity, WindowFrame, WindowFunctionDefinition, WindowUDF, @@ -103,12 +103,6 @@ pub fn create_window_expr( ) -> Result> { Ok(match fun { WindowFunctionDefinition::AggregateUDF(fun) => { - if args.is_empty() { - return not_impl_err!( - "Aggregate window function {} without arguments is not supported", - fun.name() - ); - } let aggregate = if distinct { AggregateExprBuilder::new(Arc::clone(fun), args.to_vec()) .schema(input_schema) @@ -963,29 +957,6 @@ mod tests { Ok(()) } - #[test] - fn create_window_expr_rejects_aggregate_without_args() { - let schema = Arc::new(Schema::new(vec![Field::new("a", DataType::Int32, true)])); - let err = create_window_expr( - &WindowFunctionDefinition::AggregateUDF(count_udaf()), - "count".to_owned(), - &[], - &[], - &[], - Arc::new(WindowFrame::new(None)), - schema, - false, - false, - None, - ) - .unwrap_err(); - assert!( - err.to_string() - .contains("without arguments is not supported"), - "{err}" - ); - } - #[tokio::test] async fn get_best_fitting_window_preserves_state_observer() -> Result<()> { // `EnforceSorting`/`EnforceDistribution` call `get_best_fitting_window` diff --git a/datafusion/sql/src/unparser/expr.rs b/datafusion/sql/src/unparser/expr.rs index ba73ee98f33e1..0f2cef167cc9d 100644 --- a/datafusion/sql/src/unparser/expr.rs +++ b/datafusion/sql/src/unparser/expr.rs @@ -417,11 +417,6 @@ impl Unparser<'_> { .map(|sort_expr| self.sort_to_sql(sort_expr)) .collect::>>()?; (args_to_use, within_group) - } else if args.is_empty() && func_name == "count" { - // Many dialects only accept `count(*)` for a parameterless count - let wildcard = - ast::FunctionArg::Unnamed(ast::FunctionArgExpr::Wildcard); - (vec![wildcard], Vec::new()) } else { (self.function_args_to_sql(args)?, Vec::new()) }; @@ -2283,7 +2278,6 @@ mod tests { .unwrap(), "count(*) FILTER (WHERE true)", ), - (count_udaf().call(vec![]), "count(*)"), ( Expr::from(WindowFunction { fun: WindowFunctionDefinition::WindowUDF(row_number_udwf()), diff --git a/datafusion/sqllogictest/test_files/aggregate.slt b/datafusion/sqllogictest/test_files/aggregate.slt index 8477cc92431dc..2be4e173fc701 100644 --- a/datafusion/sqllogictest/test_files/aggregate.slt +++ b/datafusion/sqllogictest/test_files/aggregate.slt @@ -8912,13 +8912,11 @@ query TT explain select count(1), count(2) from t; ---- logical_plan -01)Projection: __common_expr_1 AS count(Int64(1)), __common_expr_1 AS count(Int64(2)) -02)--Aggregate: groupBy=[[]], aggr=[[count() AS __common_expr_1]] -03)----TableScan: t projection=[] +01)Aggregate: groupBy=[[]], aggr=[[count(Int64(1)), count(Int64(2))]] +02)--TableScan: t projection=[] physical_plan -01)ProjectionExec: expr=[__common_expr_1@0 as count(Int64(1)), __common_expr_1@0 as count(Int64(2))] -02)--ProjectionExec: expr=[2 as __common_expr_1] -03)----PlaceholderRowExec +01)AggregateExec: mode=Single, gby=[], aggr=[count(Int64(1)), count(Int64(2))] +02)--DataSourceExec: partitions=1, partition_sizes=[1] query II select count(1), count() from t; @@ -8930,7 +8928,7 @@ explain select count(1), count() from t; ---- logical_plan 01)Projection: count(Int64(1)), count(Int64(1)) AS count() -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 03)----TableScan: t projection=[] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(Int64(1)), count(Int64(1))@0 as count()] @@ -8947,7 +8945,7 @@ explain select count(1), count(*) from t; ---- logical_plan 01)Projection: count(Int64(1)), count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 03)----TableScan: t projection=[] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(Int64(1)), count(Int64(1))@0 as count(*)] @@ -8964,7 +8962,7 @@ explain select count(), count(*) from t; ---- logical_plan 01)Projection: count(Int64(1)) AS count(), count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 03)----TableScan: t projection=[] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(), count(Int64(1))@0 as count(*)] @@ -8975,13 +8973,13 @@ query TT explain select count(1) * count(2) from t; ---- logical_plan -01)Projection: __common_expr_1 * __common_expr_1 AS count(Int64(1)) * count(Int64(2)) -02)--Aggregate: groupBy=[[]], aggr=[[count() AS __common_expr_1]] +01)Projection: count(Int64(1)) * count(Int64(2)) +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1)), count(Int64(2))]] 03)----TableScan: t projection=[] physical_plan -01)ProjectionExec: expr=[__common_expr_1@0 * __common_expr_1@0 as count(Int64(1)) * count(Int64(2))] -02)--ProjectionExec: expr=[2 as __common_expr_1] -03)----PlaceholderRowExec +01)ProjectionExec: expr=[count(Int64(1))@0 * count(Int64(2))@1 as count(Int64(1)) * count(Int64(2))] +02)--AggregateExec: mode=Single, gby=[], aggr=[count(Int64(1)), count(Int64(2))] +03)----DataSourceExec: partitions=1, partition_sizes=[1] statement count 0 drop table t; @@ -9493,12 +9491,12 @@ ORDER BY g; logical_plan 01)Sort: stream_test.g ASC NULLS LAST 02)--Projection: stream_test.g, count(Int64(1)) AS count(*), sum(stream_test.x), avg(stream_test.x), avg(stream_test.x) AS mean(stream_test.x), min(stream_test.x), max(stream_test.y), bit_and(stream_test.i), bit_or(stream_test.i), bit_xor(stream_test.i), bool_and(stream_test.b), bool_or(stream_test.b), median(stream_test.x), Int32(0) AS grouping(stream_test.g), var(stream_test.x), var(stream_test.x) AS var_samp(stream_test.x), var_pop(stream_test.x), var(stream_test.x) AS var_sample(stream_test.x), var_pop(stream_test.x) AS var_population(stream_test.x), stddev(stream_test.x), stddev(stream_test.x) AS stddev_samp(stream_test.x), stddev_pop(stream_test.x) -03)----Aggregate: groupBy=[[stream_test.g]], aggr=[[count() AS count(Int64(1)), sum(stream_test.x), avg(stream_test.x), min(stream_test.x), max(stream_test.y), bit_and(stream_test.i), bit_or(stream_test.i), bit_xor(stream_test.i), bool_and(stream_test.b), bool_or(stream_test.b), median(stream_test.x), var(stream_test.x), var_pop(stream_test.x), stddev(stream_test.x), stddev_pop(stream_test.x)]] +03)----Aggregate: groupBy=[[stream_test.g]], aggr=[[count(Int64(1)), sum(stream_test.x), avg(stream_test.x), min(stream_test.x), max(stream_test.y), bit_and(stream_test.i), bit_or(stream_test.i), bit_xor(stream_test.i), bool_and(stream_test.b), bool_or(stream_test.b), median(stream_test.x), var(stream_test.x), var_pop(stream_test.x), stddev(stream_test.x), stddev_pop(stream_test.x)]] 04)------Sort: stream_test.g ASC NULLS LAST, fetch=10000 05)--------TableScan: stream_test projection=[g, x, y, i, b] physical_plan 01)ProjectionExec: expr=[g@0 as g, count(Int64(1))@1 as count(*), sum(stream_test.x)@2 as sum(stream_test.x), avg(stream_test.x)@3 as avg(stream_test.x), avg(stream_test.x)@3 as mean(stream_test.x), min(stream_test.x)@4 as min(stream_test.x), max(stream_test.y)@5 as max(stream_test.y), bit_and(stream_test.i)@6 as bit_and(stream_test.i), bit_or(stream_test.i)@7 as bit_or(stream_test.i), bit_xor(stream_test.i)@8 as bit_xor(stream_test.i), bool_and(stream_test.b)@9 as bool_and(stream_test.b), bool_or(stream_test.b)@10 as bool_or(stream_test.b), median(stream_test.x)@11 as median(stream_test.x), 0 as grouping(stream_test.g), var(stream_test.x)@12 as var(stream_test.x), var(stream_test.x)@12 as var_samp(stream_test.x), var_pop(stream_test.x)@13 as var_pop(stream_test.x), var(stream_test.x)@12 as var_sample(stream_test.x), var_pop(stream_test.x)@13 as var_population(stream_test.x), stddev(stream_test.x)@14 as stddev(stream_test.x), stddev(stream_test.x)@14 as stddev_samp(stream_test.x), stddev_pop(stream_test.x)@15 as stddev_pop(stream_test.x)] -02)--AggregateExec: mode=Single, gby=[g@0 as g], aggr=[count() as count(Int64(1)), sum(stream_test.x), avg(stream_test.x), min(stream_test.x), max(stream_test.y), bit_and(stream_test.i), bit_or(stream_test.i), bit_xor(stream_test.i), bool_and(stream_test.b), bool_or(stream_test.b), median(stream_test.x), var(stream_test.x), var_pop(stream_test.x), stddev(stream_test.x), stddev_pop(stream_test.x)], ordering_mode=Sorted +02)--AggregateExec: mode=Single, gby=[g@0 as g], aggr=[count(Int64(1)), sum(stream_test.x), avg(stream_test.x), min(stream_test.x), max(stream_test.y), bit_and(stream_test.i), bit_or(stream_test.i), bit_xor(stream_test.i), bool_and(stream_test.b), bool_or(stream_test.b), median(stream_test.x), var(stream_test.x), var_pop(stream_test.x), stddev(stream_test.x), stddev_pop(stream_test.x)], ordering_mode=Sorted 03)----SortExec: TopK(fetch=10000), expr=[g@0 ASC NULLS LAST], preserve_partitioning=[false] 04)------DataSourceExec: partitions=1, partition_sizes=[1] diff --git a/datafusion/sqllogictest/test_files/aggregate_repartition.slt b/datafusion/sqllogictest/test_files/aggregate_repartition.slt index acf6553ac7e18..2302e161bfe72 100644 --- a/datafusion/sqllogictest/test_files/aggregate_repartition.slt +++ b/datafusion/sqllogictest/test_files/aggregate_repartition.slt @@ -72,13 +72,13 @@ EXPLAIN SELECT env, count(*) FROM dim_csv GROUP BY env; ---- logical_plan 01)Projection: dim_csv.env, count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[dim_csv.env]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[dim_csv.env]], aggr=[[count(Int64(1))]] 03)----TableScan: dim_csv projection=[env] physical_plan 01)ProjectionExec: expr=[env@0 as env, count(Int64(1))@1 as count(*)] -02)--AggregateExec: mode=FinalPartitioned, gby=[env@0 as env], aggr=[count() as count(Int64(1))] +02)--AggregateExec: mode=FinalPartitioned, gby=[env@0 as env], aggr=[count(Int64(1))] 03)----RepartitionExec: partitioning=Hash([env@0], 4), input_partitions=4 -04)------AggregateExec: mode=Partial, gby=[env@0 as env], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[env@0 as env], aggr=[count(Int64(1))] 05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/aggregate_repartition/dim.csv]]}, projection=[env], file_type=csv, has_header=true @@ -89,13 +89,13 @@ EXPLAIN SELECT env, count(*) FROM dim_parquet GROUP BY env; ---- logical_plan 01)Projection: dim_parquet.env, count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[dim_parquet.env]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[dim_parquet.env]], aggr=[[count(Int64(1))]] 03)----TableScan: dim_parquet projection=[env] physical_plan 01)ProjectionExec: expr=[env@0 as env, count(Int64(1))@1 as count(*)] -02)--AggregateExec: mode=FinalPartitioned, gby=[env@0 as env], aggr=[count() as count(Int64(1))] +02)--AggregateExec: mode=FinalPartitioned, gby=[env@0 as env], aggr=[count(Int64(1))] 03)----RepartitionExec: partitioning=Hash([env@0], 4), input_partitions=1 -04)------AggregateExec: mode=Partial, gby=[env@0 as env], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[env@0 as env], aggr=[count(Int64(1))] 05)--------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/aggregate_repartition/dim.parquet]]}, projection=[env], file_type=parquet # Verify the queries actually work and return the same results @@ -122,11 +122,11 @@ EXPLAIN SELECT env, count(*) FROM dim_parquet GROUP BY env; ---- logical_plan 01)Projection: dim_parquet.env, count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[dim_parquet.env]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[dim_parquet.env]], aggr=[[count(Int64(1))]] 03)----TableScan: dim_parquet projection=[env] physical_plan 01)ProjectionExec: expr=[env@0 as env, count(Int64(1))@1 as count(*)] -02)--AggregateExec: mode=Single, gby=[env@0 as env], aggr=[count() as count(Int64(1))] +02)--AggregateExec: mode=Single, gby=[env@0 as env], aggr=[count(Int64(1))] 03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/aggregate_repartition/dim.parquet]]}, projection=[env], file_type=parquet # Config reset diff --git a/datafusion/sqllogictest/test_files/array/array_has.slt b/datafusion/sqllogictest/test_files/array/array_has.slt index 4fda9253239c7..d7b6680fab062 100644 --- a/datafusion/sqllogictest/test_files/array/array_has.slt +++ b/datafusion/sqllogictest/test_files/array/array_has.slt @@ -505,7 +505,7 @@ select count(*) from test WHERE needle IN ('7f4b18de3cfeb9b4ac78c381ee2ad278', ' ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 03)----SubqueryAlias: test 04)------SubqueryAlias: t 05)--------Projection: @@ -513,9 +513,9 @@ logical_plan 07)------------TableScan: generate_series() projection=[value] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] 05)--------FilterExec: substr(md5(CAST(value@0 AS Utf8View)), 1, 32) IN (SET) ([7f4b18de3cfeb9b4ac78c381ee2ad278, a, b, c]), projection=[] 06)----------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 07)------------LazyMemoryExec: partitions=1, batch_generators=[generate_series: start=1, end=100000, batch_size=8192] @@ -532,7 +532,7 @@ select count(*) from test WHERE needle = ANY(['7f4b18de3cfeb9b4ac78c381ee2ad278' ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 03)----SubqueryAlias: test 04)------SubqueryAlias: t 05)--------Projection: @@ -540,9 +540,9 @@ logical_plan 07)------------TableScan: generate_series() projection=[value] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] 05)--------FilterExec: substr(md5(CAST(value@0 AS Utf8View)), 1, 32) IN (SET) ([7f4b18de3cfeb9b4ac78c381ee2ad278, a, b, c]), projection=[] 06)----------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 07)------------LazyMemoryExec: partitions=1, batch_generators=[generate_series: start=1, end=100000, batch_size=8192] @@ -559,7 +559,7 @@ select count(*) from test WHERE array_has(['7f4b18de3cfeb9b4ac78c381ee2ad278', ' ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 03)----SubqueryAlias: test 04)------SubqueryAlias: t 05)--------Projection: @@ -567,9 +567,9 @@ logical_plan 07)------------TableScan: generate_series() projection=[value] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] 05)--------FilterExec: substr(md5(CAST(value@0 AS Utf8View)), 1, 32) IN (SET) ([7f4b18de3cfeb9b4ac78c381ee2ad278, a, b, c]), projection=[] 06)----------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 07)------------LazyMemoryExec: partitions=1, batch_generators=[generate_series: start=1, end=100000, batch_size=8192] @@ -586,7 +586,7 @@ select count(*) from test WHERE array_has(arrow_cast(['7f4b18de3cfeb9b4ac78c381e ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 03)----SubqueryAlias: test 04)------SubqueryAlias: t 05)--------Projection: @@ -594,9 +594,9 @@ logical_plan 07)------------TableScan: generate_series() projection=[value] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] 05)--------FilterExec: substr(md5(CAST(value@0 AS Utf8View)), 1, 32) IN (SET) ([7f4b18de3cfeb9b4ac78c381ee2ad278, a, b, c]), projection=[] 06)----------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 07)------------LazyMemoryExec: partitions=1, batch_generators=[generate_series: start=1, end=100000, batch_size=8192] @@ -613,7 +613,7 @@ select count(*) from test WHERE array_has(arrow_cast(['7f4b18de3cfeb9b4ac78c381e ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 03)----SubqueryAlias: test 04)------SubqueryAlias: t 05)--------Projection: @@ -621,9 +621,9 @@ logical_plan 07)------------TableScan: generate_series() projection=[value] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] 05)--------FilterExec: substr(md5(CAST(value@0 AS Utf8View)), 1, 32) IN (SET) ([7f4b18de3cfeb9b4ac78c381ee2ad278, a, b, c]), projection=[] 06)----------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 07)------------LazyMemoryExec: partitions=1, batch_generators=[generate_series: start=1, end=100000, batch_size=8192] @@ -641,7 +641,7 @@ select count(*) from test WHERE array_has([needle], needle); ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 03)----SubqueryAlias: test 04)------SubqueryAlias: t 05)--------Projection: @@ -649,9 +649,9 @@ logical_plan 07)------------TableScan: generate_series() projection=[value] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] 05)--------FilterExec: substr(md5(CAST(value@0 AS Utf8View)), 1, 32) IS NOT NULL, projection=[] 06)----------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 07)------------LazyMemoryExec: partitions=1, batch_generators=[generate_series: start=1, end=100000, batch_size=8192] diff --git a/datafusion/sqllogictest/test_files/avro.slt b/datafusion/sqllogictest/test_files/avro.slt index 9e2382aa343f8..04e8fb7a8796d 100644 --- a/datafusion/sqllogictest/test_files/avro.slt +++ b/datafusion/sqllogictest/test_files/avro.slt @@ -265,13 +265,13 @@ EXPLAIN SELECT count(*) from alltypes_plain ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 03)----TableScan: alltypes_plain projection=[] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] 05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/testing/data/avro/alltypes_plain.avro]]}, file_type=avro diff --git a/datafusion/sqllogictest/test_files/clickbench.slt b/datafusion/sqllogictest/test_files/clickbench.slt index 192d97ee2dd02..7cb5547383c38 100644 --- a/datafusion/sqllogictest/test_files/clickbench.slt +++ b/datafusion/sqllogictest/test_files/clickbench.slt @@ -60,7 +60,7 @@ EXPLAIN SELECT COUNT(*) FROM hits; ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 03)----SubqueryAlias: hits 04)------TableScan: hits_raw projection=[] physical_plan @@ -78,16 +78,16 @@ EXPLAIN SELECT COUNT(*) FROM hits WHERE "AdvEngineID" <> 0; ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 03)----SubqueryAlias: hits 04)------Projection: 05)--------Filter: hits_raw.AdvEngineID != Int16(0) 06)----------TableScan: hits_raw projection=[AdvEngineID], partial_filters=[hits_raw.AdvEngineID != Int16(0)] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] 05)--------FilterExec: AdvEngineID@0 != 0, projection=[] 06)----------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 07)------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[AdvEngineID], file_type=parquet, predicate=AdvEngineID@40 != 0, pruning_predicate=AdvEngineID_null_count@2 != row_count@3 AND (AdvEngineID_min@0 != 0 OR 0 != AdvEngineID_max@1), required_guarantees=[AdvEngineID not in (0)] @@ -102,12 +102,12 @@ EXPLAIN SELECT SUM("AdvEngineID"), COUNT(*), AVG("ResolutionWidth") FROM hits; ---- logical_plan 01)Projection: sum(hits.AdvEngineID), count(Int64(1)) AS count(*), avg(hits.ResolutionWidth) -02)--Aggregate: groupBy=[[]], aggr=[[sum(CAST(hits.AdvEngineID AS Int64)), count() AS count(Int64(1)), avg(CAST(hits.ResolutionWidth AS Float64))]] +02)--Aggregate: groupBy=[[]], aggr=[[sum(CAST(hits.AdvEngineID AS Int64)), count(Int64(1)), avg(CAST(hits.ResolutionWidth AS Float64))]] 03)----SubqueryAlias: hits 04)------TableScan: hits_raw projection=[ResolutionWidth, AdvEngineID] physical_plan 01)ProjectionExec: expr=[sum(hits.AdvEngineID)@0 as sum(hits.AdvEngineID), count(Int64(1))@1 as count(*), avg(hits.ResolutionWidth)@2 as avg(hits.ResolutionWidth)] -02)--AggregateExec: mode=Single, gby=[], aggr=[sum(hits.AdvEngineID), count() as count(Int64(1)), avg(hits.ResolutionWidth)] +02)--AggregateExec: mode=Single, gby=[], aggr=[sum(hits.AdvEngineID), count(Int64(1)), avg(hits.ResolutionWidth)] 03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[ResolutionWidth, AdvEngineID], file_type=parquet query IIR @@ -207,7 +207,7 @@ EXPLAIN SELECT "AdvEngineID", COUNT(*) FROM hits WHERE "AdvEngineID" <> 0 GROUP logical_plan 01)Sort: count(*) DESC NULLS FIRST 02)--Projection: hits.AdvEngineID, count(Int64(1)) AS count(*) -03)----Aggregate: groupBy=[[hits.AdvEngineID]], aggr=[[count() AS count(Int64(1))]] +03)----Aggregate: groupBy=[[hits.AdvEngineID]], aggr=[[count(Int64(1))]] 04)------SubqueryAlias: hits 05)--------Filter: hits_raw.AdvEngineID != Int16(0) 06)----------TableScan: hits_raw projection=[AdvEngineID], partial_filters=[hits_raw.AdvEngineID != Int16(0)] @@ -215,9 +215,9 @@ physical_plan 01)SortPreservingMergeExec: [count(*)@1 DESC] 02)--ProjectionExec: expr=[AdvEngineID@0 as AdvEngineID, count(Int64(1))@1 as count(*)] 03)----SortExec: expr=[count(Int64(1))@1 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[AdvEngineID@0 as AdvEngineID], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[AdvEngineID@0 as AdvEngineID], aggr=[count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([AdvEngineID@0], 4), input_partitions=4 -06)----------AggregateExec: mode=Partial, gby=[AdvEngineID@0 as AdvEngineID], aggr=[count() as count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[AdvEngineID@0 as AdvEngineID], aggr=[count(Int64(1))] 07)------------FilterExec: AdvEngineID@0 != 0 08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[AdvEngineID], file_type=parquet, predicate=AdvEngineID@40 != 0, pruning_predicate=AdvEngineID_null_count@2 != row_count@3 AND (AdvEngineID_min@0 != 0 OR 0 != AdvEngineID_max@1), required_guarantees=[AdvEngineID not in (0)] @@ -264,16 +264,16 @@ EXPLAIN SELECT "RegionID", SUM("AdvEngineID"), COUNT(*) AS c, AVG("ResolutionWid logical_plan 01)Sort: c DESC NULLS FIRST, fetch=10 02)--Projection: hits.RegionID, sum(hits.AdvEngineID), count(Int64(1)) AS count(*) AS c, avg(hits.ResolutionWidth), count(DISTINCT hits.UserID) -03)----Aggregate: groupBy=[[hits.RegionID]], aggr=[[sum(CAST(hits.AdvEngineID AS Int64)), count() AS count(Int64(1)), avg(CAST(hits.ResolutionWidth AS Float64)), count(DISTINCT hits.UserID)]] +03)----Aggregate: groupBy=[[hits.RegionID]], aggr=[[sum(CAST(hits.AdvEngineID AS Int64)), count(Int64(1)), avg(CAST(hits.ResolutionWidth AS Float64)), count(DISTINCT hits.UserID)]] 04)------SubqueryAlias: hits 05)--------TableScan: hits_raw projection=[RegionID, UserID, ResolutionWidth, AdvEngineID] physical_plan 01)SortPreservingMergeExec: [c@2 DESC], fetch=10 02)--ProjectionExec: expr=[RegionID@0 as RegionID, sum(hits.AdvEngineID)@1 as sum(hits.AdvEngineID), count(Int64(1))@2 as c, avg(hits.ResolutionWidth)@3 as avg(hits.ResolutionWidth), count(DISTINCT hits.UserID)@4 as count(DISTINCT hits.UserID)] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@2 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[RegionID@0 as RegionID], aggr=[sum(hits.AdvEngineID), count() as count(Int64(1)), avg(hits.ResolutionWidth), count(DISTINCT hits.UserID)] +04)------AggregateExec: mode=FinalPartitioned, gby=[RegionID@0 as RegionID], aggr=[sum(hits.AdvEngineID), count(Int64(1)), avg(hits.ResolutionWidth), count(DISTINCT hits.UserID)] 05)--------RepartitionExec: partitioning=Hash([RegionID@0], 4), input_partitions=1 -06)----------AggregateExec: mode=Partial, gby=[RegionID@0 as RegionID], aggr=[sum(hits.AdvEngineID), count() as count(Int64(1)), avg(hits.ResolutionWidth), count(DISTINCT hits.UserID)] +06)----------AggregateExec: mode=Partial, gby=[RegionID@0 as RegionID], aggr=[sum(hits.AdvEngineID), count(Int64(1)), avg(hits.ResolutionWidth), count(DISTINCT hits.UserID)] 07)------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[RegionID, UserID, ResolutionWidth, AdvEngineID], file_type=parquet query IIIRI rowsort @@ -351,7 +351,7 @@ EXPLAIN SELECT "SearchPhrase", COUNT(*) AS c FROM hits WHERE "SearchPhrase" <> ' logical_plan 01)Sort: c DESC NULLS FIRST, fetch=10 02)--Projection: hits.SearchPhrase, count(Int64(1)) AS count(*) AS c -03)----Aggregate: groupBy=[[hits.SearchPhrase]], aggr=[[count() AS count(Int64(1))]] +03)----Aggregate: groupBy=[[hits.SearchPhrase]], aggr=[[count(Int64(1))]] 04)------SubqueryAlias: hits 05)--------Filter: hits_raw.SearchPhrase != Utf8View("") 06)----------TableScan: hits_raw projection=[SearchPhrase], partial_filters=[hits_raw.SearchPhrase != Utf8View("")] @@ -359,9 +359,9 @@ physical_plan 01)SortPreservingMergeExec: [c@1 DESC], fetch=10 02)--ProjectionExec: expr=[SearchPhrase@0 as SearchPhrase, count(Int64(1))@1 as c] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@1 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[SearchPhrase@0 as SearchPhrase], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[SearchPhrase@0 as SearchPhrase], aggr=[count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([SearchPhrase@0], 4), input_partitions=4 -06)----------AggregateExec: mode=Partial, gby=[SearchPhrase@0 as SearchPhrase], aggr=[count() as count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[SearchPhrase@0 as SearchPhrase], aggr=[count(Int64(1))] 07)------------FilterExec: SearchPhrase@0 != 08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != , pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] @@ -407,7 +407,7 @@ EXPLAIN SELECT "SearchEngineID", "SearchPhrase", COUNT(*) AS c FROM hits WHERE " logical_plan 01)Sort: c DESC NULLS FIRST, fetch=10 02)--Projection: hits.SearchEngineID, hits.SearchPhrase, count(Int64(1)) AS count(*) AS c -03)----Aggregate: groupBy=[[hits.SearchEngineID, hits.SearchPhrase]], aggr=[[count() AS count(Int64(1))]] +03)----Aggregate: groupBy=[[hits.SearchEngineID, hits.SearchPhrase]], aggr=[[count(Int64(1))]] 04)------SubqueryAlias: hits 05)--------Filter: hits_raw.SearchPhrase != Utf8View("") 06)----------TableScan: hits_raw projection=[SearchEngineID, SearchPhrase], partial_filters=[hits_raw.SearchPhrase != Utf8View("")] @@ -415,9 +415,9 @@ physical_plan 01)SortPreservingMergeExec: [c@2 DESC], fetch=10 02)--ProjectionExec: expr=[SearchEngineID@0 as SearchEngineID, SearchPhrase@1 as SearchPhrase, count(Int64(1))@2 as c] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@2 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[SearchEngineID@0 as SearchEngineID, SearchPhrase@1 as SearchPhrase], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[SearchEngineID@0 as SearchEngineID, SearchPhrase@1 as SearchPhrase], aggr=[count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([SearchEngineID@0, SearchPhrase@1], 4), input_partitions=4 -06)----------AggregateExec: mode=Partial, gby=[SearchEngineID@0 as SearchEngineID, SearchPhrase@1 as SearchPhrase], aggr=[count() as count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[SearchEngineID@0 as SearchEngineID, SearchPhrase@1 as SearchPhrase], aggr=[count(Int64(1))] 07)------------FilterExec: SearchPhrase@1 != 08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[SearchEngineID, SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != , pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] @@ -433,16 +433,16 @@ EXPLAIN SELECT "UserID", COUNT(*) FROM hits GROUP BY "UserID" ORDER BY COUNT(*) logical_plan 01)Sort: count(*) DESC NULLS FIRST, fetch=10 02)--Projection: hits.UserID, count(Int64(1)) AS count(*) -03)----Aggregate: groupBy=[[hits.UserID]], aggr=[[count() AS count(Int64(1))]] +03)----Aggregate: groupBy=[[hits.UserID]], aggr=[[count(Int64(1))]] 04)------SubqueryAlias: hits 05)--------TableScan: hits_raw projection=[UserID] physical_plan 01)SortPreservingMergeExec: [count(*)@1 DESC], fetch=10 02)--ProjectionExec: expr=[UserID@0 as UserID, count(Int64(1))@1 as count(*)] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@1 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[UserID@0 as UserID], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[UserID@0 as UserID], aggr=[count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([UserID@0], 4), input_partitions=1 -06)----------AggregateExec: mode=Partial, gby=[UserID@0 as UserID], aggr=[count() as count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[UserID@0 as UserID], aggr=[count(Int64(1))] 07)------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[UserID], file_type=parquet query II rowsort @@ -461,16 +461,16 @@ EXPLAIN SELECT "UserID", "SearchPhrase", COUNT(*) FROM hits GROUP BY "UserID", " logical_plan 01)Sort: count(*) DESC NULLS FIRST, fetch=10 02)--Projection: hits.UserID, hits.SearchPhrase, count(Int64(1)) AS count(*) -03)----Aggregate: groupBy=[[hits.UserID, hits.SearchPhrase]], aggr=[[count() AS count(Int64(1))]] +03)----Aggregate: groupBy=[[hits.UserID, hits.SearchPhrase]], aggr=[[count(Int64(1))]] 04)------SubqueryAlias: hits 05)--------TableScan: hits_raw projection=[UserID, SearchPhrase] physical_plan 01)SortPreservingMergeExec: [count(*)@2 DESC], fetch=10 02)--ProjectionExec: expr=[UserID@0 as UserID, SearchPhrase@1 as SearchPhrase, count(Int64(1))@2 as count(*)] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@2 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[UserID@0 as UserID, SearchPhrase@1 as SearchPhrase], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[UserID@0 as UserID, SearchPhrase@1 as SearchPhrase], aggr=[count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([UserID@0, SearchPhrase@1], 4), input_partitions=1 -06)----------AggregateExec: mode=Partial, gby=[UserID@0 as UserID, SearchPhrase@1 as SearchPhrase], aggr=[count() as count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[UserID@0 as UserID, SearchPhrase@1 as SearchPhrase], aggr=[count(Int64(1))] 07)------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[UserID, SearchPhrase], file_type=parquet query ITI rowsort @@ -489,15 +489,15 @@ EXPLAIN SELECT "UserID", "SearchPhrase", COUNT(*) FROM hits GROUP BY "UserID", " logical_plan 01)Projection: hits.UserID, hits.SearchPhrase, count(Int64(1)) AS count(*) 02)--Limit: skip=0, fetch=10 -03)----Aggregate: groupBy=[[hits.UserID, hits.SearchPhrase]], aggr=[[count() AS count(Int64(1))]] +03)----Aggregate: groupBy=[[hits.UserID, hits.SearchPhrase]], aggr=[[count(Int64(1))]] 04)------SubqueryAlias: hits 05)--------TableScan: hits_raw projection=[UserID, SearchPhrase] physical_plan 01)ProjectionExec: expr=[UserID@0 as UserID, SearchPhrase@1 as SearchPhrase, count(Int64(1))@2 as count(*)] 02)--CoalescePartitionsExec: fetch=10 -03)----AggregateExec: mode=FinalPartitioned, gby=[UserID@0 as UserID, SearchPhrase@1 as SearchPhrase], aggr=[count() as count(Int64(1))] +03)----AggregateExec: mode=FinalPartitioned, gby=[UserID@0 as UserID, SearchPhrase@1 as SearchPhrase], aggr=[count(Int64(1))] 04)------RepartitionExec: partitioning=Hash([UserID@0, SearchPhrase@1], 4), input_partitions=1 -05)--------AggregateExec: mode=Partial, gby=[UserID@0 as UserID, SearchPhrase@1 as SearchPhrase], aggr=[count() as count(Int64(1))] +05)--------AggregateExec: mode=Partial, gby=[UserID@0 as UserID, SearchPhrase@1 as SearchPhrase], aggr=[count(Int64(1))] 06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[UserID, SearchPhrase], file_type=parquet query ITI rowsort @@ -516,16 +516,16 @@ EXPLAIN SELECT "UserID", extract(minute FROM to_timestamp_seconds("EventTime")) logical_plan 01)Sort: count(*) DESC NULLS FIRST, fetch=10 02)--Projection: hits.UserID, date_part(Utf8("MINUTE"),to_timestamp_seconds(hits.EventTime)) AS m, hits.SearchPhrase, count(Int64(1)) AS count(*) -03)----Aggregate: groupBy=[[hits.UserID, date_part(Utf8("MINUTE"), to_timestamp_seconds(hits.EventTime)), hits.SearchPhrase]], aggr=[[count() AS count(Int64(1))]] +03)----Aggregate: groupBy=[[hits.UserID, date_part(Utf8("MINUTE"), to_timestamp_seconds(hits.EventTime)), hits.SearchPhrase]], aggr=[[count(Int64(1))]] 04)------SubqueryAlias: hits 05)--------TableScan: hits_raw projection=[EventTime, UserID, SearchPhrase] physical_plan 01)SortPreservingMergeExec: [count(*)@3 DESC], fetch=10 02)--ProjectionExec: expr=[UserID@0 as UserID, date_part(Utf8("MINUTE"),to_timestamp_seconds(hits.EventTime))@1 as m, SearchPhrase@2 as SearchPhrase, count(Int64(1))@3 as count(*)] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@3 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[UserID@0 as UserID, date_part(Utf8("MINUTE"),to_timestamp_seconds(hits.EventTime))@1 as date_part(Utf8("MINUTE"),to_timestamp_seconds(hits.EventTime)), SearchPhrase@2 as SearchPhrase], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[UserID@0 as UserID, date_part(Utf8("MINUTE"),to_timestamp_seconds(hits.EventTime))@1 as date_part(Utf8("MINUTE"),to_timestamp_seconds(hits.EventTime)), SearchPhrase@2 as SearchPhrase], aggr=[count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([UserID@0, date_part(Utf8("MINUTE"),to_timestamp_seconds(hits.EventTime))@1, SearchPhrase@2], 4), input_partitions=1 -06)----------AggregateExec: mode=Partial, gby=[UserID@1 as UserID, date_part(MINUTE, to_timestamp_seconds(EventTime@0)) as date_part(Utf8("MINUTE"),to_timestamp_seconds(hits.EventTime)), SearchPhrase@2 as SearchPhrase], aggr=[count() as count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[UserID@1 as UserID, date_part(MINUTE, to_timestamp_seconds(EventTime@0)) as date_part(Utf8("MINUTE"),to_timestamp_seconds(hits.EventTime)), SearchPhrase@2 as SearchPhrase], aggr=[count(Int64(1))] 07)------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[EventTime, UserID, SearchPhrase], file_type=parquet query IITI rowsort @@ -565,16 +565,16 @@ EXPLAIN SELECT COUNT(*) FROM hits WHERE "URL" LIKE '%google%'; ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 03)----SubqueryAlias: hits 04)------Projection: 05)--------Filter: hits_raw.URL LIKE Utf8View("%google%") 06)----------TableScan: hits_raw projection=[URL], partial_filters=[hits_raw.URL LIKE Utf8View("%google%")] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] 05)--------FilterExec: URL@0 LIKE %google%, projection=[] 06)----------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 07)------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[URL], file_type=parquet, predicate=URL@13 LIKE %google% @@ -591,7 +591,7 @@ EXPLAIN SELECT "SearchPhrase", MIN("URL"), COUNT(*) AS c FROM hits WHERE "URL" L logical_plan 01)Sort: c DESC NULLS FIRST, fetch=10 02)--Projection: hits.SearchPhrase, min(hits.URL), count(Int64(1)) AS count(*) AS c -03)----Aggregate: groupBy=[[hits.SearchPhrase]], aggr=[[min(hits.URL), count() AS count(Int64(1))]] +03)----Aggregate: groupBy=[[hits.SearchPhrase]], aggr=[[min(hits.URL), count(Int64(1))]] 04)------SubqueryAlias: hits 05)--------Filter: hits_raw.SearchPhrase != Utf8View("") AND hits_raw.URL LIKE Utf8View("%google%") 06)----------TableScan: hits_raw projection=[URL, SearchPhrase], partial_filters=[hits_raw.SearchPhrase != Utf8View(""), hits_raw.URL LIKE Utf8View("%google%")] @@ -599,9 +599,9 @@ physical_plan 01)SortPreservingMergeExec: [c@2 DESC], fetch=10 02)--ProjectionExec: expr=[SearchPhrase@0 as SearchPhrase, min(hits.URL)@1 as min(hits.URL), count(Int64(1))@2 as c] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@2 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[SearchPhrase@0 as SearchPhrase], aggr=[min(hits.URL), count() as count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[SearchPhrase@0 as SearchPhrase], aggr=[min(hits.URL), count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([SearchPhrase@0], 4), input_partitions=4 -06)----------AggregateExec: mode=Partial, gby=[SearchPhrase@1 as SearchPhrase], aggr=[min(hits.URL), count() as count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[SearchPhrase@1 as SearchPhrase], aggr=[min(hits.URL), count(Int64(1))] 07)------------FilterExec: SearchPhrase@1 != AND URL@0 LIKE %google% 08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[URL, SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != AND URL@13 LIKE %google%, pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] @@ -617,7 +617,7 @@ EXPLAIN SELECT "SearchPhrase", MIN("URL"), MIN("Title"), COUNT(*) AS c, COUNT(DI logical_plan 01)Sort: c DESC NULLS FIRST, fetch=10 02)--Projection: hits.SearchPhrase, min(hits.URL), min(hits.Title), count(Int64(1)) AS count(*) AS c, count(DISTINCT hits.UserID) -03)----Aggregate: groupBy=[[hits.SearchPhrase]], aggr=[[min(hits.URL), min(hits.Title), count() AS count(Int64(1)), count(DISTINCT hits.UserID)]] +03)----Aggregate: groupBy=[[hits.SearchPhrase]], aggr=[[min(hits.URL), min(hits.Title), count(Int64(1)), count(DISTINCT hits.UserID)]] 04)------SubqueryAlias: hits 05)--------Filter: hits_raw.SearchPhrase != Utf8View("") AND hits_raw.Title LIKE Utf8View("%Google%") AND hits_raw.URL NOT LIKE Utf8View("%.google.%") 06)----------TableScan: hits_raw projection=[Title, UserID, URL, SearchPhrase], partial_filters=[hits_raw.SearchPhrase != Utf8View(""), hits_raw.Title LIKE Utf8View("%Google%"), hits_raw.URL NOT LIKE Utf8View("%.google.%")] @@ -625,9 +625,9 @@ physical_plan 01)SortPreservingMergeExec: [c@3 DESC], fetch=10 02)--ProjectionExec: expr=[SearchPhrase@0 as SearchPhrase, min(hits.URL)@1 as min(hits.URL), min(hits.Title)@2 as min(hits.Title), count(Int64(1))@3 as c, count(DISTINCT hits.UserID)@4 as count(DISTINCT hits.UserID)] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@3 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[SearchPhrase@0 as SearchPhrase], aggr=[min(hits.URL), min(hits.Title), count() as count(Int64(1)), count(DISTINCT hits.UserID)] +04)------AggregateExec: mode=FinalPartitioned, gby=[SearchPhrase@0 as SearchPhrase], aggr=[min(hits.URL), min(hits.Title), count(Int64(1)), count(DISTINCT hits.UserID)] 05)--------RepartitionExec: partitioning=Hash([SearchPhrase@0], 4), input_partitions=4 -06)----------AggregateExec: mode=Partial, gby=[SearchPhrase@3 as SearchPhrase], aggr=[min(hits.URL), min(hits.Title), count() as count(Int64(1)), count(DISTINCT hits.UserID)] +06)----------AggregateExec: mode=Partial, gby=[SearchPhrase@3 as SearchPhrase], aggr=[min(hits.URL), min(hits.Title), count(Int64(1)), count(DISTINCT hits.UserID)] 07)------------FilterExec: SearchPhrase@3 != AND Title@0 LIKE %Google% AND URL@2 NOT LIKE %.google.% 08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[Title, UserID, URL, SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != AND Title@2 LIKE %Google% AND URL@13 NOT LIKE %.google.%, pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] @@ -734,7 +734,7 @@ logical_plan 01)Sort: l DESC NULLS FIRST, fetch=25 02)--Projection: hits.CounterID, avg(octet_length(hits.URL)) AS l, count(Int64(1)) AS count(*) AS c 03)----Filter: count(Int64(1)) > Int64(100000) -04)------Aggregate: groupBy=[[hits.CounterID]], aggr=[[avg(CAST(octet_length(hits.URL) AS Float64)), count() AS count(Int64(1))]] +04)------Aggregate: groupBy=[[hits.CounterID]], aggr=[[avg(CAST(octet_length(hits.URL) AS Float64)), count(Int64(1))]] 05)--------SubqueryAlias: hits 06)----------Filter: hits_raw.URL != Utf8View("") 07)------------TableScan: hits_raw projection=[CounterID, URL], partial_filters=[hits_raw.URL != Utf8View("")] @@ -743,9 +743,9 @@ physical_plan 02)--ProjectionExec: expr=[CounterID@0 as CounterID, avg(octet_length(hits.URL))@1 as l, count(Int64(1))@2 as c] 03)----SortExec: TopK(fetch=25), expr=[avg(octet_length(hits.URL))@1 DESC], preserve_partitioning=[true] 04)------FilterExec: count(Int64(1))@2 > 100000 -05)--------AggregateExec: mode=FinalPartitioned, gby=[CounterID@0 as CounterID], aggr=[avg(octet_length(hits.URL)), count() as count(Int64(1))] +05)--------AggregateExec: mode=FinalPartitioned, gby=[CounterID@0 as CounterID], aggr=[avg(octet_length(hits.URL)), count(Int64(1))] 06)----------RepartitionExec: partitioning=Hash([CounterID@0], 4), input_partitions=4 -07)------------AggregateExec: mode=Partial, gby=[CounterID@0 as CounterID], aggr=[avg(octet_length(hits.URL)), count() as count(Int64(1))] +07)------------AggregateExec: mode=Partial, gby=[CounterID@0 as CounterID], aggr=[avg(octet_length(hits.URL)), count(Int64(1))] 08)--------------FilterExec: URL@1 != 09)----------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 10)------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[CounterID, URL], file_type=parquet, predicate=URL@13 != , pruning_predicate=URL_null_count@2 != row_count@3 AND (URL_min@0 != OR != URL_max@1), required_guarantees=[URL not in ()] @@ -762,7 +762,7 @@ logical_plan 01)Sort: l DESC NULLS FIRST, fetch=25 02)--Projection: regexp_replace(hits.Referer,Utf8("^https?://(?:www\.)?([^/]+)/.*$"),Utf8("\1")) AS k, avg(octet_length(hits.Referer)) AS l, count(Int64(1)) AS count(*) AS c, min(hits.Referer) 03)----Filter: count(Int64(1)) > Int64(100000) -04)------Aggregate: groupBy=[[regexp_replace(hits.Referer, Utf8View("^https?://(?:www\.)?([^/]+)/.*$"), Utf8View("\1")) AS regexp_replace(hits.Referer,Utf8("^https?://(?:www\.)?([^/]+)/.*$"),Utf8("\1"))]], aggr=[[avg(CAST(octet_length(hits.Referer) AS Float64)), count() AS count(Int64(1)), min(hits.Referer)]] +04)------Aggregate: groupBy=[[regexp_replace(hits.Referer, Utf8View("^https?://(?:www\.)?([^/]+)/.*$"), Utf8View("\1")) AS regexp_replace(hits.Referer,Utf8("^https?://(?:www\.)?([^/]+)/.*$"),Utf8("\1"))]], aggr=[[avg(CAST(octet_length(hits.Referer) AS Float64)), count(Int64(1)), min(hits.Referer)]] 05)--------SubqueryAlias: hits 06)----------Filter: hits_raw.Referer != Utf8View("") 07)------------TableScan: hits_raw projection=[Referer], partial_filters=[hits_raw.Referer != Utf8View("")] @@ -771,9 +771,9 @@ physical_plan 02)--ProjectionExec: expr=[regexp_replace(hits.Referer,Utf8("^https?://(?:www\.)?([^/]+)/.*$"),Utf8("\1"))@0 as k, avg(octet_length(hits.Referer))@1 as l, count(Int64(1))@2 as c, min(hits.Referer)@3 as min(hits.Referer)] 03)----SortExec: TopK(fetch=25), expr=[avg(octet_length(hits.Referer))@1 DESC], preserve_partitioning=[true] 04)------FilterExec: count(Int64(1))@2 > 100000 -05)--------AggregateExec: mode=FinalPartitioned, gby=[regexp_replace(hits.Referer,Utf8("^https?://(?:www\.)?([^/]+)/.*$"),Utf8("\1"))@0 as regexp_replace(hits.Referer,Utf8("^https?://(?:www\.)?([^/]+)/.*$"),Utf8("\1"))], aggr=[avg(octet_length(hits.Referer)), count() as count(Int64(1)), min(hits.Referer)] +05)--------AggregateExec: mode=FinalPartitioned, gby=[regexp_replace(hits.Referer,Utf8("^https?://(?:www\.)?([^/]+)/.*$"),Utf8("\1"))@0 as regexp_replace(hits.Referer,Utf8("^https?://(?:www\.)?([^/]+)/.*$"),Utf8("\1"))], aggr=[avg(octet_length(hits.Referer)), count(Int64(1)), min(hits.Referer)] 06)----------RepartitionExec: partitioning=Hash([regexp_replace(hits.Referer,Utf8("^https?://(?:www\.)?([^/]+)/.*$"),Utf8("\1"))@0], 4), input_partitions=4 -07)------------AggregateExec: mode=Partial, gby=[regexp_replace(Referer@0, ^https?://(?:www\.)?([^/]+)/.*$, \1) as regexp_replace(hits.Referer,Utf8("^https?://(?:www\.)?([^/]+)/.*$"),Utf8("\1"))], aggr=[avg(octet_length(hits.Referer)), count() as count(Int64(1)), min(hits.Referer)] +07)------------AggregateExec: mode=Partial, gby=[regexp_replace(Referer@0, ^https?://(?:www\.)?([^/]+)/.*$, \1) as regexp_replace(hits.Referer,Utf8("^https?://(?:www\.)?([^/]+)/.*$"),Utf8("\1"))], aggr=[avg(octet_length(hits.Referer)), count(Int64(1)), min(hits.Referer)] 08)--------------FilterExec: Referer@0 != 09)----------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 10)------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[Referer], file_type=parquet, predicate=Referer@14 != , pruning_predicate=Referer_null_count@2 != row_count@3 AND (Referer_min@0 != OR != Referer_max@1), required_guarantees=[Referer not in ()] @@ -810,7 +810,7 @@ EXPLAIN SELECT "SearchEngineID", "ClientIP", COUNT(*) AS c, SUM("IsRefresh"), AV logical_plan 01)Sort: c DESC NULLS FIRST, fetch=10 02)--Projection: hits.SearchEngineID, hits.ClientIP, count(Int64(1)) AS count(*) AS c, sum(hits.IsRefresh), avg(hits.ResolutionWidth) -03)----Aggregate: groupBy=[[hits.SearchEngineID, hits.ClientIP]], aggr=[[count() AS count(Int64(1)), sum(CAST(hits.IsRefresh AS Int64)), avg(CAST(hits.ResolutionWidth AS Float64))]] +03)----Aggregate: groupBy=[[hits.SearchEngineID, hits.ClientIP]], aggr=[[count(Int64(1)), sum(CAST(hits.IsRefresh AS Int64)), avg(CAST(hits.ResolutionWidth AS Float64))]] 04)------SubqueryAlias: hits 05)--------Projection: hits_raw.ClientIP, hits_raw.IsRefresh, hits_raw.ResolutionWidth, hits_raw.SearchEngineID 06)----------Filter: hits_raw.SearchPhrase != Utf8View("") @@ -819,9 +819,9 @@ physical_plan 01)SortPreservingMergeExec: [c@2 DESC], fetch=10 02)--ProjectionExec: expr=[SearchEngineID@0 as SearchEngineID, ClientIP@1 as ClientIP, count(Int64(1))@2 as c, sum(hits.IsRefresh)@3 as sum(hits.IsRefresh), avg(hits.ResolutionWidth)@4 as avg(hits.ResolutionWidth)] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@2 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[SearchEngineID@0 as SearchEngineID, ClientIP@1 as ClientIP], aggr=[count() as count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] +04)------AggregateExec: mode=FinalPartitioned, gby=[SearchEngineID@0 as SearchEngineID, ClientIP@1 as ClientIP], aggr=[count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] 05)--------RepartitionExec: partitioning=Hash([SearchEngineID@0, ClientIP@1], 4), input_partitions=4 -06)----------AggregateExec: mode=Partial, gby=[SearchEngineID@3 as SearchEngineID, ClientIP@0 as ClientIP], aggr=[count() as count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] +06)----------AggregateExec: mode=Partial, gby=[SearchEngineID@3 as SearchEngineID, ClientIP@0 as ClientIP], aggr=[count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] 07)------------FilterExec: SearchPhrase@4 != , projection=[ClientIP@0, IsRefresh@1, ResolutionWidth@2, SearchEngineID@3] 08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[ClientIP, IsRefresh, ResolutionWidth, SearchEngineID, SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != , pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] @@ -837,7 +837,7 @@ EXPLAIN SELECT "WatchID", "ClientIP", COUNT(*) AS c, SUM("IsRefresh"), AVG("Reso logical_plan 01)Sort: c DESC NULLS FIRST, fetch=10 02)--Projection: hits.WatchID, hits.ClientIP, count(Int64(1)) AS count(*) AS c, sum(hits.IsRefresh), avg(hits.ResolutionWidth) -03)----Aggregate: groupBy=[[hits.WatchID, hits.ClientIP]], aggr=[[count() AS count(Int64(1)), sum(CAST(hits.IsRefresh AS Int64)), avg(CAST(hits.ResolutionWidth AS Float64))]] +03)----Aggregate: groupBy=[[hits.WatchID, hits.ClientIP]], aggr=[[count(Int64(1)), sum(CAST(hits.IsRefresh AS Int64)), avg(CAST(hits.ResolutionWidth AS Float64))]] 04)------SubqueryAlias: hits 05)--------Projection: hits_raw.WatchID, hits_raw.ClientIP, hits_raw.IsRefresh, hits_raw.ResolutionWidth 06)----------Filter: hits_raw.SearchPhrase != Utf8View("") @@ -846,9 +846,9 @@ physical_plan 01)SortPreservingMergeExec: [c@2 DESC], fetch=10 02)--ProjectionExec: expr=[WatchID@0 as WatchID, ClientIP@1 as ClientIP, count(Int64(1))@2 as c, sum(hits.IsRefresh)@3 as sum(hits.IsRefresh), avg(hits.ResolutionWidth)@4 as avg(hits.ResolutionWidth)] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@2 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[WatchID@0 as WatchID, ClientIP@1 as ClientIP], aggr=[count() as count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] +04)------AggregateExec: mode=FinalPartitioned, gby=[WatchID@0 as WatchID, ClientIP@1 as ClientIP], aggr=[count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] 05)--------RepartitionExec: partitioning=Hash([WatchID@0, ClientIP@1], 4), input_partitions=4 -06)----------AggregateExec: mode=Partial, gby=[WatchID@0 as WatchID, ClientIP@1 as ClientIP], aggr=[count() as count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] +06)----------AggregateExec: mode=Partial, gby=[WatchID@0 as WatchID, ClientIP@1 as ClientIP], aggr=[count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] 07)------------FilterExec: SearchPhrase@4 != , projection=[WatchID@0, ClientIP@1, IsRefresh@2, ResolutionWidth@3] 08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[WatchID, ClientIP, IsRefresh, ResolutionWidth, SearchPhrase], file_type=parquet, predicate=SearchPhrase@39 != , pruning_predicate=SearchPhrase_null_count@2 != row_count@3 AND (SearchPhrase_min@0 != OR != SearchPhrase_max@1), required_guarantees=[SearchPhrase not in ()] @@ -864,16 +864,16 @@ EXPLAIN SELECT "WatchID", "ClientIP", COUNT(*) AS c, SUM("IsRefresh"), AVG("Reso logical_plan 01)Sort: c DESC NULLS FIRST, fetch=10 02)--Projection: hits.WatchID, hits.ClientIP, count(Int64(1)) AS count(*) AS c, sum(hits.IsRefresh), avg(hits.ResolutionWidth) -03)----Aggregate: groupBy=[[hits.WatchID, hits.ClientIP]], aggr=[[count() AS count(Int64(1)), sum(CAST(hits.IsRefresh AS Int64)), avg(CAST(hits.ResolutionWidth AS Float64))]] +03)----Aggregate: groupBy=[[hits.WatchID, hits.ClientIP]], aggr=[[count(Int64(1)), sum(CAST(hits.IsRefresh AS Int64)), avg(CAST(hits.ResolutionWidth AS Float64))]] 04)------SubqueryAlias: hits 05)--------TableScan: hits_raw projection=[WatchID, ClientIP, IsRefresh, ResolutionWidth] physical_plan 01)SortPreservingMergeExec: [c@2 DESC], fetch=10 02)--ProjectionExec: expr=[WatchID@0 as WatchID, ClientIP@1 as ClientIP, count(Int64(1))@2 as c, sum(hits.IsRefresh)@3 as sum(hits.IsRefresh), avg(hits.ResolutionWidth)@4 as avg(hits.ResolutionWidth)] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@2 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[WatchID@0 as WatchID, ClientIP@1 as ClientIP], aggr=[count() as count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] +04)------AggregateExec: mode=FinalPartitioned, gby=[WatchID@0 as WatchID, ClientIP@1 as ClientIP], aggr=[count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] 05)--------RepartitionExec: partitioning=Hash([WatchID@0, ClientIP@1], 4), input_partitions=1 -06)----------AggregateExec: mode=Partial, gby=[WatchID@0 as WatchID, ClientIP@1 as ClientIP], aggr=[count() as count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] +06)----------AggregateExec: mode=Partial, gby=[WatchID@0 as WatchID, ClientIP@1 as ClientIP], aggr=[count(Int64(1)), sum(hits.IsRefresh), avg(hits.ResolutionWidth)] 07)------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[WatchID, ClientIP, IsRefresh, ResolutionWidth], file_type=parquet query IIIIR rowsort @@ -897,16 +897,16 @@ EXPLAIN SELECT "URL", COUNT(*) AS c FROM hits GROUP BY "URL" ORDER BY c DESC LIM logical_plan 01)Sort: c DESC NULLS FIRST, fetch=10 02)--Projection: hits.URL, count(Int64(1)) AS count(*) AS c -03)----Aggregate: groupBy=[[hits.URL]], aggr=[[count() AS count(Int64(1))]] +03)----Aggregate: groupBy=[[hits.URL]], aggr=[[count(Int64(1))]] 04)------SubqueryAlias: hits 05)--------TableScan: hits_raw projection=[URL] physical_plan 01)SortPreservingMergeExec: [c@1 DESC], fetch=10 02)--ProjectionExec: expr=[URL@0 as URL, count(Int64(1))@1 as c] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@1 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[URL@0 as URL], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[URL@0 as URL], aggr=[count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([URL@0], 4), input_partitions=1 -06)----------AggregateExec: mode=Partial, gby=[URL@0 as URL], aggr=[count() as count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[URL@0 as URL], aggr=[count(Int64(1))] 07)------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[URL], file_type=parquet query TI rowsort @@ -926,16 +926,16 @@ EXPLAIN SELECT 1, "URL", COUNT(*) AS c FROM hits GROUP BY 1, "URL" ORDER BY c DE logical_plan 01)Sort: c DESC NULLS FIRST, fetch=10 02)--Projection: Int64(1), hits.URL, count(Int64(1)) AS c -03)----Aggregate: groupBy=[[hits.URL]], aggr=[[count() AS count(Int64(1))]] +03)----Aggregate: groupBy=[[hits.URL]], aggr=[[count(Int64(1))]] 04)------SubqueryAlias: hits 05)--------TableScan: hits_raw projection=[URL] physical_plan 01)SortPreservingMergeExec: [c@2 DESC], fetch=10 02)--SortExec: TopK(fetch=10), expr=[c@2 DESC], preserve_partitioning=[true] 03)----ProjectionExec: expr=[1 as Int64(1), URL@0 as URL, count(Int64(1))@1 as c] -04)------AggregateExec: mode=FinalPartitioned, gby=[URL@0 as URL], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[URL@0 as URL], aggr=[count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([URL@0], 4), input_partitions=1 -06)----------AggregateExec: mode=Partial, gby=[URL@0 as URL], aggr=[count() as count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[URL@0 as URL], aggr=[count(Int64(1))] 07)------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[URL], file_type=parquet query ITI rowsort @@ -956,7 +956,7 @@ logical_plan 01)Sort: c DESC NULLS FIRST, fetch=10 02)--Projection: hits.ClientIP, __common_expr_1 - Int64(1) AS hits.ClientIP - Int64(1), __common_expr_1 - Int64(2) AS hits.ClientIP - Int64(2), __common_expr_1 - Int64(3) AS hits.ClientIP - Int64(3), count(Int64(1)) AS c 03)----Projection: CAST(hits.ClientIP AS Int64) AS __common_expr_1, hits.ClientIP, count(Int64(1)) -04)------Aggregate: groupBy=[[hits.ClientIP]], aggr=[[count() AS count(Int64(1))]] +04)------Aggregate: groupBy=[[hits.ClientIP]], aggr=[[count(Int64(1))]] 05)--------SubqueryAlias: hits 06)----------TableScan: hits_raw projection=[ClientIP] physical_plan @@ -964,9 +964,9 @@ physical_plan 02)--SortExec: TopK(fetch=10), expr=[c@4 DESC], preserve_partitioning=[true] 03)----ProjectionExec: expr=[ClientIP@1 as ClientIP, __common_expr_1@0 - 1 as hits.ClientIP - Int64(1), __common_expr_1@0 - 2 as hits.ClientIP - Int64(2), __common_expr_1@0 - 3 as hits.ClientIP - Int64(3), count(Int64(1))@2 as c] 04)------ProjectionExec: expr=[CAST(ClientIP@0 AS Int64) as __common_expr_1, ClientIP@0 as ClientIP, count(Int64(1))@1 as count(Int64(1))] -05)--------AggregateExec: mode=FinalPartitioned, gby=[ClientIP@0 as ClientIP], aggr=[count() as count(Int64(1))] +05)--------AggregateExec: mode=FinalPartitioned, gby=[ClientIP@0 as ClientIP], aggr=[count(Int64(1))] 06)----------RepartitionExec: partitioning=Hash([ClientIP@0], 4), input_partitions=1 -07)------------AggregateExec: mode=Partial, gby=[ClientIP@0 as ClientIP], aggr=[count() as count(Int64(1))] +07)------------AggregateExec: mode=Partial, gby=[ClientIP@0 as ClientIP], aggr=[count(Int64(1))] 08)--------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[ClientIP], file_type=parquet query IIIII rowsort @@ -984,7 +984,7 @@ EXPLAIN SELECT "URL", COUNT(*) AS PageViews FROM hits WHERE "CounterID" = 62 AND logical_plan 01)Sort: pageviews DESC NULLS FIRST, fetch=10 02)--Projection: hits.URL, count(Int64(1)) AS count(*) AS pageviews -03)----Aggregate: groupBy=[[hits.URL]], aggr=[[count() AS count(Int64(1))]] +03)----Aggregate: groupBy=[[hits.URL]], aggr=[[count(Int64(1))]] 04)------SubqueryAlias: hits 05)--------Projection: hits_raw.URL 06)----------Filter: hits_raw.CounterID = Int32(62) AND hits_raw.EventDate >= UInt16(15887) AND hits_raw.EventDate <= UInt16(15917) AND hits_raw.DontCountHits = Int16(0) AND hits_raw.IsRefresh = Int16(0) AND hits_raw.URL != Utf8View("") @@ -993,9 +993,9 @@ physical_plan 01)SortPreservingMergeExec: [pageviews@1 DESC], fetch=10 02)--ProjectionExec: expr=[URL@0 as URL, count(Int64(1))@1 as pageviews] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@1 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[URL@0 as URL], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[URL@0 as URL], aggr=[count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([URL@0], 4), input_partitions=4 -06)----------AggregateExec: mode=Partial, gby=[URL@0 as URL], aggr=[count() as count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[URL@0 as URL], aggr=[count(Int64(1))] 07)------------FilterExec: CounterID@1 = 62 AND EventDate@0 >= 15887 AND EventDate@0 <= 15917 AND DontCountHits@4 = 0 AND IsRefresh@3 = 0 AND URL@2 != , projection=[URL@2] 08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[EventDate, CounterID, URL, IsRefresh, DontCountHits], file_type=parquet, predicate=CounterID@6 = 62 AND EventDate@5 >= 15887 AND EventDate@5 <= 15917 AND DontCountHits@61 = 0 AND IsRefresh@15 = 0 AND URL@13 != , pruning_predicate=CounterID_null_count@2 != row_count@3 AND CounterID_min@0 <= 62 AND 62 <= CounterID_max@1 AND EventDate_null_count@5 != row_count@3 AND EventDate_max@4 >= 15887 AND EventDate_null_count@5 != row_count@3 AND EventDate_min@6 <= 15917 AND DontCountHits_null_count@9 != row_count@3 AND DontCountHits_min@7 <= 0 AND 0 <= DontCountHits_max@8 AND IsRefresh_null_count@12 != row_count@3 AND IsRefresh_min@10 <= 0 AND 0 <= IsRefresh_max@11 AND URL_null_count@15 != row_count@3 AND (URL_min@13 != OR != URL_max@14), required_guarantees=[CounterID in (62), DontCountHits in (0), IsRefresh in (0), URL not in ()] @@ -1011,7 +1011,7 @@ EXPLAIN SELECT "Title", COUNT(*) AS PageViews FROM hits WHERE "CounterID" = 62 A logical_plan 01)Sort: pageviews DESC NULLS FIRST, fetch=10 02)--Projection: hits.Title, count(Int64(1)) AS count(*) AS pageviews -03)----Aggregate: groupBy=[[hits.Title]], aggr=[[count() AS count(Int64(1))]] +03)----Aggregate: groupBy=[[hits.Title]], aggr=[[count(Int64(1))]] 04)------SubqueryAlias: hits 05)--------Projection: hits_raw.Title 06)----------Filter: hits_raw.CounterID = Int32(62) AND hits_raw.EventDate >= UInt16(15887) AND hits_raw.EventDate <= UInt16(15917) AND hits_raw.DontCountHits = Int16(0) AND hits_raw.IsRefresh = Int16(0) AND hits_raw.Title != Utf8View("") @@ -1020,9 +1020,9 @@ physical_plan 01)SortPreservingMergeExec: [pageviews@1 DESC], fetch=10 02)--ProjectionExec: expr=[Title@0 as Title, count(Int64(1))@1 as pageviews] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@1 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[Title@0 as Title], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[Title@0 as Title], aggr=[count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([Title@0], 4), input_partitions=4 -06)----------AggregateExec: mode=Partial, gby=[Title@0 as Title], aggr=[count() as count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[Title@0 as Title], aggr=[count(Int64(1))] 07)------------FilterExec: CounterID@2 = 62 AND EventDate@1 >= 15887 AND EventDate@1 <= 15917 AND DontCountHits@4 = 0 AND IsRefresh@3 = 0 AND Title@0 != , projection=[Title@0] 08)--------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 09)----------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[Title, EventDate, CounterID, IsRefresh, DontCountHits], file_type=parquet, predicate=CounterID@6 = 62 AND EventDate@5 >= 15887 AND EventDate@5 <= 15917 AND DontCountHits@61 = 0 AND IsRefresh@15 = 0 AND Title@2 != , pruning_predicate=CounterID_null_count@2 != row_count@3 AND CounterID_min@0 <= 62 AND 62 <= CounterID_max@1 AND EventDate_null_count@5 != row_count@3 AND EventDate_max@4 >= 15887 AND EventDate_null_count@5 != row_count@3 AND EventDate_min@6 <= 15917 AND DontCountHits_null_count@9 != row_count@3 AND DontCountHits_min@7 <= 0 AND 0 <= DontCountHits_max@8 AND IsRefresh_null_count@12 != row_count@3 AND IsRefresh_min@10 <= 0 AND 0 <= IsRefresh_max@11 AND Title_null_count@15 != row_count@3 AND (Title_min@13 != OR != Title_max@14), required_guarantees=[CounterID in (62), DontCountHits in (0), IsRefresh in (0), Title not in ()] @@ -1039,7 +1039,7 @@ logical_plan 01)Limit: skip=1000, fetch=10 02)--Sort: pageviews DESC NULLS FIRST, fetch=1010 03)----Projection: hits.URL, count(Int64(1)) AS count(*) AS pageviews -04)------Aggregate: groupBy=[[hits.URL]], aggr=[[count() AS count(Int64(1))]] +04)------Aggregate: groupBy=[[hits.URL]], aggr=[[count(Int64(1))]] 05)--------SubqueryAlias: hits 06)----------Projection: hits_raw.URL 07)------------Filter: hits_raw.CounterID = Int32(62) AND hits_raw.EventDate >= UInt16(15887) AND hits_raw.EventDate <= UInt16(15917) AND hits_raw.IsRefresh = Int16(0) AND hits_raw.IsLink != Int16(0) AND hits_raw.IsDownload = Int16(0) @@ -1049,9 +1049,9 @@ physical_plan 02)--SortPreservingMergeExec: [pageviews@1 DESC], fetch=1010 03)----ProjectionExec: expr=[URL@0 as URL, count(Int64(1))@1 as pageviews] 04)------SortExec: TopK(fetch=1010), expr=[count(Int64(1))@1 DESC], preserve_partitioning=[true] -05)--------AggregateExec: mode=FinalPartitioned, gby=[URL@0 as URL], aggr=[count() as count(Int64(1))] +05)--------AggregateExec: mode=FinalPartitioned, gby=[URL@0 as URL], aggr=[count(Int64(1))] 06)----------RepartitionExec: partitioning=Hash([URL@0], 4), input_partitions=4 -07)------------AggregateExec: mode=Partial, gby=[URL@0 as URL], aggr=[count() as count(Int64(1))] +07)------------AggregateExec: mode=Partial, gby=[URL@0 as URL], aggr=[count(Int64(1))] 08)--------------FilterExec: CounterID@1 = 62 AND EventDate@0 >= 15887 AND EventDate@0 <= 15917 AND IsRefresh@3 = 0 AND IsLink@4 != 0 AND IsDownload@5 = 0, projection=[URL@2] 09)----------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 10)------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[EventDate, CounterID, URL, IsRefresh, IsLink, IsDownload], file_type=parquet, predicate=CounterID@6 = 62 AND EventDate@5 >= 15887 AND EventDate@5 <= 15917 AND IsRefresh@15 = 0 AND IsLink@52 != 0 AND IsDownload@53 = 0, pruning_predicate=CounterID_null_count@2 != row_count@3 AND CounterID_min@0 <= 62 AND 62 <= CounterID_max@1 AND EventDate_null_count@5 != row_count@3 AND EventDate_max@4 >= 15887 AND EventDate_null_count@5 != row_count@3 AND EventDate_min@6 <= 15917 AND IsRefresh_null_count@9 != row_count@3 AND IsRefresh_min@7 <= 0 AND 0 <= IsRefresh_max@8 AND IsLink_null_count@12 != row_count@3 AND (IsLink_min@10 != 0 OR 0 != IsLink_max@11) AND IsDownload_null_count@15 != row_count@3 AND IsDownload_min@13 <= 0 AND 0 <= IsDownload_max@14, required_guarantees=[CounterID in (62), IsDownload in (0), IsLink not in (0), IsRefresh in (0)] @@ -1068,7 +1068,7 @@ logical_plan 01)Limit: skip=1000, fetch=10 02)--Sort: pageviews DESC NULLS FIRST, fetch=1010 03)----Projection: hits.TraficSourceID, hits.SearchEngineID, hits.AdvEngineID, CASE WHEN hits.SearchEngineID = Int64(0) AND hits.AdvEngineID = Int64(0) THEN hits.Referer ELSE Utf8("") END AS src, hits.URL AS dst, count(Int64(1)) AS count(*) AS pageviews -04)------Aggregate: groupBy=[[hits.TraficSourceID, hits.SearchEngineID, hits.AdvEngineID, CASE WHEN hits.SearchEngineID = Int16(0) AND hits.AdvEngineID = Int16(0) THEN hits.Referer ELSE Utf8View("") END AS CASE WHEN hits.SearchEngineID = Int64(0) AND hits.AdvEngineID = Int64(0) THEN hits.Referer ELSE Utf8("") END, hits.URL]], aggr=[[count() AS count(Int64(1))]] +04)------Aggregate: groupBy=[[hits.TraficSourceID, hits.SearchEngineID, hits.AdvEngineID, CASE WHEN hits.SearchEngineID = Int16(0) AND hits.AdvEngineID = Int16(0) THEN hits.Referer ELSE Utf8View("") END AS CASE WHEN hits.SearchEngineID = Int64(0) AND hits.AdvEngineID = Int64(0) THEN hits.Referer ELSE Utf8("") END, hits.URL]], aggr=[[count(Int64(1))]] 05)--------SubqueryAlias: hits 06)----------Projection: hits_raw.URL, hits_raw.Referer, hits_raw.TraficSourceID, hits_raw.SearchEngineID, hits_raw.AdvEngineID 07)------------Filter: hits_raw.CounterID = Int32(62) AND hits_raw.EventDate >= UInt16(15887) AND hits_raw.EventDate <= UInt16(15917) AND hits_raw.IsRefresh = Int16(0) @@ -1078,9 +1078,9 @@ physical_plan 02)--SortPreservingMergeExec: [pageviews@5 DESC], fetch=1010 03)----ProjectionExec: expr=[TraficSourceID@0 as TraficSourceID, SearchEngineID@1 as SearchEngineID, AdvEngineID@2 as AdvEngineID, CASE WHEN hits.SearchEngineID = Int64(0) AND hits.AdvEngineID = Int64(0) THEN hits.Referer ELSE Utf8("") END@3 as src, URL@4 as dst, count(Int64(1))@5 as pageviews] 04)------SortExec: TopK(fetch=1010), expr=[count(Int64(1))@5 DESC], preserve_partitioning=[true] -05)--------AggregateExec: mode=FinalPartitioned, gby=[TraficSourceID@0 as TraficSourceID, SearchEngineID@1 as SearchEngineID, AdvEngineID@2 as AdvEngineID, CASE WHEN hits.SearchEngineID = Int64(0) AND hits.AdvEngineID = Int64(0) THEN hits.Referer ELSE Utf8("") END@3 as CASE WHEN hits.SearchEngineID = Int64(0) AND hits.AdvEngineID = Int64(0) THEN hits.Referer ELSE Utf8("") END, URL@4 as URL], aggr=[count() as count(Int64(1))] +05)--------AggregateExec: mode=FinalPartitioned, gby=[TraficSourceID@0 as TraficSourceID, SearchEngineID@1 as SearchEngineID, AdvEngineID@2 as AdvEngineID, CASE WHEN hits.SearchEngineID = Int64(0) AND hits.AdvEngineID = Int64(0) THEN hits.Referer ELSE Utf8("") END@3 as CASE WHEN hits.SearchEngineID = Int64(0) AND hits.AdvEngineID = Int64(0) THEN hits.Referer ELSE Utf8("") END, URL@4 as URL], aggr=[count(Int64(1))] 06)----------RepartitionExec: partitioning=Hash([TraficSourceID@0, SearchEngineID@1, AdvEngineID@2, CASE WHEN hits.SearchEngineID = Int64(0) AND hits.AdvEngineID = Int64(0) THEN hits.Referer ELSE Utf8("") END@3, URL@4], 4), input_partitions=4 -07)------------AggregateExec: mode=Partial, gby=[TraficSourceID@2 as TraficSourceID, SearchEngineID@3 as SearchEngineID, AdvEngineID@4 as AdvEngineID, CASE WHEN SearchEngineID@3 = 0 AND AdvEngineID@4 = 0 THEN Referer@1 ELSE END as CASE WHEN hits.SearchEngineID = Int64(0) AND hits.AdvEngineID = Int64(0) THEN hits.Referer ELSE Utf8("") END, URL@0 as URL], aggr=[count() as count(Int64(1))] +07)------------AggregateExec: mode=Partial, gby=[TraficSourceID@2 as TraficSourceID, SearchEngineID@3 as SearchEngineID, AdvEngineID@4 as AdvEngineID, CASE WHEN SearchEngineID@3 = 0 AND AdvEngineID@4 = 0 THEN Referer@1 ELSE END as CASE WHEN hits.SearchEngineID = Int64(0) AND hits.AdvEngineID = Int64(0) THEN hits.Referer ELSE Utf8("") END, URL@0 as URL], aggr=[count(Int64(1))] 08)--------------FilterExec: CounterID@1 = 62 AND EventDate@0 >= 15887 AND EventDate@0 <= 15917 AND IsRefresh@4 = 0, projection=[URL@2, Referer@3, TraficSourceID@5, SearchEngineID@6, AdvEngineID@7] 09)----------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 10)------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[EventDate, CounterID, URL, Referer, IsRefresh, TraficSourceID, SearchEngineID, AdvEngineID], file_type=parquet, predicate=CounterID@6 = 62 AND EventDate@5 >= 15887 AND EventDate@5 <= 15917 AND IsRefresh@15 = 0, pruning_predicate=CounterID_null_count@2 != row_count@3 AND CounterID_min@0 <= 62 AND 62 <= CounterID_max@1 AND EventDate_null_count@5 != row_count@3 AND EventDate_max@4 >= 15887 AND EventDate_null_count@5 != row_count@3 AND EventDate_min@6 <= 15917 AND IsRefresh_null_count@9 != row_count@3 AND IsRefresh_min@7 <= 0 AND 0 <= IsRefresh_max@8, required_guarantees=[CounterID in (62), IsRefresh in (0)] @@ -1097,7 +1097,7 @@ logical_plan 01)Limit: skip=100, fetch=10 02)--Sort: pageviews DESC NULLS FIRST, fetch=110 03)----Projection: hits.URLHash, hits.EventDate, count(Int64(1)) AS count(*) AS pageviews -04)------Aggregate: groupBy=[[hits.URLHash, hits.EventDate]], aggr=[[count() AS count(Int64(1))]] +04)------Aggregate: groupBy=[[hits.URLHash, hits.EventDate]], aggr=[[count(Int64(1))]] 05)--------SubqueryAlias: hits 06)----------Projection: hits_raw.URLHash, CAST(CAST(hits_raw.EventDate AS Int32) AS Date32) AS EventDate 07)------------Filter: hits_raw.CounterID = Int32(62) AND hits_raw.EventDate >= UInt16(15887) AND hits_raw.EventDate <= UInt16(15917) AND hits_raw.IsRefresh = Int16(0) AND (hits_raw.TraficSourceID = Int16(-1) OR hits_raw.TraficSourceID = Int16(6)) AND hits_raw.RefererHash = Int64(3594120000172545465) @@ -1107,9 +1107,9 @@ physical_plan 02)--SortPreservingMergeExec: [pageviews@2 DESC], fetch=110 03)----ProjectionExec: expr=[URLHash@0 as URLHash, EventDate@1 as EventDate, count(Int64(1))@2 as pageviews] 04)------SortExec: TopK(fetch=110), expr=[count(Int64(1))@2 DESC], preserve_partitioning=[true] -05)--------AggregateExec: mode=FinalPartitioned, gby=[URLHash@0 as URLHash, EventDate@1 as EventDate], aggr=[count() as count(Int64(1))] +05)--------AggregateExec: mode=FinalPartitioned, gby=[URLHash@0 as URLHash, EventDate@1 as EventDate], aggr=[count(Int64(1))] 06)----------RepartitionExec: partitioning=Hash([URLHash@0, EventDate@1], 4), input_partitions=4 -07)------------AggregateExec: mode=Partial, gby=[URLHash@0 as URLHash, EventDate@1 as EventDate], aggr=[count() as count(Int64(1))] +07)------------AggregateExec: mode=Partial, gby=[URLHash@0 as URLHash, EventDate@1 as EventDate], aggr=[count(Int64(1))] 08)--------------ProjectionExec: expr=[URLHash@0 as URLHash, CAST(CAST(EventDate@1 AS Int32) AS Date32) as EventDate] 09)----------------FilterExec: CounterID@1 = 62 AND EventDate@0 >= 15887 AND EventDate@0 <= 15917 AND IsRefresh@2 = 0 AND (TraficSourceID@3 = -1 OR TraficSourceID@3 = 6) AND RefererHash@4 = 3594120000172545465, projection=[URLHash@5, EventDate@0] 10)------------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 @@ -1127,7 +1127,7 @@ logical_plan 01)Limit: skip=10000, fetch=10 02)--Sort: pageviews DESC NULLS FIRST, fetch=10010 03)----Projection: hits.WindowClientWidth, hits.WindowClientHeight, count(Int64(1)) AS count(*) AS pageviews -04)------Aggregate: groupBy=[[hits.WindowClientWidth, hits.WindowClientHeight]], aggr=[[count() AS count(Int64(1))]] +04)------Aggregate: groupBy=[[hits.WindowClientWidth, hits.WindowClientHeight]], aggr=[[count(Int64(1))]] 05)--------SubqueryAlias: hits 06)----------Projection: hits_raw.WindowClientWidth, hits_raw.WindowClientHeight 07)------------Filter: hits_raw.CounterID = Int32(62) AND hits_raw.EventDate >= UInt16(15887) AND hits_raw.EventDate <= UInt16(15917) AND hits_raw.IsRefresh = Int16(0) AND hits_raw.DontCountHits = Int16(0) AND hits_raw.URLHash = Int64(2868770270353813622) @@ -1137,9 +1137,9 @@ physical_plan 02)--SortPreservingMergeExec: [pageviews@2 DESC], fetch=10010 03)----ProjectionExec: expr=[WindowClientWidth@0 as WindowClientWidth, WindowClientHeight@1 as WindowClientHeight, count(Int64(1))@2 as pageviews] 04)------SortExec: TopK(fetch=10010), expr=[count(Int64(1))@2 DESC], preserve_partitioning=[true] -05)--------AggregateExec: mode=FinalPartitioned, gby=[WindowClientWidth@0 as WindowClientWidth, WindowClientHeight@1 as WindowClientHeight], aggr=[count() as count(Int64(1))] +05)--------AggregateExec: mode=FinalPartitioned, gby=[WindowClientWidth@0 as WindowClientWidth, WindowClientHeight@1 as WindowClientHeight], aggr=[count(Int64(1))] 06)----------RepartitionExec: partitioning=Hash([WindowClientWidth@0, WindowClientHeight@1], 4), input_partitions=4 -07)------------AggregateExec: mode=Partial, gby=[WindowClientWidth@0 as WindowClientWidth, WindowClientHeight@1 as WindowClientHeight], aggr=[count() as count(Int64(1))] +07)------------AggregateExec: mode=Partial, gby=[WindowClientWidth@0 as WindowClientWidth, WindowClientHeight@1 as WindowClientHeight], aggr=[count(Int64(1))] 08)--------------FilterExec: CounterID@1 = 62 AND EventDate@0 >= 15887 AND EventDate@0 <= 15917 AND IsRefresh@2 = 0 AND DontCountHits@5 = 0 AND URLHash@6 = 2868770270353813622, projection=[WindowClientWidth@3, WindowClientHeight@4] 09)----------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 10)------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[EventDate, CounterID, IsRefresh, WindowClientWidth, WindowClientHeight, DontCountHits, URLHash], file_type=parquet, predicate=CounterID@6 = 62 AND EventDate@5 >= 15887 AND EventDate@5 <= 15917 AND IsRefresh@15 = 0 AND DontCountHits@61 = 0 AND URLHash@103 = 2868770270353813622, pruning_predicate=CounterID_null_count@2 != row_count@3 AND CounterID_min@0 <= 62 AND 62 <= CounterID_max@1 AND EventDate_null_count@5 != row_count@3 AND EventDate_max@4 >= 15887 AND EventDate_null_count@5 != row_count@3 AND EventDate_min@6 <= 15917 AND IsRefresh_null_count@9 != row_count@3 AND IsRefresh_min@7 <= 0 AND 0 <= IsRefresh_max@8 AND DontCountHits_null_count@12 != row_count@3 AND DontCountHits_min@10 <= 0 AND 0 <= DontCountHits_max@11 AND URLHash_null_count@15 != row_count@3 AND URLHash_min@13 <= 2868770270353813622 AND 2868770270353813622 <= URLHash_max@14, required_guarantees=[CounterID in (62), DontCountHits in (0), IsRefresh in (0), URLHash in (2868770270353813622)] @@ -1156,7 +1156,7 @@ logical_plan 01)Limit: skip=1000, fetch=10 02)--Sort: date_trunc(Utf8("minute"), m) ASC NULLS LAST, fetch=1010 03)----Projection: date_trunc(Utf8("minute"),to_timestamp_seconds(hits.EventTime)) AS m, count(Int64(1)) AS count(*) AS pageviews -04)------Aggregate: groupBy=[[date_trunc(Utf8("minute"), to_timestamp_seconds(hits.EventTime))]], aggr=[[count() AS count(Int64(1))]] +04)------Aggregate: groupBy=[[date_trunc(Utf8("minute"), to_timestamp_seconds(hits.EventTime))]], aggr=[[count(Int64(1))]] 05)--------SubqueryAlias: hits 06)----------Projection: hits_raw.EventTime 07)------------Filter: hits_raw.CounterID = Int32(62) AND hits_raw.EventDate >= UInt16(15900) AND hits_raw.EventDate <= UInt16(15901) AND hits_raw.IsRefresh = Int16(0) AND hits_raw.DontCountHits = Int16(0) @@ -1166,9 +1166,9 @@ physical_plan 02)--SortPreservingMergeExec: [date_trunc(minute, m@0) ASC NULLS LAST], fetch=1010 03)----SortExec: TopK(fetch=1010), expr=[date_trunc(minute, m@0) ASC NULLS LAST], preserve_partitioning=[true] 04)------ProjectionExec: expr=[date_trunc(Utf8("minute"),to_timestamp_seconds(hits.EventTime))@0 as m, count(Int64(1))@1 as pageviews] -05)--------AggregateExec: mode=FinalPartitioned, gby=[date_trunc(Utf8("minute"),to_timestamp_seconds(hits.EventTime))@0 as date_trunc(Utf8("minute"),to_timestamp_seconds(hits.EventTime))], aggr=[count() as count(Int64(1))] +05)--------AggregateExec: mode=FinalPartitioned, gby=[date_trunc(Utf8("minute"),to_timestamp_seconds(hits.EventTime))@0 as date_trunc(Utf8("minute"),to_timestamp_seconds(hits.EventTime))], aggr=[count(Int64(1))] 06)----------RepartitionExec: partitioning=Hash([date_trunc(Utf8("minute"),to_timestamp_seconds(hits.EventTime))@0], 4), input_partitions=4 -07)------------AggregateExec: mode=Partial, gby=[date_trunc(minute, to_timestamp_seconds(EventTime@0)) as date_trunc(Utf8("minute"),to_timestamp_seconds(hits.EventTime))], aggr=[count() as count(Int64(1))] +07)------------AggregateExec: mode=Partial, gby=[date_trunc(minute, to_timestamp_seconds(EventTime@0)) as date_trunc(Utf8("minute"),to_timestamp_seconds(hits.EventTime))], aggr=[count(Int64(1))] 08)--------------FilterExec: CounterID@2 = 62 AND EventDate@1 >= 15900 AND EventDate@1 <= 15901 AND IsRefresh@3 = 0 AND DontCountHits@4 = 0, projection=[EventTime@0] 09)----------------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 10)------------------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/clickbench_hits_10.parquet]]}, projection=[EventTime, EventDate, CounterID, IsRefresh, DontCountHits], file_type=parquet, predicate=CounterID@6 = 62 AND EventDate@5 >= 15900 AND EventDate@5 <= 15901 AND IsRefresh@15 = 0 AND DontCountHits@61 = 0, pruning_predicate=CounterID_null_count@2 != row_count@3 AND CounterID_min@0 <= 62 AND 62 <= CounterID_max@1 AND EventDate_null_count@5 != row_count@3 AND EventDate_max@4 >= 15900 AND EventDate_null_count@5 != row_count@3 AND EventDate_min@6 <= 15901 AND IsRefresh_null_count@9 != row_count@3 AND IsRefresh_min@7 <= 0 AND 0 <= IsRefresh_max@8 AND DontCountHits_null_count@12 != row_count@3 AND DontCountHits_min@10 <= 0 AND 0 <= DontCountHits_max@11, required_guarantees=[CounterID in (62), DontCountHits in (0), IsRefresh in (0)] diff --git a/datafusion/sqllogictest/test_files/count_star_rule.slt b/datafusion/sqllogictest/test_files/count_star_rule.slt index 9814fd1347589..31d11e114e893 100644 --- a/datafusion/sqllogictest/test_files/count_star_rule.slt +++ b/datafusion/sqllogictest/test_files/count_star_rule.slt @@ -32,7 +32,7 @@ EXPLAIN SELECT COUNT() FROM (SELECT 1 AS a, 2 AS b) AS t; ---- logical_plan 01)Projection: count(Int64(1)) AS count() -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 03)----SubqueryAlias: t 04)------EmptyRelation: rows=1 physical_plan @@ -44,13 +44,13 @@ EXPLAIN SELECT t1.a, COUNT() FROM t1 GROUP BY t1.a; ---- logical_plan 01)Projection: t1.a, count(Int64(1)) AS count() -02)--Aggregate: groupBy=[[t1.a]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[t1.a]], aggr=[[count(Int64(1))]] 03)----TableScan: t1 projection=[a] physical_plan 01)ProjectionExec: expr=[a@0 as a, count(Int64(1))@1 as count()] -02)--AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[count() as count(Int64(1))] +02)--AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[count(Int64(1))] 03)----RepartitionExec: partitioning=Hash([a@0], 4), input_partitions=1 -04)------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[count(Int64(1))] 05)--------DataSourceExec: partitions=1, partition_sizes=[1] query TT @@ -59,14 +59,14 @@ EXPLAIN SELECT t1.a, COUNT() AS cnt FROM t1 GROUP BY t1.a HAVING COUNT() > 0; logical_plan 01)Projection: t1.a, count(Int64(1)) AS count() AS cnt 02)--Filter: count(Int64(1)) > Int64(0) -03)----Aggregate: groupBy=[[t1.a]], aggr=[[count() AS count(Int64(1))]] +03)----Aggregate: groupBy=[[t1.a]], aggr=[[count(Int64(1))]] 04)------TableScan: t1 projection=[a] physical_plan 01)ProjectionExec: expr=[a@0 as a, count(Int64(1))@1 as cnt] 02)--FilterExec: count(Int64(1))@1 > 0 -03)----AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[count() as count(Int64(1))] +03)----AggregateExec: mode=FinalPartitioned, gby=[a@0 as a], aggr=[count(Int64(1))] 04)------RepartitionExec: partitioning=Hash([a@0], 4), input_partitions=1 -05)--------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[count() as count(Int64(1))] +05)--------AggregateExec: mode=Partial, gby=[a@0 as a], aggr=[count(Int64(1))] 06)----------DataSourceExec: partitions=1, partition_sizes=[1] query II diff --git a/datafusion/sqllogictest/test_files/dynamic_filter_pushdown_config.slt b/datafusion/sqllogictest/test_files/dynamic_filter_pushdown_config.slt index f224fbf9066bd..6a6bad99f0840 100644 --- a/datafusion/sqllogictest/test_files/dynamic_filter_pushdown_config.slt +++ b/datafusion/sqllogictest/test_files/dynamic_filter_pushdown_config.slt @@ -631,15 +631,15 @@ EXPLAIN SELECT COUNT(*), MAX(score) FROM agg_parquet WHERE category = 'alpha'; ---- logical_plan 01)Projection: count(Int64(1)) AS count(*), max(agg_parquet.score) -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1)), max(agg_parquet.score)]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1)), max(agg_parquet.score)]] 03)----Projection: agg_parquet.score 04)------Filter: agg_parquet.category = Utf8View("alpha") 05)--------TableScan: agg_parquet projection=[category, score], partial_filters=[agg_parquet.category = Utf8View("alpha")] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*), max(agg_parquet.score)@1 as max(agg_parquet.score)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1)), max(agg_parquet.score)] +02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1)), max(agg_parquet.score)] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1)), max(agg_parquet.score)] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1)), max(agg_parquet.score)] 05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/dynamic_filter_pushdown_config/agg_data.parquet]]}, projection=[score], file_type=parquet, predicate=category@0 = alpha, pruning_predicate=category_null_count@2 != row_count@3 AND category_min@0 <= alpha AND alpha <= category_max@1, required_guarantees=[category in (alpha)] diff --git a/datafusion/sqllogictest/test_files/explain_analyze.slt b/datafusion/sqllogictest/test_files/explain_analyze.slt index d54630b7782d7..511ba88ed3345 100644 --- a/datafusion/sqllogictest/test_files/explain_analyze.slt +++ b/datafusion/sqllogictest/test_files/explain_analyze.slt @@ -500,7 +500,7 @@ GROUP BY k; ---- Plan with Metrics 01)ProjectionExec: expr=[k@0 as k, count(Int64(1))@1 as count(*)], metrics=[output_bytes=1056.0 B] -02)--AggregateExec: mode=Single, gby=[k@0 as k], aggr=[count() as count(Int64(1))], metrics=[output_bytes=1056.0 B, spilled_bytes=0.0 B, peak_mem_used=9.2 KB] +02)--AggregateExec: mode=Single, gby=[k@0 as k], aggr=[count(Int64(1))], metrics=[output_bytes=1056.0 B, spilled_bytes=0.0 B, peak_mem_used=9.2 KB] 03)----ProjectionExec: expr=[column1@0 as k], metrics=[output_bytes=32.0 B] 04)------DataSourceExec: partitions=1, partition_sizes=[1], metrics=[] diff --git a/datafusion/sqllogictest/test_files/explain_tree.slt b/datafusion/sqllogictest/test_files/explain_tree.slt index ba2e1d2078993..fcac86aa21a2e 100644 --- a/datafusion/sqllogictest/test_files/explain_tree.slt +++ b/datafusion/sqllogictest/test_files/explain_tree.slt @@ -1239,47 +1239,45 @@ physical_plan 07)┌─────────────┴─────────────┐ 08)│ AggregateExec │ 09)│ -------------------- │ -10)│ aggr: │ -11)│ count() as count(Int64(1))│ +10)│ aggr: count(1) │ +11)│ group_by: name │ 12)│ │ -13)│ group_by: name │ -14)│ │ -15)│ mode: │ -16)│ SinglePartitioned │ -17)└─────────────┬─────────────┘ -18)┌─────────────┴─────────────┐ -19)│ InterleaveExec ├──────────────┐ -20)└─────────────┬─────────────┘ │ -21)┌─────────────┴─────────────┐┌─────────────┴─────────────┐ -22)│ AggregateExec ││ AggregateExec │ -23)│ -------------------- ││ -------------------- │ -24)│ group_by: name ││ group_by: name │ -25)│ ││ │ -26)│ mode: ││ mode: │ -27)│ FinalPartitioned ││ FinalPartitioned │ -28)└─────────────┬─────────────┘└─────────────┬─────────────┘ -29)┌─────────────┴─────────────┐┌─────────────┴─────────────┐ -30)│ RepartitionExec ││ RepartitionExec │ -31)│ -------------------- ││ -------------------- │ -32)│ partition_count(in->out): ││ partition_count(in->out): │ -33)│ 1 -> 4 ││ 1 -> 4 │ -34)│ ││ │ -35)│ partitioning_scheme: ││ partitioning_scheme: │ -36)│ Hash([name@0], 4) ││ Hash([name@0], 4) │ -37)└─────────────┬─────────────┘└─────────────┬─────────────┘ -38)┌─────────────┴─────────────┐┌─────────────┴─────────────┐ -39)│ AggregateExec ││ AggregateExec │ -40)│ -------------------- ││ -------------------- │ -41)│ group_by: name ││ group_by: name │ -42)│ mode: Partial ││ mode: Partial │ -43)└─────────────┬─────────────┘└─────────────┬─────────────┘ -44)┌─────────────┴─────────────┐┌─────────────┴─────────────┐ -45)│ DataSourceExec ││ DataSourceExec │ -46)│ -------------------- ││ -------------------- │ -47)│ bytes: 288 ││ bytes: 280 │ -48)│ format: memory ││ format: memory │ -49)│ rows: 3 ││ rows: 3 │ -50)└───────────────────────────┘└───────────────────────────┘ +13)│ mode: │ +14)│ SinglePartitioned │ +15)└─────────────┬─────────────┘ +16)┌─────────────┴─────────────┐ +17)│ InterleaveExec ├──────────────┐ +18)└─────────────┬─────────────┘ │ +19)┌─────────────┴─────────────┐┌─────────────┴─────────────┐ +20)│ AggregateExec ││ AggregateExec │ +21)│ -------------------- ││ -------------------- │ +22)│ group_by: name ││ group_by: name │ +23)│ ││ │ +24)│ mode: ││ mode: │ +25)│ FinalPartitioned ││ FinalPartitioned │ +26)└─────────────┬─────────────┘└─────────────┬─────────────┘ +27)┌─────────────┴─────────────┐┌─────────────┴─────────────┐ +28)│ RepartitionExec ││ RepartitionExec │ +29)│ -------------------- ││ -------------------- │ +30)│ partition_count(in->out): ││ partition_count(in->out): │ +31)│ 1 -> 4 ││ 1 -> 4 │ +32)│ ││ │ +33)│ partitioning_scheme: ││ partitioning_scheme: │ +34)│ Hash([name@0], 4) ││ Hash([name@0], 4) │ +35)└─────────────┬─────────────┘└─────────────┬─────────────┘ +36)┌─────────────┴─────────────┐┌─────────────┴─────────────┐ +37)│ AggregateExec ││ AggregateExec │ +38)│ -------------------- ││ -------------------- │ +39)│ group_by: name ││ group_by: name │ +40)│ mode: Partial ││ mode: Partial │ +41)└─────────────┬─────────────┘└─────────────┬─────────────┘ +42)┌─────────────┴─────────────┐┌─────────────┴─────────────┐ +43)│ DataSourceExec ││ DataSourceExec │ +44)│ -------------------- ││ -------------------- │ +45)│ bytes: 288 ││ bytes: 280 │ +46)│ format: memory ││ format: memory │ +47)│ rows: 3 ││ rows: 3 │ +48)└───────────────────────────┘└───────────────────────────┘ # Test explain tree for UnionExec query TT @@ -1715,53 +1713,49 @@ physical_plan 07)┌─────────────┴─────────────┐ 08)│ AggregateExec │ 09)│ -------------------- │ -10)│ aggr: │ -11)│ count() as count(Int64(1))│ -12)│ │ -13)│ mode: Final │ -14)└─────────────┬─────────────┘ -15)┌─────────────┴─────────────┐ -16)│ CoalescePartitionsExec │ -17)└─────────────┬─────────────┘ -18)┌─────────────┴─────────────┐ -19)│ AggregateExec │ -20)│ -------------------- │ -21)│ aggr: │ -22)│ count() as count(Int64(1))│ -23)│ │ -24)│ mode: Partial │ -25)└─────────────┬─────────────┘ -26)┌─────────────┴─────────────┐ -27)│ RepartitionExec │ -28)│ -------------------- │ -29)│ partition_count(in->out): │ -30)│ 1 -> 4 │ -31)│ │ -32)│ partitioning_scheme: │ -33)│ RoundRobinBatch(4) │ -34)└─────────────┬─────────────┘ -35)┌─────────────┴─────────────┐ -36)│ ProjectionExec │ -37)└─────────────┬─────────────┘ -38)┌─────────────┴─────────────┐ -39)│ GlobalLimitExec │ -40)│ -------------------- │ -41)│ limit: 3 │ -42)│ skip: 6 │ -43)└─────────────┬─────────────┘ -44)┌─────────────┴─────────────┐ -45)│ FilterExec │ -46)│ -------------------- │ -47)│ fetch: 9 │ -48)│ predicate: a > 3 │ -49)└─────────────┬─────────────┘ -50)┌─────────────┴─────────────┐ -51)│ DataSourceExec │ -52)│ -------------------- │ -53)│ bytes: 160 │ -54)│ format: memory │ -55)│ rows: 10 │ -56)└───────────────────────────┘ +10)│ aggr: count(1) │ +11)│ mode: Final │ +12)└─────────────┬─────────────┘ +13)┌─────────────┴─────────────┐ +14)│ CoalescePartitionsExec │ +15)└─────────────┬─────────────┘ +16)┌─────────────┴─────────────┐ +17)│ AggregateExec │ +18)│ -------------------- │ +19)│ aggr: count(1) │ +20)│ mode: Partial │ +21)└─────────────┬─────────────┘ +22)┌─────────────┴─────────────┐ +23)│ RepartitionExec │ +24)│ -------------------- │ +25)│ partition_count(in->out): │ +26)│ 1 -> 4 │ +27)│ │ +28)│ partitioning_scheme: │ +29)│ RoundRobinBatch(4) │ +30)└─────────────┬─────────────┘ +31)┌─────────────┴─────────────┐ +32)│ ProjectionExec │ +33)└─────────────┬─────────────┘ +34)┌─────────────┴─────────────┐ +35)│ GlobalLimitExec │ +36)│ -------------------- │ +37)│ limit: 3 │ +38)│ skip: 6 │ +39)└─────────────┬─────────────┘ +40)┌─────────────┴─────────────┐ +41)│ FilterExec │ +42)│ -------------------- │ +43)│ fetch: 9 │ +44)│ predicate: a > 3 │ +45)└─────────────┬─────────────┘ +46)┌─────────────┴─────────────┐ +47)│ DataSourceExec │ +48)│ -------------------- │ +49)│ bytes: 160 │ +50)│ format: memory │ +51)│ rows: 10 │ +52)└───────────────────────────┘ # clean up statement ok diff --git a/datafusion/sqllogictest/test_files/functional_dependencies.slt b/datafusion/sqllogictest/test_files/functional_dependencies.slt index db74d9404511c..c49004190dc60 100644 --- a/datafusion/sqllogictest/test_files/functional_dependencies.slt +++ b/datafusion/sqllogictest/test_files/functional_dependencies.slt @@ -160,7 +160,7 @@ EXPLAIN SELECT x, cnt FROM (SELECT x, count(*) AS cnt FROM t_uniq GROUP BY x) OR logical_plan 01)Sort: t_uniq.x ASC NULLS LAST 02)--Projection: t_uniq.x, count(Int64(1)) AS cnt -03)----Aggregate: groupBy=[[t_uniq.x]], aggr=[[count() AS count(Int64(1))]] +03)----Aggregate: groupBy=[[t_uniq.x]], aggr=[[count(Int64(1))]] 04)------TableScan: t_uniq projection=[x] @@ -268,14 +268,14 @@ EXPLAIN SELECT g.x, count(*) AS c ---- logical_plan 01)Projection: g.x, count(Int64(1)) AS count(*) AS c -02)--Aggregate: groupBy=[[g.x, g.cnt]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[g.x, g.cnt]], aggr=[[count(Int64(1))]] 03)----Projection: g.x, g.cnt 04)------Left Join: CAST(a.z AS Int64) = g.cnt 05)--------SubqueryAlias: a 06)----------TableScan: t_probe projection=[z] 07)--------SubqueryAlias: g 08)----------Projection: t_null.x, count(Int64(1)) AS count(*) AS cnt -09)------------Aggregate: groupBy=[[t_null.x]], aggr=[[count() AS count(Int64(1))]] +09)------------Aggregate: groupBy=[[t_null.x]], aggr=[[count(Int64(1))]] 10)--------------TableScan: t_null projection=[x] # 5.2 The ORDER BY variant: `g.x` is NULL for both rows, so the `g.cnt` diff --git a/datafusion/sqllogictest/test_files/joins.slt b/datafusion/sqllogictest/test_files/joins.slt index 36f19eaada0d7..b594c68ce17d6 100644 --- a/datafusion/sqllogictest/test_files/joins.slt +++ b/datafusion/sqllogictest/test_files/joins.slt @@ -1423,16 +1423,16 @@ group by t1_id ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[join_t1.t1_id]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[join_t1.t1_id]], aggr=[[count(Int64(1))]] 03)----Projection: join_t1.t1_id 04)------Inner Join: join_t1.t1_id = join_t2.t2_id 05)--------TableScan: join_t1 projection=[t1_id] 06)--------TableScan: join_t2 projection=[t2_id] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@1 as count(*)] -02)--AggregateExec: mode=FinalPartitioned, gby=[t1_id@0 as t1_id], aggr=[count() as count(Int64(1))] +02)--AggregateExec: mode=FinalPartitioned, gby=[t1_id@0 as t1_id], aggr=[count(Int64(1))] 03)----RepartitionExec: partitioning=Hash([t1_id@0], 2), input_partitions=2 -04)------AggregateExec: mode=Partial, gby=[t1_id@0 as t1_id], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[t1_id@0 as t1_id], aggr=[count(Int64(1))] 05)--------HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(t1_id@0, t2_id@0)], projection=[t1_id@0] 06)----------DataSourceExec: partitions=1, partition_sizes=[1] 07)----------RepartitionExec: partitioning=RoundRobinBatch(2), input_partitions=1 @@ -4524,7 +4524,7 @@ JOIN my_catalog.my_schema.table_with_many_types AS r ON l.binary_col = r.binary_ ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 03)----Projection: 04)------Inner Join: l.binary_col = r.binary_col 05)--------SubqueryAlias: l @@ -4533,7 +4533,7 @@ logical_plan 08)----------TableScan: my_catalog.my_schema.table_with_many_types projection=[binary_col] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Single, gby=[], aggr=[count() as count(Int64(1))] +02)--AggregateExec: mode=Single, gby=[], aggr=[count(Int64(1))] 03)----HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(binary_col@0, binary_col@0)], projection=[] 04)------DataSourceExec: partitions=1, partition_sizes=[1] 05)------DataSourceExec: partitions=1, partition_sizes=[1] @@ -5766,7 +5766,7 @@ EXPLAIN SELECT count(*) FROM elim_orders LEFT JOIN elim_users ON user_id = id; ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 03)----TableScan: elim_orders projection=[] physical_plan 01)ProjectionExec: expr=[3 as count(*)] @@ -5862,16 +5862,16 @@ EXPLAIN SELECT count(*) FROM elim_users LEFT JOIN elim_orders ON elim_users.id = ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 03)----Projection: 04)------Left Join: elim_users.id = elim_orders.user_id 05)--------TableScan: elim_users projection=[id] 06)--------TableScan: elim_orders projection=[user_id] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] 05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 06)----------HashJoinExec: mode=CollectLeft, join_type=Left, on=[(id@0, user_id@0)], projection=[] 07)------------DataSourceExec: partitions=1, partition_sizes=[1] @@ -6083,7 +6083,7 @@ EXPLAIN SELECT count(*) FROM elim_users RIGHT JOIN elim_orders ON id = user_id; ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 03)----TableScan: elim_orders projection=[] physical_plan 01)ProjectionExec: expr=[3 as count(*)] diff --git a/datafusion/sqllogictest/test_files/json.slt b/datafusion/sqllogictest/test_files/json.slt index e9ff1bfafc2ec..e3b3d6b4e11ed 100644 --- a/datafusion/sqllogictest/test_files/json.slt +++ b/datafusion/sqllogictest/test_files/json.slt @@ -55,13 +55,13 @@ EXPLAIN SELECT count(*) from json_test ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 03)----TableScan: json_test projection=[] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] 05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 06)----------DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/core/tests/data/2.json]]}, file_type=json diff --git a/datafusion/sqllogictest/test_files/lateral_join.slt b/datafusion/sqllogictest/test_files/lateral_join.slt index 5779c14f08e42..cae3e67153246 100644 --- a/datafusion/sqllogictest/test_files/lateral_join.slt +++ b/datafusion/sqllogictest/test_files/lateral_join.slt @@ -764,7 +764,7 @@ logical_plan 04)------TableScan: t1 projection=[id] 05)------SubqueryAlias: sub 06)--------Projection: count(Int64(1)) AS cnt, t2.t1_id, Boolean(true) AS __always_true -07)----------Aggregate: groupBy=[[t2.t1_id]], aggr=[[count() AS count(Int64(1))]] +07)----------Aggregate: groupBy=[[t2.t1_id]], aggr=[[count(Int64(1))]] 08)------------TableScan: t2 projection=[t1_id] physical_plan 01)SortPreservingMergeExec: [id@0 ASC NULLS LAST] @@ -773,9 +773,9 @@ physical_plan 04)------HashJoinExec: mode=CollectLeft, join_type=Left, on=[(id@0, t1_id@1)], projection=[id@0, __always_true@3, cnt@1] 05)--------DataSourceExec: partitions=1, partition_sizes=[1] 06)--------ProjectionExec: expr=[count(Int64(1))@1 as cnt, t1_id@0 as t1_id, true as __always_true] -07)----------AggregateExec: mode=FinalPartitioned, gby=[t1_id@0 as t1_id], aggr=[count() as count(Int64(1))] +07)----------AggregateExec: mode=FinalPartitioned, gby=[t1_id@0 as t1_id], aggr=[count(Int64(1))] 08)------------RepartitionExec: partitioning=Hash([t1_id@0], 4), input_partitions=1 -09)--------------AggregateExec: mode=Partial, gby=[t1_id@0 as t1_id], aggr=[count() as count(Int64(1))] +09)--------------AggregateExec: mode=Partial, gby=[t1_id@0 as t1_id], aggr=[count(Int64(1))] 10)----------------DataSourceExec: partitions=1, partition_sizes=[1] # Verify LEFT lateral without aggregate decorrelates to left join diff --git a/datafusion/sqllogictest/test_files/limit.slt b/datafusion/sqllogictest/test_files/limit.slt index c8ea8ca15ad78..1ff6ca4fb0253 100644 --- a/datafusion/sqllogictest/test_files/limit.slt +++ b/datafusion/sqllogictest/test_files/limit.slt @@ -308,7 +308,7 @@ EXPLAIN SELECT COUNT(*) FROM (SELECT a FROM t1 LIMIT 3 OFFSET 11); ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 03)----Limit: skip=11, fetch=3 04)------TableScan: t1 projection=[], fetch=14 physical_plan @@ -327,7 +327,7 @@ EXPLAIN SELECT COUNT(*) FROM (SELECT a FROM t1 LIMIT 3 OFFSET 8); ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 03)----Limit: skip=8, fetch=3 04)------TableScan: t1 projection=[], fetch=11 physical_plan @@ -369,7 +369,7 @@ EXPLAIN SELECT COUNT(*) FROM (SELECT a FROM t1 OFFSET 8); ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 03)----Limit: skip=8, fetch=None 04)------TableScan: t1 projection=[] physical_plan @@ -387,16 +387,16 @@ EXPLAIN SELECT COUNT(*) FROM (SELECT a FROM t1 WHERE a > 3 LIMIT 3 OFFSET 6); ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 03)----Projection: 04)------Limit: skip=6, fetch=3 05)--------Filter: t1.a > Int32(3) 06)----------TableScan: t1 projection=[a] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] 05)--------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 06)----------ProjectionExec: expr=[] 07)------------GlobalLimitExec: skip=6, fetch=3 diff --git a/datafusion/sqllogictest/test_files/nested_loop_join_spill.slt b/datafusion/sqllogictest/test_files/nested_loop_join_spill.slt index 4e12314e3cb00..6eaf2b38a5bcf 100644 --- a/datafusion/sqllogictest/test_files/nested_loop_join_spill.slt +++ b/datafusion/sqllogictest/test_files/nested_loop_join_spill.slt @@ -50,7 +50,7 @@ INNER JOIN generate_series(1, 1) AS t2(v2) ---- Plan with Metrics 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)], metrics=[] -02)--AggregateExec: mode=Single, gby=[], aggr=[count() as count(Int64(1))], metrics=[] +02)--AggregateExec: mode=Single, gby=[], aggr=[count(Int64(1))], metrics=[] 03)----NestedLoopJoinExec: join_type=Inner, filter=v1@0 + v2@1 > 0, projection=[], metrics=[output_rows=100.0 K, spill_count=2, ] 04)------ProjectionExec: expr=[value@0 as v1], metrics=[] 05)--------LazyMemoryExec: partitions=1, batch_generators=[generate_series: start=1, end=100000, batch_size=8192], metrics=[] @@ -104,7 +104,7 @@ RIGHT JOIN generate_series(1, 200) AS t2(v2) ---- Plan with Metrics 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)], metrics=[] -02)--AggregateExec: mode=Single, gby=[], aggr=[count() as count(Int64(1))], metrics=[] +02)--AggregateExec: mode=Single, gby=[], aggr=[count(Int64(1))], metrics=[] 03)----ProjectionExec: expr=[], metrics=[] 04)------NestedLoopJoinExec: join_type=Right, filter=v1@0 + v2@1 = 2 AND join_proj_push_down_1@2, projection=[v1@0, v2@1], metrics=[output_rows=200, spill_count=2, ] 05)--------ProjectionExec: expr=[value@0 as v1], metrics=[] @@ -175,9 +175,9 @@ LEFT JOIN generate_series(1, 100) AS t2(v2) ---- Plan with Metrics 01)ProjectionExec: expr=[count(Int64(1))@0 as cnt], metrics=[] -02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))], metrics=[] +02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))], metrics=[] 03)----CoalescePartitionsExec, metrics=[] -04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))], metrics=[] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))], metrics=[] 05)--------NestedLoopJoinExec: join_type=Left, filter=v1@0 + v2@1 = 101, projection=[], metrics=[output_rows=5.00 K, spill_count=2, ] 06)----------ProjectionExec: expr=[value@0 as v1], metrics=[] 07)------------LazyMemoryExec: partitions=1, batch_generators=[generate_series: start=1, end=5000, batch_size=8192], metrics=[] diff --git a/datafusion/sqllogictest/test_files/optimizer_group_by_constant.slt b/datafusion/sqllogictest/test_files/optimizer_group_by_constant.slt index fd2020e8f7383..8c0556547496a 100644 --- a/datafusion/sqllogictest/test_files/optimizer_group_by_constant.slt +++ b/datafusion/sqllogictest/test_files/optimizer_group_by_constant.slt @@ -49,7 +49,7 @@ GROUP BY 1, 2, 3, 4 ---- logical_plan 01)Projection: t.c1, Int64(99999), t.c5 + t.c8, Utf8("test"), count(Int64(1)) -02)--Aggregate: groupBy=[[t.c1, t.c5 + t.c8]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[t.c1, t.c5 + t.c8]], aggr=[[count(Int64(1))]] 03)----SubqueryAlias: t 04)------TableScan: test_table projection=[c1, c5, c8] @@ -60,7 +60,7 @@ FROM test_table t group by 1, 2, 3 ---- logical_plan -01)Aggregate: groupBy=[[Int64(123), Int64(456), Int64(789)]], aggr=[[count() AS count(Int64(1)), avg(t.c12)]] +01)Aggregate: groupBy=[[Int64(123), Int64(456), Int64(789)]], aggr=[[count(Int64(1)), avg(t.c12)]] 02)--SubqueryAlias: t 03)----TableScan: test_table projection=[c12] @@ -72,7 +72,7 @@ GROUP BY 1, 2 ---- logical_plan 01)Projection: to_date(Utf8("2023-05-04")) AS dt, date_part(Utf8("DAY"),now()) < Int64(1000) AS today_filter, count(Int64(1)) -02)--Aggregate: groupBy=[[Date32("2023-05-04") AS to_date(Utf8("2023-05-04")), Boolean(true) AS date_part(Utf8("DAY"),now()) < Int64(1000)]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[Date32("2023-05-04") AS to_date(Utf8("2023-05-04")), Boolean(true) AS date_part(Utf8("DAY"),now()) < Int64(1000)]], aggr=[[count(Int64(1))]] 03)----SubqueryAlias: t 04)------TableScan: test_table projection=[] @@ -89,7 +89,7 @@ FROM test_table t GROUP BY 1 ---- logical_plan -01)Aggregate: groupBy=[[Boolean(true) AS NOT date_part(Utf8("MONTH"),now()) BETWEEN Int64(50) AND Int64(60)]], aggr=[[count() AS count(Int64(1))]] +01)Aggregate: groupBy=[[Boolean(true) AS NOT date_part(Utf8("MONTH"),now()) BETWEEN Int64(50) AND Int64(60)]], aggr=[[count(Int64(1))]] 02)--SubqueryAlias: t 03)----TableScan: test_table projection=[] diff --git a/datafusion/sqllogictest/test_files/piecewise_merge_join_batches.slt b/datafusion/sqllogictest/test_files/piecewise_merge_join_batches.slt index 95d642c08c04b..d8b3c16163aff 100644 --- a/datafusion/sqllogictest/test_files/piecewise_merge_join_batches.slt +++ b/datafusion/sqllogictest/test_files/piecewise_merge_join_batches.slt @@ -38,7 +38,7 @@ EXPLAIN SELECT count(*) FROM pb_l l WHERE EXISTS (SELECT 1 FROM range(3, 8) r WH ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 03)----Projection: 04)------LeftSemi Join: Filter: CAST(l.v AS Int64) > __correlated_sq_1.value 05)--------SubqueryAlias: l @@ -48,9 +48,9 @@ logical_plan 09)------------TableScan: range() projection=[value] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] 05)--------ProjectionExec: expr=[] 06)----------PiecewiseMergeJoin: operator=Gt, join_type=LeftSemi, on=(CAST(v AS Int64) > value) 07)------------SortPreservingMergeExec: [CAST(v@0 AS Int64) ASC] @@ -71,7 +71,7 @@ EXPLAIN SELECT count(*) FROM pb_l l JOIN range(3, 8) r ON l.v < r.value; ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 03)----Projection: 04)------Inner Join: Filter: CAST(l.v AS Int64) < r.value 05)--------SubqueryAlias: l @@ -80,9 +80,9 @@ logical_plan 08)----------TableScan: range() projection=[value] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as count(*)] -02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] 05)--------ProjectionExec: expr=[] 06)----------PiecewiseMergeJoin: operator=Lt, join_type=Inner, on=(CAST(v AS Int64) < value) 07)------------SortPreservingMergeExec: [CAST(v@0 AS Int64) DESC] diff --git a/datafusion/sqllogictest/test_files/preserve_file_partitioning.slt b/datafusion/sqllogictest/test_files/preserve_file_partitioning.slt index 38654f36adc46..e2dd22cc82bba 100644 --- a/datafusion/sqllogictest/test_files/preserve_file_partitioning.slt +++ b/datafusion/sqllogictest/test_files/preserve_file_partitioning.slt @@ -223,13 +223,13 @@ EXPLAIN SELECT f_dkey, count(*), sum(value) FROM fact_table GROUP BY f_dkey; ---- logical_plan 01)Projection: fact_table.f_dkey, count(Int64(1)) AS count(*), sum(fact_table.value) -02)--Aggregate: groupBy=[[fact_table.f_dkey]], aggr=[[count() AS count(Int64(1)), sum(fact_table.value)]] +02)--Aggregate: groupBy=[[fact_table.f_dkey]], aggr=[[count(Int64(1)), sum(fact_table.value)]] 03)----TableScan: fact_table projection=[value, f_dkey] physical_plan 01)ProjectionExec: expr=[f_dkey@0 as f_dkey, count(Int64(1))@1 as count(*), sum(fact_table.value)@2 as sum(fact_table.value)] -02)--AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[count() as count(Int64(1)), sum(fact_table.value)] +02)--AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[count(Int64(1)), sum(fact_table.value)] 03)----RepartitionExec: partitioning=Hash([f_dkey@0], 3), input_partitions=3 -04)------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[count() as count(Int64(1)), sum(fact_table.value)] +04)------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[count(Int64(1)), sum(fact_table.value)] 05)--------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=C/data.parquet]]}, projection=[value, f_dkey], file_type=parquet # Verify results without optimization @@ -253,11 +253,11 @@ EXPLAIN SELECT f_dkey, count(*), sum(value) FROM fact_table GROUP BY f_dkey; ---- logical_plan 01)Projection: fact_table.f_dkey, count(Int64(1)) AS count(*), sum(fact_table.value) -02)--Aggregate: groupBy=[[fact_table.f_dkey]], aggr=[[count() AS count(Int64(1)), sum(fact_table.value)]] +02)--Aggregate: groupBy=[[fact_table.f_dkey]], aggr=[[count(Int64(1)), sum(fact_table.value)]] 03)----TableScan: fact_table projection=[value, f_dkey] physical_plan 01)ProjectionExec: expr=[f_dkey@0 as f_dkey, count(Int64(1))@1 as count(*), sum(fact_table.value)@2 as sum(fact_table.value)] -02)--AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[count() as count(Int64(1)), sum(fact_table.value)] +02)--AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[count(Int64(1)), sum(fact_table.value)] 03)----DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=C/data.parquet]]}, projection=[value, f_dkey], output_partitioning=Hash([f_dkey@1], 3), file_type=parquet # Verify results with optimization match results without optimization @@ -282,14 +282,14 @@ EXPLAIN SELECT f_dkey, count(*), avg(value) FROM fact_table_ordered GROUP BY f_d logical_plan 01)Sort: fact_table_ordered.f_dkey ASC NULLS LAST 02)--Projection: fact_table_ordered.f_dkey, count(Int64(1)) AS count(*), avg(fact_table_ordered.value) -03)----Aggregate: groupBy=[[fact_table_ordered.f_dkey]], aggr=[[count() AS count(Int64(1)), avg(fact_table_ordered.value)]] +03)----Aggregate: groupBy=[[fact_table_ordered.f_dkey]], aggr=[[count(Int64(1)), avg(fact_table_ordered.value)]] 04)------TableScan: fact_table_ordered projection=[value, f_dkey] physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, count(Int64(1))@1 as count(*), avg(fact_table_ordered.value)@2 as avg(fact_table_ordered.value)] -03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[count() as count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted +03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted 04)------RepartitionExec: partitioning=Hash([f_dkey@0], 3), input_partitions=3, preserve_order=true, sort_exprs=f_dkey@0 ASC NULLS LAST -05)--------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[count() as count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted +05)--------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted 06)----------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=C/data.parquet]]}, projection=[value, f_dkey], output_ordering=[f_dkey@1 ASC NULLS LAST], file_type=parquet # Verify results without optimization @@ -314,12 +314,12 @@ EXPLAIN SELECT f_dkey, count(*), avg(value) FROM fact_table_ordered GROUP BY f_d logical_plan 01)Sort: fact_table_ordered.f_dkey ASC NULLS LAST 02)--Projection: fact_table_ordered.f_dkey, count(Int64(1)) AS count(*), avg(fact_table_ordered.value) -03)----Aggregate: groupBy=[[fact_table_ordered.f_dkey]], aggr=[[count() AS count(Int64(1)), avg(fact_table_ordered.value)]] +03)----Aggregate: groupBy=[[fact_table_ordered.f_dkey]], aggr=[[count(Int64(1)), avg(fact_table_ordered.value)]] 04)------TableScan: fact_table_ordered projection=[value, f_dkey] physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, count(Int64(1))@1 as count(*), avg(fact_table_ordered.value)@2 as avg(fact_table_ordered.value)] -03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[count() as count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted +03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted 04)------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=C/data.parquet]]}, projection=[value, f_dkey], output_ordering=[f_dkey@1 ASC NULLS LAST], output_partitioning=Hash([f_dkey@1], 3), file_type=parquet query TIR @@ -348,7 +348,7 @@ ORDER BY f.f_dkey; logical_plan 01)Sort: f.f_dkey ASC NULLS LAST 02)--Projection: f.f_dkey, max(d.env), max(d.service), count(Int64(1)) AS count(*), sum(f.value) -03)----Aggregate: groupBy=[[f.f_dkey]], aggr=[[max(d.env), max(d.service), count() AS count(Int64(1)), sum(f.value)]] +03)----Aggregate: groupBy=[[f.f_dkey]], aggr=[[max(d.env), max(d.service), count(Int64(1)), sum(f.value)]] 04)------Projection: f.value, f.f_dkey, d.env, d.service 05)--------Inner Join: f.f_dkey = d.d_dkey 06)----------SubqueryAlias: f @@ -359,9 +359,9 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, max(d.env)@1 as max(d.env), max(d.service)@2 as max(d.service), count(Int64(1))@3 as count(*), sum(f.value)@4 as sum(f.value)] -03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[max(d.env), max(d.service), count() as count(Int64(1)), sum(f.value)], ordering_mode=Sorted +03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[max(d.env), max(d.service), count(Int64(1)), sum(f.value)], ordering_mode=Sorted 04)------RepartitionExec: partitioning=Hash([f_dkey@0], 3), input_partitions=3, preserve_order=true, sort_exprs=f_dkey@0 ASC NULLS LAST -05)--------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[max(d.env), max(d.service), count() as count(Int64(1)), sum(f.value)], ordering_mode=Sorted +05)--------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[max(d.env), max(d.service), count(Int64(1)), sum(f.value)], ordering_mode=Sorted 06)----------HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(d_dkey@0, f_dkey@1)], projection=[value@3, f_dkey@4, env@1, service@2] 07)------------CoalescePartitionsExec 08)--------------FilterExec: service@2 = log @@ -401,7 +401,7 @@ ORDER BY f.f_dkey; logical_plan 01)Sort: f.f_dkey ASC NULLS LAST 02)--Projection: f.f_dkey, max(d.env), max(d.service), count(Int64(1)) AS count(*), sum(f.value) -03)----Aggregate: groupBy=[[f.f_dkey]], aggr=[[max(d.env), max(d.service), count() AS count(Int64(1)), sum(f.value)]] +03)----Aggregate: groupBy=[[f.f_dkey]], aggr=[[max(d.env), max(d.service), count(Int64(1)), sum(f.value)]] 04)------Projection: f.value, f.f_dkey, d.env, d.service 05)--------Inner Join: f.f_dkey = d.d_dkey 06)----------SubqueryAlias: f @@ -412,7 +412,7 @@ logical_plan physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, max(d.env)@1 as max(d.env), max(d.service)@2 as max(d.service), count(Int64(1))@3 as count(*), sum(f.value)@4 as sum(f.value)] -03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[max(d.env), max(d.service), count() as count(Int64(1)), sum(f.value)], ordering_mode=Sorted +03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[max(d.env), max(d.service), count(Int64(1)), sum(f.value)], ordering_mode=Sorted 04)------HashJoinExec: mode=CollectLeft, join_type=Inner, on=[(d_dkey@0, f_dkey@1)], projection=[value@3, f_dkey@4, env@1, service@2] 05)--------CoalescePartitionsExec 06)----------FilterExec: service@2 = log @@ -543,11 +543,11 @@ EXPLAIN SELECT f_dkey, count(*), sum(value) FROM high_cardinality_table GROUP BY ---- logical_plan 01)Projection: high_cardinality_table.f_dkey, count(Int64(1)) AS count(*), sum(high_cardinality_table.value) -02)--Aggregate: groupBy=[[high_cardinality_table.f_dkey]], aggr=[[count() AS count(Int64(1)), sum(high_cardinality_table.value)]] +02)--Aggregate: groupBy=[[high_cardinality_table.f_dkey]], aggr=[[count(Int64(1)), sum(high_cardinality_table.value)]] 03)----TableScan: high_cardinality_table projection=[value, f_dkey] physical_plan 01)ProjectionExec: expr=[f_dkey@0 as f_dkey, count(Int64(1))@1 as count(*), sum(high_cardinality_table.value)@2 as sum(high_cardinality_table.value)] -02)--AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[count() as count(Int64(1)), sum(high_cardinality_table.value)] +02)--AggregateExec: mode=SinglePartitioned, gby=[f_dkey@1 as f_dkey], aggr=[count(Int64(1)), sum(high_cardinality_table.value)] 03)----DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/high_cardinality/f_dkey=A/data.parquet, WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/high_cardinality/f_dkey=D/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/high_cardinality/f_dkey=B/data.parquet, WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/high_cardinality/f_dkey=E/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/high_cardinality/f_dkey=C/data.parquet]]}, projection=[value, f_dkey], output_partitioning=Hash([f_dkey@1], 3), file_type=parquet # Verify results with optimization match results without optimization @@ -585,13 +585,13 @@ EXPLAIN SELECT f_dkey, count(*), sum(value) FROM high_cardinality_table GROUP BY ---- logical_plan 01)Projection: high_cardinality_table.f_dkey, count(Int64(1)) AS count(*), sum(high_cardinality_table.value) -02)--Aggregate: groupBy=[[high_cardinality_table.f_dkey]], aggr=[[count() AS count(Int64(1)), sum(high_cardinality_table.value)]] +02)--Aggregate: groupBy=[[high_cardinality_table.f_dkey]], aggr=[[count(Int64(1)), sum(high_cardinality_table.value)]] 03)----TableScan: high_cardinality_table projection=[value, f_dkey] physical_plan 01)ProjectionExec: expr=[f_dkey@0 as f_dkey, count(Int64(1))@1 as count(*), sum(high_cardinality_table.value)@2 as sum(high_cardinality_table.value)] -02)--AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[count() as count(Int64(1)), sum(high_cardinality_table.value)] +02)--AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey], aggr=[count(Int64(1)), sum(high_cardinality_table.value)] 03)----RepartitionExec: partitioning=Hash([f_dkey@0], 3), input_partitions=3 -04)------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[count() as count(Int64(1)), sum(high_cardinality_table.value)] +04)------AggregateExec: mode=Partial, gby=[f_dkey@1 as f_dkey], aggr=[count(Int64(1)), sum(high_cardinality_table.value)] 05)--------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/high_cardinality/f_dkey=A/data.parquet, WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/high_cardinality/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/high_cardinality/f_dkey=C/data.parquet, WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/high_cardinality/f_dkey=D/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/high_cardinality/f_dkey=E/data.parquet]]}, projection=[value, f_dkey], file_type=parquet query TIR rowsort @@ -717,11 +717,11 @@ GROUP BY f_dkey, timestamp; ---- logical_plan 01)Projection: fact_table.f_dkey, fact_table.timestamp, count(Int64(1)) AS count(*), avg(fact_table.value) -02)--Aggregate: groupBy=[[fact_table.f_dkey, fact_table.timestamp]], aggr=[[count() AS count(Int64(1)), avg(fact_table.value)]] +02)--Aggregate: groupBy=[[fact_table.f_dkey, fact_table.timestamp]], aggr=[[count(Int64(1)), avg(fact_table.value)]] 03)----TableScan: fact_table projection=[timestamp, value, f_dkey] physical_plan 01)ProjectionExec: expr=[f_dkey@0 as f_dkey, timestamp@1 as timestamp, count(Int64(1))@2 as count(*), avg(fact_table.value)@3 as avg(fact_table.value)] -02)--AggregateExec: mode=SinglePartitioned, gby=[f_dkey@2 as f_dkey, timestamp@0 as timestamp], aggr=[count() as count(Int64(1)), avg(fact_table.value)] +02)--AggregateExec: mode=SinglePartitioned, gby=[f_dkey@2 as f_dkey, timestamp@0 as timestamp], aggr=[count(Int64(1)), avg(fact_table.value)] 03)----DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/preserve_file_partitioning/fact/f_dkey=C/data.parquet]]}, projection=[timestamp, value, f_dkey], output_partitioning=Hash([f_dkey@2], 3), file_type=parquet query TPIR rowsort diff --git a/datafusion/sqllogictest/test_files/projection_pushdown.slt b/datafusion/sqllogictest/test_files/projection_pushdown.slt index 1020741e21aa4..c92c95fdfbc59 100644 --- a/datafusion/sqllogictest/test_files/projection_pushdown.slt +++ b/datafusion/sqllogictest/test_files/projection_pushdown.slt @@ -1858,11 +1858,11 @@ FROM simple_struct GROUP BY s; ---- logical_plan 01)Projection: get_field(simple_struct.s, Utf8("label")) IS NOT NULL AS has_label, count(Int64(1)) -02)--Aggregate: groupBy=[[simple_struct.s]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[simple_struct.s]], aggr=[[count(Int64(1))]] 03)----TableScan: simple_struct projection=[s] physical_plan 01)ProjectionExec: expr=[get_field(s@0, label) IS NOT NULL as has_label, count(Int64(1))@1 as count(Int64(1))] -02)--AggregateExec: mode=Single, gby=[s@0 as s], aggr=[count() as count(Int64(1))] +02)--AggregateExec: mode=Single, gby=[s@0 as s], aggr=[count(Int64(1))] 03)----DataSourceExec: file_groups={1 group: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/projection_pushdown/simple.parquet]]}, projection=[s], file_type=parquet # Verify correctness - all labels are non-null diff --git a/datafusion/sqllogictest/test_files/push_down_filter_regression.slt b/datafusion/sqllogictest/test_files/push_down_filter_regression.slt index 4bb6e293a6d1f..c260efae95334 100644 --- a/datafusion/sqllogictest/test_files/push_down_filter_regression.slt +++ b/datafusion/sqllogictest/test_files/push_down_filter_regression.slt @@ -649,14 +649,14 @@ EXPLAIN SELECT k, c FROM (SELECT random() < 0.5 AS k, count(*) AS c FROM generat logical_plan 01)Projection: random() < Float64(0.5) AS k, count(Int64(1)) AS c 02)--Filter: random() < Float64(0.5) -03)----Aggregate: groupBy=[[random() < Float64(0.5)]], aggr=[[count() AS count(Int64(1))]] +03)----Aggregate: groupBy=[[random() < Float64(0.5)]], aggr=[[count(Int64(1))]] 04)------TableScan: generate_series() projection=[] physical_plan 01)ProjectionExec: expr=[random() < Float64(0.5)@0 as k, count(Int64(1))@1 as c] 02)--FilterExec: random() < Float64(0.5)@0 -03)----AggregateExec: mode=FinalPartitioned, gby=[random() < Float64(0.5)@0 as random() < Float64(0.5)], aggr=[count() as count(Int64(1))] +03)----AggregateExec: mode=FinalPartitioned, gby=[random() < Float64(0.5)@0 as random() < Float64(0.5)], aggr=[count(Int64(1))] 04)------RepartitionExec: partitioning=Hash([random() < Float64(0.5)@0], 4), input_partitions=4 -05)--------AggregateExec: mode=Partial, gby=[random() < 0.5 as random() < Float64(0.5)], aggr=[count() as count(Int64(1))] +05)--------AggregateExec: mode=Partial, gby=[random() < 0.5 as random() < Float64(0.5)], aggr=[count(Int64(1))] 06)----------RepartitionExec: partitioning=RoundRobinBatch(4), input_partitions=1 07)------------LazyMemoryExec: partitions=1, batch_generators=[generate_series: start=1, end=10000, batch_size=8192] diff --git a/datafusion/sqllogictest/test_files/repartition_subset_satisfaction.slt b/datafusion/sqllogictest/test_files/repartition_subset_satisfaction.slt index d6399f2419c18..5371ca59beea1 100644 --- a/datafusion/sqllogictest/test_files/repartition_subset_satisfaction.slt +++ b/datafusion/sqllogictest/test_files/repartition_subset_satisfaction.slt @@ -156,14 +156,14 @@ ORDER BY f_dkey, time_bin; logical_plan 01)Sort: fact_table_ordered.f_dkey ASC NULLS LAST, time_bin ASC NULLS LAST 02)--Projection: fact_table_ordered.f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp) AS time_bin, count(Int64(1)) AS count(*), avg(fact_table_ordered.value) -03)----Aggregate: groupBy=[[fact_table_ordered.f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"), fact_table_ordered.timestamp)]], aggr=[[count() AS count(Int64(1)), avg(fact_table_ordered.value)]] +03)----Aggregate: groupBy=[[fact_table_ordered.f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"), fact_table_ordered.timestamp)]], aggr=[[count(Int64(1)), avg(fact_table_ordered.value)]] 04)------TableScan: fact_table_ordered projection=[timestamp, value, f_dkey] physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST, time_bin@1 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1 as time_bin, count(Int64(1))@2 as count(*), avg(fact_table_ordered.value)@3 as avg(fact_table_ordered.value)] -03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count() as count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted +03)----AggregateExec: mode=FinalPartitioned, gby=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1 as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted 04)------RepartitionExec: partitioning=Hash([f_dkey@0, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1], 3), input_partitions=3, preserve_order=true, sort_exprs=f_dkey@0 ASC NULLS LAST, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1 ASC NULLS LAST -05)--------AggregateExec: mode=Partial, gby=[f_dkey@2 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@0) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count() as count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted +05)--------AggregateExec: mode=Partial, gby=[f_dkey@2 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@0) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted 06)----------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=C/data.parquet]]}, projection=[timestamp, value, f_dkey], output_ordering=[f_dkey@2 ASC NULLS LAST, timestamp@0 ASC NULLS LAST], output_partitioning=Hash([f_dkey@2], 3), file_type=parquet # Verify results without subset satisfaction @@ -198,12 +198,12 @@ ORDER BY f_dkey, time_bin; logical_plan 01)Sort: fact_table_ordered.f_dkey ASC NULLS LAST, time_bin ASC NULLS LAST 02)--Projection: fact_table_ordered.f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp) AS time_bin, count(Int64(1)) AS count(*), avg(fact_table_ordered.value) -03)----Aggregate: groupBy=[[fact_table_ordered.f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"), fact_table_ordered.timestamp)]], aggr=[[count() AS count(Int64(1)), avg(fact_table_ordered.value)]] +03)----Aggregate: groupBy=[[fact_table_ordered.f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"), fact_table_ordered.timestamp)]], aggr=[[count(Int64(1)), avg(fact_table_ordered.value)]] 04)------TableScan: fact_table_ordered projection=[timestamp, value, f_dkey] physical_plan 01)SortPreservingMergeExec: [f_dkey@0 ASC NULLS LAST, time_bin@1 ASC NULLS LAST] 02)--ProjectionExec: expr=[f_dkey@0 as f_dkey, date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)@1 as time_bin, count(Int64(1))@2 as count(*), avg(fact_table_ordered.value)@3 as avg(fact_table_ordered.value)] -03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@2 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@0) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count() as count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted +03)----AggregateExec: mode=SinglePartitioned, gby=[f_dkey@2 as f_dkey, date_bin(IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }, timestamp@0) as date_bin(IntervalMonthDayNano("IntervalMonthDayNano { months: 0, days: 0, nanoseconds: 30000000000 }"),fact_table_ordered.timestamp)], aggr=[count(Int64(1)), avg(fact_table_ordered.value)], ordering_mode=Sorted 04)------DataSourceExec: file_groups={3 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=A/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=B/data.parquet], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/scratch/repartition_subset_satisfaction/fact/f_dkey=C/data.parquet]]}, projection=[timestamp, value, f_dkey], output_ordering=[f_dkey@2 ASC NULLS LAST, timestamp@0 ASC NULLS LAST], output_partitioning=Hash([f_dkey@2], 3), file_type=parquet # Verify results match with subset satisfaction diff --git a/datafusion/sqllogictest/test_files/select.slt b/datafusion/sqllogictest/test_files/select.slt index 1ac0943b17edd..d65cfa900c249 100644 --- a/datafusion/sqllogictest/test_files/select.slt +++ b/datafusion/sqllogictest/test_files/select.slt @@ -1565,16 +1565,16 @@ GROUP BY c2; ---- logical_plan 01)Projection: aggregate_test_100.c2, count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[aggregate_test_100.c2]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[aggregate_test_100.c2]], aggr=[[count(Int64(1))]] 03)----Projection: aggregate_test_100.c2 04)------Sort: aggregate_test_100.c1 ASC NULLS LAST, aggregate_test_100.c2 ASC NULLS LAST, fetch=4 05)--------Projection: aggregate_test_100.c2, aggregate_test_100.c1 06)----------TableScan: aggregate_test_100 projection=[c1, c2] physical_plan 01)ProjectionExec: expr=[c2@0 as c2, count(Int64(1))@1 as count(*)] -02)--AggregateExec: mode=FinalPartitioned, gby=[c2@0 as c2], aggr=[count() as count(Int64(1))] +02)--AggregateExec: mode=FinalPartitioned, gby=[c2@0 as c2], aggr=[count(Int64(1))] 03)----RepartitionExec: partitioning=Hash([c2@0], 2), input_partitions=2 -04)------AggregateExec: mode=Partial, gby=[c2@0 as c2], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[c2@0 as c2], aggr=[count(Int64(1))] 05)--------RepartitionExec: partitioning=RoundRobinBatch(2), input_partitions=1 06)----------ProjectionExec: expr=[c2@1 as c2] 07)------------SortExec: TopK(fetch=4), expr=[c1@0 ASC NULLS LAST, c2@1 ASC NULLS LAST], preserve_partitioning=[false] diff --git a/datafusion/sqllogictest/test_files/single_distinct_to_groupby.slt b/datafusion/sqllogictest/test_files/single_distinct_to_groupby.slt index cc414d025ef13..fddfc661b8cc5 100644 --- a/datafusion/sqllogictest/test_files/single_distinct_to_groupby.slt +++ b/datafusion/sqllogictest/test_files/single_distinct_to_groupby.slt @@ -60,7 +60,7 @@ EXPLAIN SELECT g, count(*) AS records, count(DISTINCT x) AS distinct_x FROM t GR logical_plan 01)Projection: t.g, CASE WHEN sum(alias2) IS NOT NULL THEN sum(alias2) ELSE Int64(0) END AS records, count(alias1) AS distinct_x 02)--Aggregate: groupBy=[[t.g]], aggr=[[sum(alias2), count(alias1)]] -03)----Aggregate: groupBy=[[t.g, t.x AS alias1]], aggr=[[count() AS alias2]] +03)----Aggregate: groupBy=[[t.g, t.x AS alias1]], aggr=[[count(Int64(1)) AS alias2]] 04)------TableScan: t projection=[g, x] # A count next to min, max and sum, all beside the distinct count @@ -73,7 +73,7 @@ FROM t GROUP BY g; logical_plan 01)Projection: t.g, CASE WHEN sum(alias2) IS NOT NULL THEN sum(alias2) ELSE Int64(0) END AS records, count(alias1) AS distinct_x, min(alias3) AS min_v, max(alias4) AS max_v, sum(alias5) AS sum_v 02)--Aggregate: groupBy=[[t.g]], aggr=[[sum(alias2), count(alias1), min(alias3), max(alias4), sum(alias5)]] -03)----Aggregate: groupBy=[[t.g, t.x AS alias1]], aggr=[[count() AS alias2, min(t.v) AS alias3, max(t.v) AS alias4, sum(CAST(t.v AS Int64)) AS alias5]] +03)----Aggregate: groupBy=[[t.g, t.x AS alias1]], aggr=[[count(Int64(1)) AS alias2, min(t.v) AS alias3, max(t.v) AS alias4, sum(CAST(t.v AS Int64)) AS alias5]] 04)------TableScan: t projection=[g, x, v] # A count with a FILTER still blocks the rewrite: the filter is per input row, @@ -83,7 +83,7 @@ EXPLAIN SELECT g, count(*) FILTER (WHERE v > 1) AS records, count(DISTINCT x) AS ---- logical_plan 01)Projection: t.g, count(Int64(1)) FILTER (WHERE t.v > Int64(1)) AS count(*) FILTER (WHERE t.v > Int64(1)) AS records, count(DISTINCT t.x) AS distinct_x -02)--Aggregate: groupBy=[[t.g]], aggr=[[count() FILTER (WHERE t.v > Int32(1)) AS count(Int64(1)) FILTER (WHERE t.v > Int64(1)), count(DISTINCT t.x)]] +02)--Aggregate: groupBy=[[t.g]], aggr=[[count(Int64(1)) FILTER (WHERE t.v > Int32(1)) AS count(Int64(1)) FILTER (WHERE t.v > Int64(1)), count(DISTINCT t.x)]] 03)----TableScan: t projection=[g, x, v] # An unsupported non-distinct aggregate still blocks the rewrite @@ -108,7 +108,7 @@ EXPLAIN SELECT g, count(*) AS records, count(DISTINCT xi) AS distinct_xi FROM t ---- logical_plan 01)Projection: t.g, count(Int64(1)) AS count(*) AS records, count(DISTINCT t.xi) AS distinct_xi -02)--Aggregate: groupBy=[[t.g]], aggr=[[count() AS count(Int64(1)), count(DISTINCT t.xi)]] +02)--Aggregate: groupBy=[[t.g]], aggr=[[count(Int64(1)), count(DISTINCT t.xi)]] 03)----TableScan: t projection=[g, xi] # `sum` reports nothing about `GroupsAccumulator` support from the argument @@ -121,7 +121,7 @@ EXPLAIN SELECT g, count(*) AS records, sum(DISTINCT v) AS distinct_v FROM t GROU ---- logical_plan 01)Projection: t.g, count(Int64(1)) AS count(*) AS records, sum(DISTINCT t.v) AS distinct_v -02)--Aggregate: groupBy=[[t.g]], aggr=[[count() AS count(Int64(1)), sum(DISTINCT CAST(t.v AS Int64))]] +02)--Aggregate: groupBy=[[t.g]], aggr=[[count(Int64(1)), sum(DISTINCT CAST(t.v AS Int64))]] 03)----TableScan: t projection=[g, v] # `min(DISTINCT v)` is the same value as `min(v)`. EliminateAggregateDistinct @@ -133,7 +133,7 @@ EXPLAIN SELECT g, count(*) AS records, min(DISTINCT v) AS distinct_min_v FROM t ---- logical_plan 01)Projection: t.g, count(Int64(1)) AS count(*) AS records, min(DISTINCT t.v) AS distinct_min_v -02)--Aggregate: groupBy=[[t.g]], aggr=[[count() AS count(Int64(1)), min(t.v) AS min(DISTINCT t.v)]] +02)--Aggregate: groupBy=[[t.g]], aggr=[[count(Int64(1)), min(t.v) AS min(DISTINCT t.v)]] 03)----TableScan: t projection=[g, v] # The gate covers only the count. A plan that already qualified through sum, @@ -293,7 +293,7 @@ logical_plan 02)--Projection: t.g, CASE WHEN sum(alias2) IS NOT NULL THEN sum(alias2) ELSE Int64(0) END AS records, count(alias1) AS distinct_x 03)----Filter: CASE WHEN sum(alias2) IS NOT NULL THEN sum(alias2) ELSE Int64(0) END > Int64(1) AND count(alias1) > Int64(0) 04)------Aggregate: groupBy=[[t.g]], aggr=[[sum(alias2), count(alias1)]] -05)--------Aggregate: groupBy=[[t.g, t.x AS alias1]], aggr=[[count() AS alias2]] +05)--------Aggregate: groupBy=[[t.g, t.x AS alias1]], aggr=[[count(Int64(1)) AS alias2]] 06)----------TableScan: t projection=[g, x] statement ok diff --git a/datafusion/sqllogictest/test_files/subquery.slt b/datafusion/sqllogictest/test_files/subquery.slt index 994ec93d8d37f..ca0b0b1ff8719 100644 --- a/datafusion/sqllogictest/test_files/subquery.slt +++ b/datafusion/sqllogictest/test_files/subquery.slt @@ -541,7 +541,7 @@ logical_plan 03)----Subquery: 04)------Projection: count(Int64(1)) AS count(*) 05)--------Filter: sum(outer_ref(t1.t1_int) + t2.t2_id) > Int64(0) -06)----------Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1)), sum(CAST(outer_ref(t1.t1_int) + t2.t2_id AS Int64))]] +06)----------Aggregate: groupBy=[[]], aggr=[[count(Int64(1)), sum(CAST(outer_ref(t1.t1_int) + t2.t2_id AS Int64))]] 07)------------Filter: outer_ref(t1.t1_name) = t2.t2_name 08)--------------TableScan: t2 09)----TableScan: t1 projection=[t1_id, t1_name, t1_int] @@ -725,7 +725,7 @@ logical_plan 01)Projection: () AS b 02)--Subquery: 03)----Projection: count(Int64(1)) AS count(*) -04)------Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +04)------Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 05)--------TableScan: t1 projection=[] 06)--EmptyRelation: rows=1 @@ -739,10 +739,10 @@ logical_plan 01)Projection: () AS b, () 02)--Subquery: 03)----Projection: count(Int64(1)) AS count(*) -04)------Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +04)------Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 05)--------TableScan: t1 projection=[] 06)--Subquery: -07)----Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +07)----Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 08)------TableScan: t2 projection=[] 09)--EmptyRelation: rows=1 @@ -808,10 +808,10 @@ logical_plan 01)Projection: () AS b, () 02)--Subquery: 03)----Projection: count(Int64(1)) AS count(*) -04)------Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +04)------Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 05)--------TableScan: t1 projection=[] 06)--Subquery: -07)----Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +07)----Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 08)------TableScan: t2 projection=[] 09)--EmptyRelation: rows=1 physical_plan @@ -841,7 +841,7 @@ logical_plan 03)----TableScan: t1 projection=[t1_id, t1_int] 04)----SubqueryAlias: __scalar_sq_1 05)------Projection: count(Int64(1)) AS count(*), t2.t2_int, Boolean(true) AS __always_true -06)--------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count() AS count(Int64(1))]] +06)--------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count(Int64(1))]] 07)----------TableScan: t2 projection=[t2_int] query II rowsort @@ -879,7 +879,7 @@ logical_plan 03)----TableScan: t1 projection=[t1_id, t1_int] 04)----SubqueryAlias: __scalar_sq_1 05)------Projection: count(Int64(1)) AS count(*), t2.t2_int, Boolean(true) AS __always_true -06)--------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count() AS count(Int64(1))]] +06)--------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count(Int64(1))]] 07)----------TableScan: t2 projection=[t2_int] query II rowsort @@ -962,7 +962,7 @@ logical_plan 03)----TableScan: t1 projection=[t1_id, t1_int] 04)----SubqueryAlias: __scalar_sq_1 05)------Projection: count(Int64(1)) AS _cnt, t2.t2_int, Boolean(true) AS __always_true -06)--------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count() AS count(Int64(1))]] +06)--------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count(Int64(1))]] 07)----------TableScan: t2 projection=[t2_int] query II rowsort @@ -983,7 +983,7 @@ logical_plan 03)----TableScan: t1 projection=[t1_id, t1_int] 04)----SubqueryAlias: __scalar_sq_1 05)------Projection: count(Int64(1)) + Int64(2) AS _cnt, t2.t2_int, Boolean(true) AS __always_true -06)--------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count() AS count(Int64(1))]] +06)--------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count(Int64(1))]] 07)----------TableScan: t2 projection=[t2_int] query II rowsort @@ -1006,7 +1006,7 @@ logical_plan 05)--------TableScan: t1 projection=[t1_id, t1_int] 06)--------SubqueryAlias: __scalar_sq_1 07)----------Projection: count(Int64(1)) AS count(*), t2.t2_id, Boolean(true) AS __always_true -08)------------Aggregate: groupBy=[[t2.t2_id]], aggr=[[count() AS count(Int64(1))]] +08)------------Aggregate: groupBy=[[t2.t2_id]], aggr=[[count(Int64(1))]] 09)--------------TableScan: t2 projection=[t2_id] query I rowsort @@ -1028,7 +1028,7 @@ logical_plan 04)----SubqueryAlias: __scalar_sq_1 05)------Projection: count(Int64(1)) AS count(*) + Int64(2) AS cnt_plus_2, t2.t2_int 06)--------Filter: count(Int64(1)) > Int64(1) -07)----------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count() AS count(Int64(1))]] +07)----------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count(Int64(1))]] 08)------------TableScan: t2 projection=[t2_int] query II rowsort @@ -1050,7 +1050,7 @@ logical_plan 03)----TableScan: t1 projection=[t1_id, t1_int] 04)----SubqueryAlias: __scalar_sq_1 05)------Projection: count(Int64(1)) + Int64(2) AS cnt_plus_2, t2.t2_int, count(Int64(1)), Boolean(true) AS __always_true -06)--------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count() AS count(Int64(1))]] +06)--------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count(Int64(1))]] 07)----------TableScan: t2 projection=[t2_int] query II rowsort @@ -1074,7 +1074,7 @@ logical_plan 06)----------TableScan: t1 projection=[t1_int] 07)--------SubqueryAlias: __scalar_sq_1 08)----------Projection: count(Int64(1)) AS count(*), t2.t2_int, Boolean(true) AS __always_true -09)------------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count() AS count(Int64(1))]] +09)------------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count(Int64(1))]] 10)--------------TableScan: t2 projection=[t2_int] query I rowsort @@ -1095,7 +1095,7 @@ logical_plan 05)--------TableScan: t1 projection=[t1_int] 06)--------SubqueryAlias: __scalar_sq_1 07)----------Projection: count(Int64(1)) AS cnt, t2.t2_int, Boolean(true) AS __always_true -08)------------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count() AS count(Int64(1))]] +08)------------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count(Int64(1))]] 09)--------------TableScan: t2 projection=[t2_int] @@ -1125,7 +1125,7 @@ logical_plan 05)--------TableScan: t1 projection=[t1_int] 06)--------SubqueryAlias: __scalar_sq_1 07)----------Projection: count(Int64(1)) + Int64(1) + Int64(1) AS cnt_plus_two, t2.t2_int, count(Int64(1)), Boolean(true) AS __always_true -08)------------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count() AS count(Int64(1))]] +08)------------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count(Int64(1))]] 09)--------------TableScan: t2 projection=[t2_int] query I rowsort @@ -1154,7 +1154,7 @@ logical_plan 05)--------TableScan: t1 projection=[t1_int] 06)--------SubqueryAlias: __scalar_sq_1 07)----------Projection: CASE WHEN count(Int64(1)) = Int64(1) THEN Int64(NULL) ELSE count(Int64(1)) END AS cnt, t2.t2_int, Boolean(true) AS __always_true -08)------------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count() AS count(Int64(1))]] +08)------------Aggregate: groupBy=[[t2.t2_int]], aggr=[[count(Int64(1))]] 09)--------------TableScan: t2 projection=[t2_int] diff --git a/datafusion/sqllogictest/test_files/tpch/plans/q1.slt.part b/datafusion/sqllogictest/test_files/tpch/plans/q1.slt.part index 5d7670090c744..b227f94553e2f 100644 --- a/datafusion/sqllogictest/test_files/tpch/plans/q1.slt.part +++ b/datafusion/sqllogictest/test_files/tpch/plans/q1.slt.part @@ -42,7 +42,7 @@ explain select logical_plan 01)Sort: lineitem.l_returnflag ASC NULLS LAST, lineitem.l_linestatus ASC NULLS LAST 02)--Projection: lineitem.l_returnflag, lineitem.l_linestatus, sum(lineitem.l_quantity) AS sum_qty, sum(lineitem.l_extendedprice) AS sum_base_price, sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount) AS sum_disc_price, sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount * Int64(1) + lineitem.l_tax) AS sum_charge, avg(lineitem.l_quantity) AS avg_qty, avg(lineitem.l_extendedprice) AS avg_price, avg(lineitem.l_discount) AS avg_disc, count(Int64(1)) AS count(*) AS count_order -03)----Aggregate: groupBy=[[lineitem.l_returnflag, lineitem.l_linestatus]], aggr=[[sum(lineitem.l_quantity), sum(lineitem.l_extendedprice), sum(__common_expr_1) AS sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount), sum(__common_expr_1 * (Decimal128(1,20,0) + lineitem.l_tax)) AS sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount * Int64(1) + lineitem.l_tax), avg(lineitem.l_quantity), avg(lineitem.l_extendedprice), avg(lineitem.l_discount), count() AS count(Int64(1))]] +03)----Aggregate: groupBy=[[lineitem.l_returnflag, lineitem.l_linestatus]], aggr=[[sum(lineitem.l_quantity), sum(lineitem.l_extendedprice), sum(__common_expr_1) AS sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount), sum(__common_expr_1 * (Decimal128(1,20,0) + lineitem.l_tax)) AS sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount * Int64(1) + lineitem.l_tax), avg(lineitem.l_quantity), avg(lineitem.l_extendedprice), avg(lineitem.l_discount), count(Int64(1))]] 04)------Projection: lineitem.l_extendedprice * (Decimal128(1,20,0) - lineitem.l_discount) AS __common_expr_1, lineitem.l_quantity, lineitem.l_extendedprice, lineitem.l_discount, lineitem.l_tax, lineitem.l_returnflag, lineitem.l_linestatus 05)--------Filter: lineitem.l_shipdate <= Date32("1998-09-02") 06)----------TableScan: lineitem projection=[l_quantity, l_extendedprice, l_discount, l_tax, l_returnflag, l_linestatus, l_shipdate], partial_filters=[lineitem.l_shipdate <= Date32("1998-09-02")] @@ -50,9 +50,9 @@ physical_plan 01)SortPreservingMergeExec: [l_returnflag@0 ASC NULLS LAST, l_linestatus@1 ASC NULLS LAST] 02)--ProjectionExec: expr=[l_returnflag@0 as l_returnflag, l_linestatus@1 as l_linestatus, sum(lineitem.l_quantity)@2 as sum_qty, sum(lineitem.l_extendedprice)@3 as sum_base_price, sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount)@4 as sum_disc_price, sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount * Int64(1) + lineitem.l_tax)@5 as sum_charge, avg(lineitem.l_quantity)@6 as avg_qty, avg(lineitem.l_extendedprice)@7 as avg_price, avg(lineitem.l_discount)@8 as avg_disc, count(Int64(1))@9 as count_order] 03)----SortExec: expr=[l_returnflag@0 ASC NULLS LAST, l_linestatus@1 ASC NULLS LAST], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[l_returnflag@0 as l_returnflag, l_linestatus@1 as l_linestatus], aggr=[sum(lineitem.l_quantity), sum(lineitem.l_extendedprice), sum(__common_expr_1) as sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount), sum(__common_expr_1 * 1 + lineitem.l_tax) as sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount * Int64(1) + lineitem.l_tax), avg(lineitem.l_quantity), avg(lineitem.l_extendedprice), avg(lineitem.l_discount), count() as count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[l_returnflag@0 as l_returnflag, l_linestatus@1 as l_linestatus], aggr=[sum(lineitem.l_quantity), sum(lineitem.l_extendedprice), sum(__common_expr_1) as sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount), sum(__common_expr_1 * 1 + lineitem.l_tax) as sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount * Int64(1) + lineitem.l_tax), avg(lineitem.l_quantity), avg(lineitem.l_extendedprice), avg(lineitem.l_discount), count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([l_returnflag@0, l_linestatus@1], 4), input_partitions=4 -06)----------AggregateExec: mode=Partial, gby=[l_returnflag@5 as l_returnflag, l_linestatus@6 as l_linestatus], aggr=[sum(lineitem.l_quantity), sum(lineitem.l_extendedprice), sum(__common_expr_1) as sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount), sum(__common_expr_1 * 1 + lineitem.l_tax) as sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount * Int64(1) + lineitem.l_tax), avg(lineitem.l_quantity), avg(lineitem.l_extendedprice), avg(lineitem.l_discount), count() as count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[l_returnflag@5 as l_returnflag, l_linestatus@6 as l_linestatus], aggr=[sum(lineitem.l_quantity), sum(lineitem.l_extendedprice), sum(__common_expr_1) as sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount), sum(__common_expr_1 * 1 + lineitem.l_tax) as sum(lineitem.l_extendedprice * Int64(1) - lineitem.l_discount * Int64(1) + lineitem.l_tax), avg(lineitem.l_quantity), avg(lineitem.l_extendedprice), avg(lineitem.l_discount), count(Int64(1))] 07)------------ProjectionExec: expr=[l_extendedprice@0 * (1 - l_discount@1) as __common_expr_1, l_quantity@2 as l_quantity, l_extendedprice@0 as l_extendedprice, l_discount@1 as l_discount, l_tax@3 as l_tax, l_returnflag@4 as l_returnflag, l_linestatus@5 as l_linestatus] 08)--------------FilterExec: l_shipdate@6 <= 1998-09-02, projection=[l_extendedprice@1, l_discount@2, l_quantity@0, l_tax@3, l_returnflag@4, l_linestatus@5] 09)----------------DataSourceExec: file_groups={4 groups: [[WORKSPACE_ROOT/datafusion/sqllogictest/test_files/tpch/data/lineitem.tbl:0..18561749], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/tpch/data/lineitem.tbl:18561749..37123498], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/tpch/data/lineitem.tbl:37123498..55685247], [WORKSPACE_ROOT/datafusion/sqllogictest/test_files/tpch/data/lineitem.tbl:55685247..74246996]]}, projection=[l_quantity, l_extendedprice, l_discount, l_tax, l_returnflag, l_linestatus, l_shipdate], constraints=[PrimaryKey([0, 3])], file_type=csv, has_header=false diff --git a/datafusion/sqllogictest/test_files/tpch/plans/q13.slt.part b/datafusion/sqllogictest/test_files/tpch/plans/q13.slt.part index d5b36f1398809..9f9cbb3b6af68 100644 --- a/datafusion/sqllogictest/test_files/tpch/plans/q13.slt.part +++ b/datafusion/sqllogictest/test_files/tpch/plans/q13.slt.part @@ -42,7 +42,7 @@ limit 10; logical_plan 01)Sort: custdist DESC NULLS FIRST, c_orders.c_count DESC NULLS FIRST, fetch=10 02)--Projection: c_orders.c_count, count(Int64(1)) AS count(*) AS custdist -03)----Aggregate: groupBy=[[c_orders.c_count]], aggr=[[count() AS count(Int64(1))]] +03)----Aggregate: groupBy=[[c_orders.c_count]], aggr=[[count(Int64(1))]] 04)------SubqueryAlias: c_orders 05)--------Projection: count(orders.o_orderkey) AS c_count 06)----------Aggregate: groupBy=[[customer.c_custkey]], aggr=[[count(orders.o_orderkey)]] @@ -56,9 +56,9 @@ physical_plan 01)SortPreservingMergeExec: [custdist@1 DESC, c_count@0 DESC], fetch=10 02)--ProjectionExec: expr=[c_count@0 as c_count, count(Int64(1))@1 as custdist] 03)----SortExec: TopK(fetch=10), expr=[count(Int64(1))@1 DESC, c_count@0 DESC], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[c_count@0 as c_count], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[c_count@0 as c_count], aggr=[count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([c_count@0], 4), input_partitions=4 -06)----------AggregateExec: mode=Partial, gby=[c_count@0 as c_count], aggr=[count() as count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[c_count@0 as c_count], aggr=[count(Int64(1))] 07)------------ProjectionExec: expr=[count(orders.o_orderkey)@1 as c_count] 08)--------------AggregateExec: mode=SinglePartitioned, gby=[c_custkey@0 as c_custkey], aggr=[count(orders.o_orderkey)] 09)----------------HashJoinExec: mode=Partitioned, join_type=Left, on=[(c_custkey@0, o_custkey@1)], projection=[c_custkey@0, o_orderkey@1] diff --git a/datafusion/sqllogictest/test_files/tpch/plans/q21.slt.part b/datafusion/sqllogictest/test_files/tpch/plans/q21.slt.part index c4794d150eea3..47e5d6d888dc5 100644 --- a/datafusion/sqllogictest/test_files/tpch/plans/q21.slt.part +++ b/datafusion/sqllogictest/test_files/tpch/plans/q21.slt.part @@ -60,7 +60,7 @@ order by logical_plan 01)Sort: numwait DESC NULLS FIRST, supplier.s_name ASC NULLS LAST 02)--Projection: supplier.s_name, count(Int64(1)) AS count(*) AS numwait -03)----Aggregate: groupBy=[[supplier.s_name]], aggr=[[count() AS count(Int64(1))]] +03)----Aggregate: groupBy=[[supplier.s_name]], aggr=[[count(Int64(1))]] 04)------Projection: supplier.s_name 05)--------LeftAnti Join: l1.l_orderkey = __correlated_sq_2.l_orderkey Filter: __correlated_sq_2.l_suppkey != l1.l_suppkey 06)----------LeftSemi Join: l1.l_orderkey = __correlated_sq_1.l_orderkey Filter: __correlated_sq_1.l_suppkey != l1.l_suppkey @@ -92,9 +92,9 @@ physical_plan 01)SortPreservingMergeExec: [numwait@1 DESC, s_name@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[s_name@0 as s_name, count(Int64(1))@1 as numwait] 03)----SortExec: expr=[count(Int64(1))@1 DESC, s_name@0 ASC NULLS LAST], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[s_name@0 as s_name], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[s_name@0 as s_name], aggr=[count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([s_name@0], 4), input_partitions=4 -06)----------AggregateExec: mode=Partial, gby=[s_name@0 as s_name], aggr=[count() as count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[s_name@0 as s_name], aggr=[count(Int64(1))] 07)------------HashJoinExec: mode=Partitioned, join_type=LeftAnti, on=[(l_orderkey@1, l_orderkey@0)], filter=l_suppkey@1 != l_suppkey@0, projection=[s_name@0] 08)--------------HashJoinExec: mode=Partitioned, join_type=LeftSemi, on=[(l_orderkey@1, l_orderkey@0)], filter=l_suppkey@1 != l_suppkey@0 09)----------------RepartitionExec: partitioning=Hash([l_orderkey@1], 4), input_partitions=4 diff --git a/datafusion/sqllogictest/test_files/tpch/plans/q22.slt.part b/datafusion/sqllogictest/test_files/tpch/plans/q22.slt.part index 44933c238cc7d..d3f27021f1781 100644 --- a/datafusion/sqllogictest/test_files/tpch/plans/q22.slt.part +++ b/datafusion/sqllogictest/test_files/tpch/plans/q22.slt.part @@ -58,7 +58,7 @@ order by logical_plan 01)Sort: custsale.cntrycode ASC NULLS LAST 02)--Projection: custsale.cntrycode, count(Int64(1)) AS count(*) AS numcust, sum(custsale.c_acctbal) AS totacctbal -03)----Aggregate: groupBy=[[custsale.cntrycode]], aggr=[[count() AS count(Int64(1)), sum(custsale.c_acctbal)]] +03)----Aggregate: groupBy=[[custsale.cntrycode]], aggr=[[count(Int64(1)), sum(custsale.c_acctbal)]] 04)------SubqueryAlias: custsale 05)--------Projection: substr(customer.c_phone, Int64(1), Int64(2)) AS cntrycode, customer.c_acctbal 06)----------LeftAnti Join: customer.c_custkey = __correlated_sq_1.o_custkey @@ -76,9 +76,9 @@ physical_plan 02)--SortPreservingMergeExec: [cntrycode@0 ASC NULLS LAST] 03)----ProjectionExec: expr=[cntrycode@0 as cntrycode, count(Int64(1))@1 as numcust, sum(custsale.c_acctbal)@2 as totacctbal] 04)------SortExec: expr=[cntrycode@0 ASC NULLS LAST], preserve_partitioning=[true] -05)--------AggregateExec: mode=FinalPartitioned, gby=[cntrycode@0 as cntrycode], aggr=[count() as count(Int64(1)), sum(custsale.c_acctbal)] +05)--------AggregateExec: mode=FinalPartitioned, gby=[cntrycode@0 as cntrycode], aggr=[count(Int64(1)), sum(custsale.c_acctbal)] 06)----------RepartitionExec: partitioning=Hash([cntrycode@0], 4), input_partitions=4 -07)------------AggregateExec: mode=Partial, gby=[cntrycode@0 as cntrycode], aggr=[count() as count(Int64(1)), sum(custsale.c_acctbal)] +07)------------AggregateExec: mode=Partial, gby=[cntrycode@0 as cntrycode], aggr=[count(Int64(1)), sum(custsale.c_acctbal)] 08)--------------ProjectionExec: expr=[substr(c_phone@0, 1, 2) as cntrycode, c_acctbal@1 as c_acctbal] 09)----------------HashJoinExec: mode=Partitioned, join_type=LeftAnti, on=[(c_custkey@0, o_custkey@0)], projection=[c_phone@1, c_acctbal@2] 10)------------------RepartitionExec: partitioning=Hash([c_custkey@0], 4), input_partitions=4 diff --git a/datafusion/sqllogictest/test_files/tpch/plans/q4.slt.part b/datafusion/sqllogictest/test_files/tpch/plans/q4.slt.part index 6c48abe9fb888..470d7a6527a52 100644 --- a/datafusion/sqllogictest/test_files/tpch/plans/q4.slt.part +++ b/datafusion/sqllogictest/test_files/tpch/plans/q4.slt.part @@ -42,7 +42,7 @@ order by logical_plan 01)Sort: orders.o_orderpriority ASC NULLS LAST 02)--Projection: orders.o_orderpriority, count(Int64(1)) AS count(*) AS order_count -03)----Aggregate: groupBy=[[orders.o_orderpriority]], aggr=[[count() AS count(Int64(1))]] +03)----Aggregate: groupBy=[[orders.o_orderpriority]], aggr=[[count(Int64(1))]] 04)------Projection: orders.o_orderpriority 05)--------LeftSemi Join: orders.o_orderkey = __correlated_sq_1.l_orderkey 06)----------Projection: orders.o_orderkey, orders.o_orderpriority @@ -56,9 +56,9 @@ physical_plan 01)SortPreservingMergeExec: [o_orderpriority@0 ASC NULLS LAST] 02)--ProjectionExec: expr=[o_orderpriority@0 as o_orderpriority, count(Int64(1))@1 as order_count] 03)----SortExec: expr=[o_orderpriority@0 ASC NULLS LAST], preserve_partitioning=[true] -04)------AggregateExec: mode=FinalPartitioned, gby=[o_orderpriority@0 as o_orderpriority], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=FinalPartitioned, gby=[o_orderpriority@0 as o_orderpriority], aggr=[count(Int64(1))] 05)--------RepartitionExec: partitioning=Hash([o_orderpriority@0], 4), input_partitions=4 -06)----------AggregateExec: mode=Partial, gby=[o_orderpriority@0 as o_orderpriority], aggr=[count() as count(Int64(1))] +06)----------AggregateExec: mode=Partial, gby=[o_orderpriority@0 as o_orderpriority], aggr=[count(Int64(1))] 07)------------HashJoinExec: mode=Partitioned, join_type=LeftSemi, on=[(o_orderkey@0, l_orderkey@0)], projection=[o_orderpriority@1] 08)--------------RepartitionExec: partitioning=Hash([o_orderkey@0], 4), input_partitions=4 09)----------------FilterExec: o_orderdate@1 >= 1993-07-01 AND o_orderdate@1 < 1993-10-01, projection=[o_orderkey@0, o_orderpriority@2] diff --git a/datafusion/sqllogictest/test_files/union.slt b/datafusion/sqllogictest/test_files/union.slt index 1a8e758262d33..115a010330103 100644 --- a/datafusion/sqllogictest/test_files/union.slt +++ b/datafusion/sqllogictest/test_files/union.slt @@ -600,7 +600,7 @@ SELECT count(*) FROM ( ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) -02)--Aggregate: groupBy=[[name]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[name]], aggr=[[count(Int64(1))]] 03)----Union 04)------Aggregate: groupBy=[[t1.name]], aggr=[[]] 05)--------TableScan: t1 projection=[name] @@ -608,7 +608,7 @@ logical_plan 07)--------TableScan: t2 projection=[name] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@1 as count(*)] -02)--AggregateExec: mode=SinglePartitioned, gby=[name@0 as name], aggr=[count() as count(Int64(1))] +02)--AggregateExec: mode=SinglePartitioned, gby=[name@0 as name], aggr=[count(Int64(1))] 03)----InterleaveExec 04)------AggregateExec: mode=FinalPartitioned, gby=[name@0 as name], aggr=[] 05)--------RepartitionExec: partitioning=Hash([name@0], 4), input_partitions=4 @@ -642,7 +642,7 @@ logical_plan 02)--Union 03)----Projection: count(Int64(1)) AS count(*) AS cnt 04)------Limit: skip=0, fetch=3 -05)--------Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +05)--------Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 06)----------SubqueryAlias: a 07)------------Projection: 08)--------------Aggregate: groupBy=[[aggregate_test_100.c1]], aggr=[[]] @@ -663,9 +663,9 @@ physical_plan 02)--UnionExec 03)----ProjectionExec: expr=[CAST(count(Int64(1))@0 AS Int64) as cnt] 04)------GlobalLimitExec: skip=0, fetch=3 -05)--------AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] +05)--------AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] 06)----------CoalescePartitionsExec -07)------------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] +07)------------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] 08)--------------ProjectionExec: expr=[] 09)----------------AggregateExec: mode=FinalPartitioned, gby=[c1@0 as c1], aggr=[] 10)------------------RepartitionExec: partitioning=Hash([c1@0], 4), input_partitions=4 @@ -799,7 +799,7 @@ select x, y from (select 1 as x , max(10) as y) b logical_plan 01)Union 02)--Projection: count(Int64(1)) AS count(*) AS count, a.n -03)----Aggregate: groupBy=[[a.n]], aggr=[[count() AS count(Int64(1))]] +03)----Aggregate: groupBy=[[a.n]], aggr=[[count(Int64(1))]] 04)------SubqueryAlias: a 05)--------Projection: Int64(5) AS n 06)----------EmptyRelation: rows=1 @@ -811,7 +811,7 @@ logical_plan physical_plan 01)UnionExec 02)--ProjectionExec: expr=[count(Int64(1))@1 as count, CAST(n@0 AS Int64) as n] -03)----AggregateExec: mode=SinglePartitioned, gby=[n@0 as n], aggr=[count() as count(Int64(1))] +03)----AggregateExec: mode=SinglePartitioned, gby=[n@0 as n], aggr=[count(Int64(1))] 04)------ProjectionExec: expr=[5 as n] 05)--------PlaceholderRowExec 06)--ProjectionExec: expr=[1 as count, max(Int64(10))@0 as n] diff --git a/datafusion/sqllogictest/test_files/window.slt b/datafusion/sqllogictest/test_files/window.slt index fa65f8902709e..af12fad670796 100644 --- a/datafusion/sqllogictest/test_files/window.slt +++ b/datafusion/sqllogictest/test_files/window.slt @@ -1868,7 +1868,7 @@ EXPLAIN SELECT count(*) as global_count FROM ---- logical_plan 01)Projection: count(Int64(1)) AS count(*) AS global_count -02)--Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]] +02)--Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]] 03)----SubqueryAlias: a 04)------Projection: 05)--------Aggregate: groupBy=[[aggregate_test_100.c1]], aggr=[[]] @@ -1877,9 +1877,9 @@ logical_plan 08)--------------TableScan: aggregate_test_100 projection=[c1, c13], partial_filters=[aggregate_test_100.c13 != Utf8View("C2GT5KVyOPZpgKVl110TyZO0NcJ434")] physical_plan 01)ProjectionExec: expr=[count(Int64(1))@0 as global_count] -02)--AggregateExec: mode=Final, gby=[], aggr=[count() as count(Int64(1))] +02)--AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] 03)----CoalescePartitionsExec -04)------AggregateExec: mode=Partial, gby=[], aggr=[count() as count(Int64(1))] +04)------AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] 05)--------ProjectionExec: expr=[] 06)----------AggregateExec: mode=FinalPartitioned, gby=[c1@0 as c1], aggr=[] 07)------------RepartitionExec: partitioning=Hash([c1@0], 2), input_partitions=2 @@ -7277,7 +7277,7 @@ statement error DataFusion error: This feature is not implemented: Unsupported O SELECT a FROM window_in_filter_t OFFSET row_number() OVER (); # ... and the same for an aggregate function -statement error DataFusion error: This feature is not implemented: Unsupported LIMIT expression: count\(\) AS count\(\*\) +statement error DataFusion error: This feature is not implemented: Unsupported LIMIT expression: count\(Int64\(1\)\) AS count\(\*\) SELECT a FROM window_in_filter_t LIMIT count(*); statement ok diff --git a/datafusion/substrait/tests/cases/roundtrip_logical_plan.rs b/datafusion/substrait/tests/cases/roundtrip_logical_plan.rs index 38fdef982f015..6d783af16e461 100644 --- a/datafusion/substrait/tests/cases/roundtrip_logical_plan.rs +++ b/datafusion/substrait/tests/cases/roundtrip_logical_plan.rs @@ -1328,7 +1328,7 @@ async fn simple_intersect() -> Result<()> { async fn check_wildcard(syntax: &str) -> Result<()> { let expected_plan_str = format!( "Projection: count(Int64(1)) AS {syntax}\ - \n Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]]\ + \n Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]]\ \n Projection:\ \n LeftSemi Join: data.a = data2.a\ \n Aggregate: groupBy=[[data.a]], aggr=[[]]\ @@ -1362,9 +1362,13 @@ async fn simple_intersect() -> Result<()> { check_wildcard("count(*)").await?; check_wildcard("count()").await?; - check_constant("count(1)", "count() AS count(Int64(1))").await?; - check_constant("count(2)", "count() AS count(Int64(2))").await?; - check_constant("count(1 + 2)", "count() AS count(Int64(1) + Int64(2))").await?; + check_constant("count(1)", "count(Int64(1))").await?; + check_constant("count(2)", "count(Int64(2))").await?; + check_constant( + "count(1 + 2)", + "count(Int64(3)) AS count(Int64(1) + Int64(2))", + ) + .await?; Ok(()) } @@ -1542,7 +1546,7 @@ async fn simple_intersect_table_reuse() -> Result<()> { async fn check_wildcard(syntax: &str) -> Result<()> { let expected_plan_str = format!( "Projection: count(Int64(1)) AS {syntax}\ - \n Aggregate: groupBy=[[]], aggr=[[count() AS count(Int64(1))]]\ + \n Aggregate: groupBy=[[]], aggr=[[count(Int64(1))]]\ \n Projection:\ \n LeftSemi Join: left.a = right.a\ \n SubqueryAlias: left\ @@ -1580,9 +1584,13 @@ async fn simple_intersect_table_reuse() -> Result<()> { check_wildcard("count(*)").await?; check_wildcard("count()").await?; - check_constant("count(1)", "count() AS count(Int64(1))").await?; - check_constant("count(2)", "count() AS count(Int64(2))").await?; - check_constant("count(1 + 2)", "count() AS count(Int64(1) + Int64(2))").await?; + check_constant("count(1)", "count(Int64(1))").await?; + check_constant("count(2)", "count(Int64(2))").await?; + check_constant( + "count(1 + 2)", + "count(Int64(3)) AS count(Int64(1) + Int64(2))", + ) + .await?; Ok(()) } From 262e7934e9ae84054dd9f2cf0ce850c09ec671e2 Mon Sep 17 00:00:00 2001 From: wudidapaopao <664920313@qq.com> Date: Thu, 8 Oct 2026 15:32:07 +0800 Subject: [PATCH 5/6] refactor: evaluate literal aggregate arguments once --- datafusion/physical-expr-common/src/utils.rs | 117 +-------- .../physical-plan/benches/aggregate_filter.rs | 44 +++- .../src/aggregates/aggregate_argument.rs | 234 ++++++++++++++++++ .../aggregates/aggregate_hash_table/common.rs | 50 ++-- .../src/aggregates/aggregate_stream.rs | 53 ++-- .../physical-plan/src/aggregates/mod.rs | 1 + 6 files changed, 311 insertions(+), 188 deletions(-) create mode 100644 datafusion/physical-plan/src/aggregates/aggregate_argument.rs diff --git a/datafusion/physical-expr-common/src/utils.rs b/datafusion/physical-expr-common/src/utils.rs index 823f4f84ea417..b67e391dd497d 100644 --- a/datafusion/physical-expr-common/src/utils.rs +++ b/datafusion/physical-expr-common/src/utils.rs @@ -36,8 +36,7 @@ use arrow::datatypes::{ }; use arrow::record_batch::RecordBatch; use arrow::{downcast_dictionary_array, downcast_primitive_array}; -use datafusion_common::{Result, ScalarValue, assert_eq_or_internal_err}; -use datafusion_expr_common::columnar_value::ColumnarValue; +use datafusion_common::Result; use datafusion_expr_common::sort_properties::ExprProperties; /// Represents a [`PhysicalExpr`] node with associated properties (order and @@ -426,68 +425,6 @@ pub fn evaluate_expressions_to_arrays_with_metrics<'a>( .collect::>>() } -/// Largest array, in bytes, that a [`ScalarArrayCache`] keeps between calls. -const MAX_CACHED_SCALAR_ARRAY_BYTES: usize = 1024 * 1024; - -/// Reuses the array expanded from a scalar expression result across batches. -#[derive(Debug, Default)] -pub struct ScalarArrayCache { - cached: Option<(ScalarValue, ArrayRef)>, -} - -impl ScalarArrayCache { - /// Like [`ColumnarValue::into_array_of_size`], but reuses the array built - /// for an earlier, equal scalar. - pub fn into_array_of_size( - &mut self, - value: ColumnarValue, - num_rows: usize, - ) -> Result { - let scalar = match value { - ColumnarValue::Scalar(scalar) => scalar, - array @ ColumnarValue::Array(_) => { - return array.into_array_of_size(num_rows); - } - }; - - if let Some((cached_scalar, array)) = &self.cached - && array.len() >= num_rows - && *cached_scalar == scalar - { - return Ok(array.slice(0, num_rows)); - } - - let array = scalar.to_array_of_size(num_rows)?; - if array.get_array_memory_size() <= MAX_CACHED_SCALAR_ARRAY_BYTES { - self.cached = Some((scalar, Arc::clone(&array))); - } - Ok(array) - } -} - -/// Like [`evaluate_expressions_to_arrays`], but expands scalar results through -/// one cache per expression. -pub fn evaluate_expressions_to_arrays_with_cache( - exprs: &[Arc], - caches: &mut [ScalarArrayCache], - batch: &RecordBatch, -) -> Result> { - assert_eq_or_internal_err!( - exprs.len(), - caches.len(), - "expected one scalar array cache per expression" - ); - let num_rows = batch.num_rows(); - exprs - .iter() - .zip(caches) - .map(|(expr, cache)| { - expr.evaluate(batch) - .and_then(|value| cache.into_array_of_size(value, num_rows)) - }) - .collect() -} - #[cfg(test)] mod tests { @@ -719,56 +656,4 @@ mod tests { assert_eq!(scattered.value(4), 50); Ok(()) } - - #[test] - fn scalar_array_cache_reuses_equal_scalars() -> Result<()> { - let mut cache = ScalarArrayCache::default(); - let one = || ColumnarValue::Scalar(ScalarValue::Int32(Some(1))); - - let first = cache.into_array_of_size(one(), 4)?; - let second = cache.into_array_of_size(one(), 3)?; - assert_eq!(as_int32_array(&second)?, &Int32Array::from(vec![1, 1, 1])); - assert_eq!( - as_int32_array(&second)?.values().as_ptr(), - as_int32_array(&first)?.values().as_ptr() - ); - - let larger = cache.into_array_of_size(one(), 6)?; - assert_eq!(as_int32_array(&larger)?, &Int32Array::from(vec![1; 6])); - let smaller = cache.into_array_of_size(one(), 5)?; - assert_eq!( - as_int32_array(&smaller)?.values().as_ptr(), - as_int32_array(&larger)?.values().as_ptr() - ); - - let two = ColumnarValue::Scalar(ScalarValue::Int32(Some(2))); - let two = cache.into_array_of_size(two, 2)?; - assert_eq!(as_int32_array(&two)?, &Int32Array::from(vec![2, 2])); - - let array: ArrayRef = Arc::new(Int32Array::from(vec![7, 8])); - let result = - cache.into_array_of_size(ColumnarValue::Array(Arc::clone(&array)), 2)?; - assert!(Arc::ptr_eq(&result, &array)); - assert!( - cache - .into_array_of_size(ColumnarValue::Array(array), 3) - .is_err() - ); - Ok(()) - } - - #[test] - fn scalar_array_cache_skips_large_arrays() -> Result<()> { - let mut cache = ScalarArrayCache::default(); - let value = || ColumnarValue::Scalar(ScalarValue::from("x".repeat(1024))); - - let first = cache.into_array_of_size(value(), 2048)?; - let second = cache.into_array_of_size(value(), 2048)?; - assert_eq!(first.as_ref(), second.as_ref()); - assert_ne!( - as_string_array(&first).values().as_ptr(), - as_string_array(&second).values().as_ptr() - ); - Ok(()) - } } diff --git a/datafusion/physical-plan/benches/aggregate_filter.rs b/datafusion/physical-plan/benches/aggregate_filter.rs index ae97cb7388609..417d153abce5c 100644 --- a/datafusion/physical-plan/benches/aggregate_filter.rs +++ b/datafusion/physical-plan/benches/aggregate_filter.rs @@ -29,11 +29,12 @@ use arrow::datatypes::{DataType, Field, Schema, SchemaRef}; use arrow::record_batch::RecordBatch; use criterion::{BenchmarkId, Criterion, criterion_group, criterion_main}; use datafusion_execution::TaskContext; -use datafusion_expr::Operator; +use datafusion_expr::{AggregateUDF, Operator}; +use datafusion_functions_aggregate::count::count_udaf; use datafusion_functions_aggregate::sum::sum_udaf; use datafusion_physical_expr::PhysicalExpr; use datafusion_physical_expr::aggregate::{AggregateExprBuilder, AggregateFunctionExpr}; -use datafusion_physical_expr::expressions::{BinaryExpr, col}; +use datafusion_physical_expr::expressions::{BinaryExpr, col, lit}; use datafusion_physical_plan::aggregates::{ AggregateExec, AggregateMode, PhysicalGroupBy, }; @@ -61,6 +62,7 @@ const FILTER_CASES: &[Option] = &[ enum ArgumentKind { Column, Multiply, + Literal, } impl ArgumentKind { @@ -68,6 +70,7 @@ impl ArgumentKind { match self { Self::Column => "column", Self::Multiply => "multiply", + Self::Literal => "literal", } } } @@ -151,19 +154,30 @@ fn aggregate_expr( argument_kind: ArgumentKind, ) -> Arc { let value = col("value", schema).unwrap(); - let argument: Arc = match argument_kind { - ArgumentKind::Column => value, - ArgumentKind::Multiply => Arc::new(BinaryExpr::new( - Arc::clone(&value), - Operator::Multiply, - value, - )), - }; + let (function, argument): (Arc, Arc) = + match argument_kind { + ArgumentKind::Column => (sum_udaf(), value), + ArgumentKind::Multiply => ( + sum_udaf(), + Arc::new(BinaryExpr::new( + Arc::clone(&value), + Operator::Multiply, + value, + )), + ), + // `COUNT(*)` is planned as `COUNT(1)` + ArgumentKind::Literal => (count_udaf(), lit(1i64)), + }; + let alias = format!( + "{}_{}_{aggregate_index}", + function.name(), + argument_kind.name() + ); Arc::new( - AggregateExprBuilder::new(sum_udaf(), vec![argument]) + AggregateExprBuilder::new(function, vec![argument]) .schema(Arc::clone(schema)) - .alias(format!("sum_{}_{aggregate_index}", argument_kind.name())) + .alias(alias) .build() .unwrap(), ) @@ -266,7 +280,11 @@ fn aggregate_filter_benchmark(c: &mut Criterion) { c, &runtime, AggregateLayout::OneAggregate, - &[ArgumentKind::Column, ArgumentKind::Multiply], + &[ + ArgumentKind::Column, + ArgumentKind::Multiply, + ArgumentKind::Literal, + ], ); benchmark_case( c, diff --git a/datafusion/physical-plan/src/aggregates/aggregate_argument.rs b/datafusion/physical-plan/src/aggregates/aggregate_argument.rs new file mode 100644 index 0000000000000..dd5fe8d50e40d --- /dev/null +++ b/datafusion/physical-plan/src/aggregates/aggregate_argument.rs @@ -0,0 +1,234 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Evaluation of aggregate function arguments against input batches. + +use std::sync::Arc; + +use crate::PhysicalExpr; +use arrow::array::{ArrayRef, BooleanArray}; +use arrow::record_batch::RecordBatch; +use datafusion_common::{Result, ScalarValue}; +use datafusion_physical_expr::expressions::Literal; + +/// Largest literal array, in bytes, that an [`AggregateArgument`] keeps for +/// later batches. +/// +/// The kept array lives as long as the aggregate stream and is not tracked by +/// a memory pool, so larger arrays are rebuilt for every batch instead. +const MAX_KEPT_LITERAL_ARRAY_BYTES: usize = 1024 * 1024; + +/// One argument of an aggregate function, such as `x` in `SUM(x)`. +/// +/// Accumulators take one value per input row, so an argument that evaluates to +/// a single value must be expanded to an array for every batch. A literal +/// argument, such as the `1` in `COUNT(1)` (which is how `COUNT(*)` is planned) +/// or the `','` in `STRING_AGG(x, ',')`, has the same value for every batch of +/// the query. Its array is built once, and later batches receive a zero-copy +/// slice of it. +/// +/// Only literals are treated this way. Other expressions are evaluated against +/// every batch: a [`ColumnarValue::Scalar`] result holds one value for the rows +/// of a single batch, and the next batch may produce a different value. +/// +/// [`ColumnarValue::Scalar`]: datafusion_expr::ColumnarValue::Scalar +pub(in crate::aggregates) struct AggregateArgument { + expr: Arc, + /// Set when `expr` is a [`Literal`]. + literal: Option, +} + +impl AggregateArgument { + pub(in crate::aggregates) fn new(expr: Arc) -> Self { + let literal = expr + .downcast_ref::() + .map(|literal| LiteralArray::new(literal.value().clone())); + Self { expr, literal } + } + + pub(in crate::aggregates) fn expr(&self) -> &Arc { + &self.expr + } + + /// Evaluates the argument against `batch`, returning one value per row. + pub(in crate::aggregates) fn evaluate( + &mut self, + batch: &RecordBatch, + ) -> Result { + match &mut self.literal { + Some(literal) => literal.array_of_size(batch.num_rows()), + None => self + .expr + .evaluate(batch)? + .into_array_of_size(batch.num_rows()), + } + } + + /// Evaluates the argument for the rows of `batch` where `selection` is + /// true, returning one value per row of `batch`. + /// + /// Rows where `selection` is not true may hold any value, because the + /// selection is also passed to [`GroupsAccumulator::convert_to_state`], + /// which ignores them. + /// + /// [`GroupsAccumulator::convert_to_state`]: datafusion_expr::GroupsAccumulator::convert_to_state + pub(in crate::aggregates) fn evaluate_selection( + &mut self, + batch: &RecordBatch, + selection: &BooleanArray, + ) -> Result { + match &mut self.literal { + Some(literal) if selection.has_true() => { + literal.array_of_size(batch.num_rows()) + } + _ => self + .expr + .evaluate_selection(batch, selection)? + .into_array_of_size(batch.num_rows()), + } + } +} + +/// A literal value and the longest array built from it so far. +struct LiteralArray { + value: ScalarValue, + /// `None` until the first array is built, or while every array built has + /// been larger than [`MAX_KEPT_LITERAL_ARRAY_BYTES`]. + array: Option, +} + +impl LiteralArray { + fn new(value: ScalarValue) -> Self { + Self { value, array: None } + } + + fn array_of_size(&mut self, num_rows: usize) -> Result { + if let Some(array) = &self.array + && array.len() >= num_rows + { + return Ok(array.slice(0, num_rows)); + } + + let array = self.value.to_array_of_size(num_rows)?; + if array.get_array_memory_size() <= MAX_KEPT_LITERAL_ARRAY_BYTES { + self.array = Some(Arc::clone(&array)); + } + Ok(array) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + use arrow::array::{AsArray, Int32Array}; + use arrow::datatypes::{DataType, Field, Int32Type, Schema}; + use datafusion_physical_expr::expressions::{col, lit}; + + fn batch(num_rows: usize) -> RecordBatch { + let schema = Arc::new(Schema::new(vec![Field::new("a", DataType::Int32, true)])); + let values = Int32Array::from_iter_values(0..num_rows as i32); + RecordBatch::try_new(schema, vec![Arc::new(values)]).unwrap() + } + + fn values_ptr(array: &ArrayRef) -> *const i32 { + array.as_primitive::().values().as_ptr() + } + + #[test] + fn literal_array_is_reused_across_batches() -> Result<()> { + let mut argument = AggregateArgument::new(lit(1i32)); + + let first = argument.evaluate(&batch(4))?; + let smaller = argument.evaluate(&batch(3))?; + assert_eq!( + smaller.as_primitive::(), + &Int32Array::from(vec![1; 3]) + ); + assert_eq!(values_ptr(&smaller), values_ptr(&first)); + + // A larger batch builds a longer array, which serves later batches + let larger = argument.evaluate(&batch(6))?; + assert_eq!( + larger.as_primitive::(), + &Int32Array::from(vec![1; 6]) + ); + let after_larger = argument.evaluate(&batch(5))?; + assert_eq!(values_ptr(&after_larger), values_ptr(&larger)); + + assert!(argument.evaluate(&batch(0))?.is_empty()); + Ok(()) + } + + #[test] + fn large_literal_array_is_not_kept() -> Result<()> { + let mut argument = AggregateArgument::new(lit("x".repeat(1024))); + + let first = argument.evaluate(&batch(2048))?; + let second = argument.evaluate(&batch(2048))?; + assert_eq!(first.as_ref(), second.as_ref()); + assert_ne!( + first.as_string::().values().as_ptr(), + second.as_string::().values().as_ptr() + ); + Ok(()) + } + + #[test] + fn non_literal_is_evaluated_for_every_batch() -> Result<()> { + let schema = batch(0).schema(); + let mut argument = AggregateArgument::new(col("a", &schema)?); + + let first = argument.evaluate(&batch(3))?; + assert_eq!( + first.as_primitive::(), + &Int32Array::from(vec![0, 1, 2]) + ); + let second = argument.evaluate(&batch(2))?; + assert_eq!( + second.as_primitive::(), + &Int32Array::from(vec![0, 1]) + ); + Ok(()) + } + + #[test] + fn literal_selection_matches_evaluate_selection() -> Result<()> { + let batch = batch(3); + let expr = lit(7i32); + let mut argument = AggregateArgument::new(Arc::clone(&expr)); + + for selection in [ + BooleanArray::from(vec![true, true, true]), + BooleanArray::from(vec![false, true, false]), + BooleanArray::from(vec![Some(false), None, Some(true)]), + BooleanArray::from(vec![false, false, false]), + BooleanArray::from(vec![None, Some(false), None]), + ] { + let expected = expr + .evaluate_selection(&batch, &selection)? + .into_array(batch.num_rows())?; + let actual = argument.evaluate_selection(&batch, &selection)?; + assert_eq!( + actual.as_ref(), + expected.as_ref(), + "selection {selection:?}" + ); + } + Ok(()) + } +} diff --git a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common.rs b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common.rs index 3a217c9bec14b..8ad11ddd3f4e3 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_hash_table/common.rs @@ -31,10 +31,10 @@ use datafusion_execution::memory_pool::proxy::VecAllocExt; use datafusion_expr::{AggregateMetrics, EmitTo, GroupsAccumulator}; use datafusion_physical_expr::GroupsAccumulatorAdapter; use datafusion_physical_expr::aggregate::AggregateFunctionExpr; -use datafusion_physical_expr_common::utils::ScalarArrayCache; use log::debug; use crate::PhysicalExpr; +use crate::aggregates::aggregate_argument::AggregateArgument; use crate::aggregates::group_values::{ AccumulatorPhase, AggregateAccumulatorMetrics, AggregateArgumentMetrics, GroupByMetrics, GroupValues, new_group_values, @@ -527,11 +527,8 @@ pub(super) struct HashAggregateAccumulator { /// Arguments to pass to this accumulator. /// - /// Example: `CORR(x, y)` stores two expressions here, while `SUM(x)` stores one. - arguments: Vec>, - - /// One cache per aggregate argument. - argument_caches: Vec, + /// Example: `CORR(x, y)` stores two arguments here, while `SUM(x)` stores one. + arguments: Vec, /// Optional `FILTER` expression for this accumulator. /// @@ -722,14 +719,9 @@ impl HashAggregateAccumulator { accumulator: Box, submetrics: Arc, ) -> Self { - let argument_caches = arguments - .iter() - .map(|_| ScalarArrayCache::default()) - .collect(); Self { aggregate_expr, - arguments, - argument_caches, + arguments: arguments.into_iter().map(AggregateArgument::new).collect(), filter, accumulator, submetrics, @@ -743,7 +735,10 @@ impl HashAggregateAccumulator { create_group_accumulator(&self.aggregate_expr, Arc::clone(&self.submetrics))?; Ok(Self::new( Arc::clone(&self.aggregate_expr), - self.arguments.clone(), + self.arguments + .iter() + .map(|argument| Arc::clone(argument.expr())) + .collect(), self.filter.clone(), accumulator, Arc::clone(&self.submetrics), @@ -780,15 +775,13 @@ impl HashAggregateAccumulator { }; let arguments = self .arguments - .iter() - .zip(&mut self.argument_caches) - .map(|(expr, cache)| { + .iter_mut() + .map(|argument| { if let Some(argument_batch) = argument_batch { - expr.evaluate(argument_batch).and_then(|value| { - cache.into_array_of_size(value, argument_batch.num_rows()) - }) + argument.evaluate(argument_batch) } else { - let data_type = expr.data_type(batch.schema_ref().as_ref())?; + let data_type = + argument.expr().data_type(batch.schema_ref().as_ref())?; Ok(new_empty_array(&data_type)) } }) @@ -813,15 +806,10 @@ impl HashAggregateAccumulator { let selection = filter.as_ref(); let arguments = self .arguments - .iter() - .zip(&mut self.argument_caches) - .map(|(expr, cache)| { - selection - .map_or_else( - || expr.evaluate(batch), - |selection| expr.evaluate_selection(batch, selection), - ) - .and_then(|value| cache.into_array_of_size(value, batch.num_rows())) + .iter_mut() + .map(|argument| match selection { + Some(selection) => argument.evaluate_selection(batch, selection), + None => argument.evaluate(batch), }) .collect::>()?; @@ -911,8 +899,8 @@ impl HashAggregateAccumulator { ) -> Result> { self.arguments .iter() - .map(|expr| { - let data_type = expr.data_type(input_schema)?; + .map(|argument| { + let data_type = argument.expr().data_type(input_schema)?; Ok(new_null_array(&data_type, num_rows)) }) .collect() diff --git a/datafusion/physical-plan/src/aggregates/aggregate_stream.rs b/datafusion/physical-plan/src/aggregates/aggregate_stream.rs index b19b52ac95513..fb24c8c313cf4 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_stream.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_stream.rs @@ -17,6 +17,7 @@ //! Aggregate without grouping columns +use crate::aggregates::aggregate_argument::AggregateArgument; use crate::aggregates::group_values::{ AccumulatorPhase, AggregateAccumulatorMetrics, AggregateArgumentMetrics, aggregate_sub_metrics, @@ -47,9 +48,6 @@ use std::task::{Context, Poll}; use super::AggregateExec; use crate::filter::batch_filter; use datafusion_execution::memory_pool::{MemoryConsumer, MemoryReservation}; -use datafusion_physical_expr_common::utils::{ - ScalarArrayCache, evaluate_expressions_to_arrays_with_cache, -}; use futures::stream::{Stream, StreamExt}; /// stream struct for aggregation without grouping columns @@ -70,8 +68,7 @@ struct AggregateStreamInner { schema: SchemaRef, mode: AggregateMode, input: SendableRecordBatchStream, - aggregate_expressions: Vec>>, - aggregate_argument_caches: Vec>, + aggregate_arguments: Vec>, filter_expressions: Arc<[Option>]>, aggregate_argument_metrics: AggregateArgumentMetrics, aggregate_accumulator_metrics: AggregateAccumulatorMetrics, @@ -125,8 +122,8 @@ impl AggregateStreamInner { guard.clone() }; - let agg_exprs = self - .aggregate_expressions + let agg_args = self + .aggregate_arguments .get(acc_info.aggr_index) .ok_or_else(|| { internal_datafusion_err!( @@ -135,12 +132,15 @@ impl AggregateStreamInner { ) })?; // Only aggregates with a single argument are supported. - let column_expr = agg_exprs.first().ok_or_else(|| { - internal_datafusion_err!( - "Aggregate expression at index {} expected a single argument", - acc_info.aggr_index - ) - })?; + let column_expr = agg_args + .first() + .ok_or_else(|| { + internal_datafusion_err!( + "Aggregate expression at index {} expected a single argument", + acc_info.aggr_index + ) + })? + .expr(); let literal = lit(bound); let predicate: Arc = match acc_info.aggr_type { @@ -303,10 +303,9 @@ impl AggregateStream { let baseline_metrics = BaselineMetrics::new(&agg.metrics, partition); let input = agg.input.execute(partition, Arc::clone(context))?; - let aggregate_expressions = aggregate_expressions(agg.aggr_expr(), &agg.mode, 0)?; - let aggregate_argument_caches = aggregate_expressions - .iter() - .map(|exprs| exprs.iter().map(|_| ScalarArrayCache::default()).collect()) + let aggregate_arguments = aggregate_expressions(agg.aggr_expr(), &agg.mode, 0)? + .into_iter() + .map(|exprs| exprs.into_iter().map(AggregateArgument::new).collect()) .collect(); let filter_expressions = match agg.mode.input_mode() { AggregateInputMode::Raw => agg_filter_expr, @@ -368,8 +367,7 @@ impl AggregateStream { mode: agg.mode, input, baseline_metrics, - aggregate_expressions, - aggregate_argument_caches, + aggregate_arguments, filter_expressions, aggregate_argument_metrics, aggregate_accumulator_metrics, @@ -394,8 +392,7 @@ impl AggregateStream { &this.mode, &batch, &mut this.accumulators, - &this.aggregate_expressions, - &mut this.aggregate_argument_caches, + &mut this.aggregate_arguments, &this.filter_expressions, &this.aggregate_argument_metrics, &this.aggregate_accumulator_metrics, @@ -481,13 +478,11 @@ impl RecordBatchStream for AggregateStream { /// If successful, this returns the additional number of bytes that were allocated during this process. /// /// TODO: Make this a member function -#[expect(clippy::too_many_arguments)] fn aggregate_batch( mode: &AggregateMode, batch: &RecordBatch, accumulators: &mut [AccumulatorItem], - expressions: &[Vec>], - argument_caches: &mut [Vec], + arguments: &mut [Vec], filters: &[Option>], aggregate_argument_metrics: &AggregateArgumentMetrics, aggregate_accumulator_metrics: &AggregateAccumulatorMetrics, @@ -502,18 +497,20 @@ fn aggregate_batch( // 1.1 accumulators .iter_mut() - .zip(expressions) - .zip(argument_caches) + .zip(arguments) .zip(filters) .enumerate() - .try_for_each(|(index, (((accum, expr), caches), filter))| { + .try_for_each(|(index, ((accum, arguments), filter))| { // 1.2 and 1.3 let values = aggregate_argument_metrics.time(index, || { let batch = match filter { Some(filter) => Cow::Owned(batch_filter(batch, filter)?), None => Cow::Borrowed(batch), }; - evaluate_expressions_to_arrays_with_cache(expr, caches, batch.as_ref()) + arguments + .iter_mut() + .map(|argument| argument.evaluate(batch.as_ref())) + .collect::>>() })?; // 1.4 diff --git a/datafusion/physical-plan/src/aggregates/mod.rs b/datafusion/physical-plan/src/aggregates/mod.rs index 6552bd95e06a5..b6f48a9967b3a 100644 --- a/datafusion/physical-plan/src/aggregates/mod.rs +++ b/datafusion/physical-plan/src/aggregates/mod.rs @@ -210,6 +210,7 @@ use itertools::Itertools; use topk::hash_table::is_supported_hash_key_type; use topk::heap::is_supported_heap_type; +mod aggregate_argument; mod aggregate_hash_table; mod aggregate_stream; pub mod group_values; From 0c8fad6fe6c70738eadfb22dee685da260e7667f Mon Sep 17 00:00:00 2001 From: wudidapaopao <664920313@qq.com> Date: Thu, 8 Oct 2026 16:12:44 +0800 Subject: [PATCH 6/6] docs: clarify literal array reuse --- .../physical-plan/src/aggregates/aggregate_argument.rs | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/datafusion/physical-plan/src/aggregates/aggregate_argument.rs b/datafusion/physical-plan/src/aggregates/aggregate_argument.rs index dd5fe8d50e40d..56ca9e283754b 100644 --- a/datafusion/physical-plan/src/aggregates/aggregate_argument.rs +++ b/datafusion/physical-plan/src/aggregates/aggregate_argument.rs @@ -38,8 +38,8 @@ const MAX_KEPT_LITERAL_ARRAY_BYTES: usize = 1024 * 1024; /// a single value must be expanded to an array for every batch. A literal /// argument, such as the `1` in `COUNT(1)` (which is how `COUNT(*)` is planned) /// or the `','` in `STRING_AGG(x, ',')`, has the same value for every batch of -/// the query. Its array is built once, and later batches receive a zero-copy -/// slice of it. +/// the query. Its array is built on demand and reused for later batches, which +/// receive a zero-copy slice of it. /// /// Only literals are treated this way. Other expressions are evaluated against /// every batch: a [`ColumnarValue::Scalar`] result holds one value for the rows @@ -103,7 +103,7 @@ impl AggregateArgument { } } -/// A literal value and the longest array built from it so far. +/// A literal value and the largest reusable array retained so far. struct LiteralArray { value: ScalarValue, /// `None` until the first array is built, or while every array built has