diff --git a/Cargo.toml b/Cargo.toml index 5b8b06972..8a6f9a648 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -18,6 +18,11 @@ license = "Apache-2.0" documentation = "https://datafusion-contrib.github.io/datafusion-distributed/" repository = "https://github.com/datafusion-contrib/datafusion-distributed" +[[test]] +name = "dynamic_filtering" +path = "tests/dynamic_filtering/main.rs" +required-features = ["integration"] + [dependencies] chrono = { version = "0.4.44" } datafusion = { workspace = true, features = [ diff --git a/docs/source/user-guide/05-metrics.md b/docs/source/user-guide/05-metrics.md index 98d4f1440..1bdb2defd 100644 --- a/docs/source/user-guide/05-metrics.md +++ b/docs/source/user-guide/05-metrics.md @@ -26,8 +26,12 @@ channel, so they are not lost even if the result stream is dropped early (for ex ## Rendering a plan with metrics -Two functions, both exported from the crate root, do the work: +These functions, all exported from the crate root, do the work: +- `rewrite_distributed_plan_with_dynamic_filters(plan, task_ctx)` — folds the completed dynamic + filters reported by each worker task into an isolated copy of the plan. Pass the same task context + used to execute the plan. When displaying both dynamic filters and metrics, apply the + dynamic-filter rewrite first. - `rewrite_distributed_plan_with_metrics(plan, format)` — folds every task's metrics back into the coordinator's copy of the plan. It waits for all worker metrics to arrive, so the result is always complete. The `format` is a `DistributedMetricsFormat`: @@ -42,23 +46,29 @@ The order of operations matters: **the plan must be fully executed before its me ```rust use datafusion::physical_plan::execute_stream; use datafusion_distributed::{ - DistributedMetricsFormat, display_plan_ascii, rewrite_distributed_plan_with_metrics, + DistributedMetricsFormat, display_plan_ascii, + rewrite_distributed_plan_with_dynamic_filters, rewrite_distributed_plan_with_metrics, }; use futures::TryStreamExt; // 1. Plan the query. let plan = ctx.sql(sql).await?.create_physical_plan().await?; +let task_ctx = ctx.task_ctx(); // 2. Execute it to completion (collect, or otherwise fully drain the stream). -execute_stream(plan.clone(), ctx.task_ctx())? +execute_stream(plan.clone(), task_ctx.clone())? .try_collect::>() .await?; -// 3. Fold the per-task metrics back into the plan... +// 3. Fold the completed per-task dynamic filters back into the plan... +let plan = + rewrite_distributed_plan_with_dynamic_filters(plan, &task_ctx).await?; + +// 4. Fold the per-task metrics back into the plan... let plan = rewrite_distributed_plan_with_metrics(plan, DistributedMetricsFormat::Aggregated).await?; -// 4. ...and render it. +// 5. ...and render it. println!("{}", display_plan_ascii(plan.as_ref(), true)); ``` diff --git a/src/codec/mod.rs b/src/codec/mod.rs index 020937ded..fc03f09fb 100644 --- a/src/codec/mod.rs +++ b/src/codec/mod.rs @@ -4,7 +4,8 @@ mod user_codec; pub use distributed_codec::DistributedCodec; pub(crate) use physical_plan::{ - decode_execution_plan, decode_partitioning, encode_execution_plan, encode_partitioning, + decode_execution_plan, decode_partitioning, decode_physical_expr, encode_execution_plan, + encode_partitioning, encode_physical_expr, roundtrip_pb, }; pub(crate) use user_codec::{ get_distributed_user_codecs, set_distributed_user_codec, set_distributed_user_codec_arc, diff --git a/src/codec/physical_plan.rs b/src/codec/physical_plan.rs index 6ff9836c6..632178cc3 100644 --- a/src/codec/physical_plan.rs +++ b/src/codec/physical_plan.rs @@ -1,15 +1,17 @@ use super::DistributedCodec; -use datafusion::arrow::datatypes::SchemaRef; +use datafusion::arrow::datatypes::{Schema, SchemaRef}; use datafusion::common::Result; use datafusion::execution::TaskContext; -use datafusion::physical_expr::Partitioning; +use datafusion::physical_expr::{Partitioning, PhysicalExpr}; use datafusion::physical_plan::ExecutionPlan; use datafusion_proto::bytes::{ physical_plan_from_bytes_with_proto_converter, physical_plan_to_bytes_with_proto_converter, }; use datafusion_proto::physical_plan::from_proto::parse_protobuf_partitioning; use datafusion_proto::physical_plan::to_proto::serialize_partitioning; -use datafusion_proto::physical_plan::{DeduplicatingProtoConverter, PhysicalPlanDecodeContext}; +use datafusion_proto::physical_plan::{ + DeduplicatingProtoConverter, PhysicalPlanDecodeContext, PhysicalProtoConverterExtension, +}; use datafusion_proto::protobuf; use datafusion_proto::protobuf::proto_error; use prost::Message; @@ -42,6 +44,35 @@ pub(crate) fn decode_execution_plan( physical_plan_from_bytes_with_proto_converter(encoded, task_ctx, &codec, &converter) } +/// Round-trips a plan through protobuf to produce an independent [`ExecutionPlan`]. +pub(crate) fn roundtrip_pb( + plan: Arc, + task_ctx: &TaskContext, +) -> Result> { + let encoded = encode_execution_plan(plan, task_ctx)?; + decode_execution_plan(&encoded, task_ctx) +} + +pub(crate) fn encode_physical_expr( + expression: &Arc, + task_ctx: &TaskContext, +) -> Result { + let codec = DistributedCodec::new_combined_with_user(task_ctx.session_config()); + let converter = new_proto_converter(); + converter.physical_expr_to_proto(expression, &codec) +} + +pub(crate) fn decode_physical_expr( + proto: &protobuf::PhysicalExprNode, + input_schema: &Schema, + task_ctx: &TaskContext, +) -> Result> { + let codec = DistributedCodec::new_combined_with_user(task_ctx.session_config()); + let decode_ctx = PhysicalPlanDecodeContext::new(task_ctx, &codec); + let converter = new_proto_converter(); + converter.proto_to_physical_expr(proto, input_schema, &decode_ctx) +} + pub(crate) fn encode_partitioning( partitioning: &Partitioning, task_ctx: &TaskContext, diff --git a/src/common/maybe_encoded.rs b/src/common/maybe_encoded.rs index faef314c1..7e01ad2b3 100644 --- a/src/common/maybe_encoded.rs +++ b/src/common/maybe_encoded.rs @@ -1,17 +1,21 @@ use crate::codec::{ - decode_execution_plan, decode_partitioning, encode_execution_plan, encode_partitioning, + decode_execution_plan, decode_partitioning, decode_physical_expr, encode_execution_plan, + encode_partitioning, encode_physical_expr, }; -use datafusion::arrow::datatypes::SchemaRef; +use datafusion::arrow::datatypes::{Schema, SchemaRef}; use datafusion::common::{Result, internal_err}; use datafusion::execution::TaskContext; -use datafusion::physical_expr::Partitioning; +use datafusion::physical_expr::{Partitioning, PhysicalExpr}; use datafusion::physical_plan::ExecutionPlan; +use datafusion_proto::protobuf::PhysicalExprNode; +use datafusion_proto::protobuf::proto_error; +use prost::Message; use std::sync::Arc; /// A value that a transport may either leave encoded or materialize in memory. /// Users are free to pass [MaybeEncoded::Encoded] or [MaybeEncoded::Decoded] at any /// moment and Distributed DataFusion's code will internally know how to handle it. -#[derive(Clone)] +#[derive(Clone, Debug)] pub enum MaybeEncoded { Encoded(Vec), Decoded(T), @@ -82,6 +86,44 @@ impl MaybeEncoded { } } +impl MaybeEncoded> { + /// Returns the encoded [`PhysicalExpr`] as protobuf bytes: + /// - If in `Decoded` state, it encodes it using the codecs registered in the [`TaskContext`]. + /// - If in `Encoded` state, it passes through the existing bytes. + pub fn encode(self, ctx: &Arc) -> Result> { + match self { + Self::Encoded(encoded) => Ok(encoded), + Self::Decoded(expression) => { + Ok(encode_physical_expr(&expression, ctx)?.encode_to_vec()) + } + } + } + + /// Returns the decoded [`PhysicalExpr`]. + /// - If in `Decoded` state, it passes through the expression. + /// - If in `Encoded` state, it decodes it using the provided schema and task context. + pub fn decode( + self, + input_schema: &Schema, + task_ctx: &TaskContext, + ) -> Result> { + self.decode_with(|encoded| { + let proto = PhysicalExprNode::decode(encoded.as_slice()) + .map_err(|error| proto_error(error.to_string()))?; + decode_physical_expr(&proto, input_schema, task_ctx) + }) + } + + /// Materializes the expression's protobuf representation without changing the stored form. + pub(crate) fn to_proto(&self, task_ctx: &TaskContext) -> Result { + match self { + Self::Encoded(encoded) => PhysicalExprNode::decode(encoded.as_slice()) + .map_err(|error| proto_error(error.to_string())), + Self::Decoded(expression) => encode_physical_expr(expression, task_ctx), + } + } +} + #[cfg(test)] mod tests { use super::*; diff --git a/src/coordinator/distributed.rs b/src/coordinator/distributed.rs index 1e5fc796f..a22001a8e 100644 --- a/src/coordinator/distributed.rs +++ b/src/coordinator/distributed.rs @@ -1,13 +1,13 @@ use crate::common::require_one_child; -use crate::coordinator::metrics_store::MetricsStore; use crate::coordinator::prepare_dynamic_plan::prepare_dynamic_plan; use crate::coordinator::prepare_static_plan::prepare_static_plan; use crate::coordinator::query_coordinator::QueryCoordinator; -use crate::distributed_planner::NetworkBoundaryExt; -use crate::{DistributedConfig, TaskKey}; +use crate::coordinator::store::{Store, task_keys_for_plan}; +use crate::dynamic_filtering::sever_dynamic_filter_relationships_in_plan_for_display; +use crate::{DistributedConfig, TaskCompletedDynamicFilters, TaskKey, TaskMetrics}; use datafusion::common::internal_datafusion_err; -use datafusion::common::tree_node::{TreeNode, TreeNodeRecursion}; -use datafusion::common::{Result, exec_err}; +use datafusion::common::tree_node::TreeNodeRecursion; +use datafusion::common::{HashMap, Result, exec_err}; use datafusion::execution::{SendableRecordBatchStream, TaskContext}; use datafusion::physical_expr::PhysicalExpr; use datafusion::physical_expr_common::metrics::MetricsSet; @@ -16,7 +16,7 @@ use datafusion::physical_plan::stream::RecordBatchReceiverStreamBuilder; use datafusion::physical_plan::{DisplayAs, DisplayFormatType, ExecutionPlan, PlanProperties}; use futures::StreamExt; use std::fmt::Formatter; -use std::sync::{Arc, Mutex}; +use std::sync::{Arc, OnceLock}; /// [ExecutionPlan] that executes the inner plan in distributed mode. /// Before executing it, two modifications are lazily performed on the plan: @@ -27,30 +27,34 @@ use std::sync::{Arc, Mutex}; /// over the wire. #[derive(Debug)] pub struct DistributedExec { - /// Initial [ExecutionPlan] present before execution. + /// [ExecutionPlan] exposed through [`ExecutionPlan::children`] and used as the input to + /// execution. + /// + /// Initially, this is the plan present before execution: /// - If the plan was distributed statically, this will be the final distributed plan with all /// the appropriate network boundaries in it. /// - If the plan is going to be distributed dynamically during execution, this is the initial /// non-distributed plan. + /// + /// Post-execution rewrites replace this plan in the returned clone while leaving the original + /// [`DistributedExec`] unchanged. base_plan: Arc, - /// Resulting [ExecutionPlan] after execution ready for visualization purposes. - /// - If the plan was distributed statically, this is equal to the base plan. - /// - If the plan is going to be distributed dynamically during execution, this is the resulting - /// plan re-calculated based on runtime statistics. - plan_for_viz: Arc>>>, - /// The head stage meant to be executed locally on [DistributedExec::execute]. - head_stage: Arc>>>, + /// Complete plans produced during static or dynamic preparation. + prepared_plan: Arc>, /// DataFusion metrics. metrics: ExecutionPlanMetricsSet, /// Storage where metrics collected from workers at runtime will place their results as they /// finish their respective remote tasks. - pub(crate) metrics_store: Option>, + pub(crate) metrics_store: Option>>, + /// Storage for the completed dynamic filters reported by each worker task. + pub(crate) completed_dynamic_filter_store: Option>>, } +#[derive(Debug, Clone)] pub(super) struct PreparedPlan { - /// The head stage meant to be executed locally by the coordinator. + /// The coordinator-side plan prepared for execution. pub(super) head_stage: Arc, - /// A final representation of the plan for visualization purposes. + /// The complete distributed plan reconstructed for visualization, including all stages. pub(super) plan_for_viz: Arc, } @@ -58,82 +62,95 @@ impl DistributedExec { pub fn new(base_plan: Arc) -> Self { Self { base_plan, - plan_for_viz: Arc::new(Mutex::new(None)), - head_stage: Arc::new(Mutex::new(None)), + prepared_plan: Arc::new(OnceLock::new()), metrics: ExecutionPlanMetricsSet::new(), metrics_store: None, + completed_dynamic_filter_store: None, } } /// Enables task metrics collection from remote workers. pub fn with_metrics_collection(mut self, enabled: bool) -> Self { self.metrics_store = match enabled { - true => Some(Arc::new(MetricsStore::new())), + true => Some(Arc::new(Store::new())), false => None, }; self } - /// Waits until all worker tasks have reported their metrics back via the coordinator channel. - /// - /// Metrics are delivered asynchronously after query execution completes, so callers that need - /// complete metrics (e.g. for observability or display) should await this before inspecting - /// [`Self::task_metrics`] or calling [`rewrite_distributed_plan_with_metrics`]. - /// - /// [`rewrite_distributed_plan_with_metrics`]: crate::rewrite_distributed_plan_with_metrics - pub async fn wait_for_metrics(&self) { - let mut expected_keys: Vec = Vec::new(); - let Some(task_metrics) = &self.metrics_store else { - return; - }; - let Some(plan) = self.plan_for_viz.lock().unwrap().as_ref().cloned() else { - return; + /// Enables collection of completed dynamic filters from remote workers for display. + pub fn with_dynamic_filter_collection(mut self, enabled: bool) -> Self { + self.completed_dynamic_filter_store = match enabled { + true => Some(Arc::new(Store::new())), + false => None, }; - let _ = plan.apply(|plan| { - if let Some(boundary) = plan.as_network_boundary() { - let stage = boundary.input_stage(); - for i in 0..stage.task_count() { - expected_keys.push(TaskKey { - query_id: stage.query_id(), - stage_id: stage.num(), - task_number: i, - }); - } - } - Ok(TreeNodeRecursion::Continue) - }); - if expected_keys.is_empty() { - return; - } - let mut rx = task_metrics.rx.clone(); - let _ = rx - .wait_for(|map| expected_keys.iter().all(|key| map.contains_key(key))) - .await; + self + } + + /// Waits until all worker tasks have reported their metrics back via the coordinator channel + /// if metrics collection is enabled. + pub async fn wait_for_metrics(&self) -> Option> { + let task_metrics = self.metrics_store.as_ref()?; + let plan = &self.prepared_plan.get()?.plan_for_viz; + Some(task_metrics.wait_for(&task_keys_for_plan(plan)).await) + } + + /// Waits until all worker tasks have reported their completed dynamic filters back via + /// the coordinator channel if dynamic filter collection is enabled. + pub(crate) async fn wait_for_dynamic_filters( + &self, + ) -> Option> { + let store = self.completed_dynamic_filter_store.as_ref()?; + let plan = &self.prepared_plan.get()?.plan_for_viz; + Some(store.wait_for(&task_keys_for_plan(plan)).await) + } + + fn prepared_plan(&self) -> Result { + self.prepared_plan.get().cloned().ok_or_else(|| { + internal_datafusion_err!("No prepared plan found. Was execute() called?") + }) } - /// Returns the plan which is lazily prepared on `execute()` and actually gets executed. - /// It is updated on every call to `execute()`. Returns an error if `.execute()` has not been - /// called. + /// Returns the plan reconstructed during preparation for visualization and rewriting. pub(crate) fn plan_for_viz(&self) -> Result> { - self.plan_for_viz - .lock() - .map_err(|e| internal_datafusion_err!("Failed to lock prepared plan: {}", e))? - .clone() - .ok_or_else(|| { - internal_datafusion_err!("No prepared plan found. Was execute() called?") - }) - } - - /// Returns the head stage that was actually executed. Unlike [`Self::plan_for_viz`] (which is - /// reconstructed for visualization, with `Stage::Local` boundaries and rebuilt ancestor - /// `Arc`s), this returns the original `Arc` instances whose metrics were populated during - /// execution. + Ok(self.prepared_plan()?.plan_for_viz) + } + + /// Returns the prepared visualization plan when available, or the original optimized plan + /// before execution has prepared one. + pub(crate) fn plan_for_viz_or_base_plan(&self) -> Arc { + self.prepared_plan + .get() + .map(|prepared| Arc::clone(&prepared.plan_for_viz)) + .unwrap_or_else(|| Arc::clone(&self.base_plan)) + } + + /// Returns the coordinator-side plan executed by [`DistributedExec`]. + /// + /// Unlike [`Self::plan_for_viz`], this contains [`Stage::Remote`] boundaries instead of the + /// remote execution-plan nodes. It also retains the original plan-node instances whose + /// metrics were populated during execution. + /// + /// [`Stage::Remote`]: crate::stage::Stage::Remote pub(crate) fn head_stage(&self) -> Result> { - self.head_stage - .lock() - .map_err(|e| internal_datafusion_err!("Failed to lock head stage: {}", e))? - .clone() - .ok_or_else(|| internal_datafusion_err!("No head stage found. Was execute() called?")) + Ok(self.prepared_plan()?.head_stage) + } + + /// Returns a new [`DistributedExec`] with an updated visualization plan while leaving its + /// public child unchanged. + pub(crate) fn with_plan_for_viz( + &self, + plan_for_viz: Arc, + ) -> Result> { + let mut prepared_plan = self.prepared_plan()?; + prepared_plan.plan_for_viz = plan_for_viz; + Ok(Arc::new(Self { + base_plan: Arc::clone(&self.base_plan), + prepared_plan: Arc::new(OnceLock::from(prepared_plan)), + metrics: self.metrics.clone(), + metrics_store: self.metrics_store.clone(), + completed_dynamic_filter_store: self.completed_dynamic_filter_store.clone(), + })) } } @@ -167,12 +184,20 @@ impl ExecutionPlan for DistributedExec { self: Arc, children: Vec>, ) -> Result> { + let child = require_one_child(&children)?; + // Replacing the public child is independent from replacing the visualization plan. A + // post-execution rewrite updates the latter explicitly via `Self::with_plan_for_viz`. + let prepared_plan = self + .prepared_plan + .get() + .cloned() + .map_or_else(OnceLock::new, OnceLock::from); Ok(Arc::new(DistributedExec { - base_plan: require_one_child(&children)?, - plan_for_viz: Arc::new(Mutex::new(None)), - head_stage: Arc::new(Mutex::new(None)), + base_plan: child, + prepared_plan: Arc::new(prepared_plan), metrics: self.metrics.clone(), metrics_store: self.metrics_store.clone(), + completed_dynamic_filter_store: self.completed_dynamic_filter_store.clone(), })) } @@ -192,13 +217,14 @@ impl ExecutionPlan for DistributedExec { } let base_plan = Arc::clone(&self.base_plan); - let plan_for_viz = Arc::clone(&self.plan_for_viz); - let head_stage = Arc::clone(&self.head_stage); + let prepared_plan = Arc::clone(&self.prepared_plan); + let collect_dynamic_filters = self.completed_dynamic_filter_store.is_some(); let query_coordinator = QueryCoordinator::new( Arc::clone(&context), &self.metrics, self.metrics_store.clone(), + self.completed_dynamic_filter_store.clone(), ); let mut builder = RecordBatchReceiverStreamBuilder::new(self.schema(), 1); @@ -218,27 +244,30 @@ impl ExecutionPlan for DistributedExec { let guard = query_coordinator.end_query_guard(); let d_cfg = DistributedConfig::from_config_options(context.session_config().options())?; - let result = match d_cfg.dynamic_task_count { + let mut prepared = match d_cfg.dynamic_task_count { true => prepare_dynamic_plan(&query_coordinator, &base_plan).await?, false => prepare_static_plan(&query_coordinator, &base_plan)?, }; - plan_for_viz - .lock() - .expect("poisoned lock") - .replace(result.plan_for_viz); - head_stage - .lock() - .expect("poisoned lock") - .replace(Arc::clone(&result.head_stage)); - let mut stream = result.head_stage.execute(partition, context)?; + prepared.plan_for_viz = match collect_dynamic_filters { + true => sever_dynamic_filter_relationships_in_plan_for_display( + prepared.plan_for_viz, + &context, + )?, + false => prepared.plan_for_viz, + }; + let head_stage = Arc::clone(&prepared.head_stage); + prepared_plan.set(prepared).map_err(|_| { + internal_datafusion_err!("DistributedExec was already prepared for execution") + })?; + let mut stream = head_stage.execute(partition, context)?; while let Some(msg) = stream.next().await { if tx.send(msg).await.is_err() { break; // channel closed } } - drop(tx); drop(guard); + drop(tx); query_coordinator.drain_pending_tasks().await?; Ok(()) }); diff --git a/src/coordinator/metrics_store.rs b/src/coordinator/metrics_store.rs deleted file mode 100644 index 8ccc1f7c0..000000000 --- a/src/coordinator/metrics_store.rs +++ /dev/null @@ -1,36 +0,0 @@ -use crate::{TaskKey, TaskMetrics}; -use datafusion::common::HashMap; -use tokio::sync::watch; - -type MetricsMap = HashMap; - -/// Stores the metrics collected from all worker tasks, and notifies waiters when new entries arrive. -#[derive(Debug, Clone)] -pub struct MetricsStore { - tx: watch::Sender, - pub(crate) rx: watch::Receiver, -} - -impl MetricsStore { - pub(crate) fn new() -> Self { - let (tx, rx) = watch::channel(HashMap::new()); - Self { tx, rx } - } - - pub(crate) fn insert(&self, key: TaskKey, metrics: TaskMetrics) { - self.tx.send_modify(|map| { - map.insert(key, metrics); - }); - } - - pub(crate) fn get(&self, key: &TaskKey) -> Option { - self.rx.borrow().get(key).cloned() - } - - #[cfg(test)] - pub(crate) fn from_entries(entries: impl IntoIterator) -> Self { - let map: HashMap<_, _> = entries.into_iter().collect(); - let (tx, rx) = watch::channel(map); - Self { tx, rx } - } -} diff --git a/src/coordinator/mod.rs b/src/coordinator/mod.rs index c1a8a8dd2..fd7706fda 100644 --- a/src/coordinator/mod.rs +++ b/src/coordinator/mod.rs @@ -1,9 +1,9 @@ mod distributed; mod latency_metric; -mod metrics_store; mod prepare_dynamic_plan; mod prepare_static_plan; mod query_coordinator; +mod store; pub use distributed::DistributedExec; -pub(crate) use metrics_store::MetricsStore; +pub(crate) use store::Store; diff --git a/src/coordinator/query_coordinator.rs b/src/coordinator/query_coordinator.rs index 2f9c99fb0..d28061a02 100644 --- a/src/coordinator/query_coordinator.rs +++ b/src/coordinator/query_coordinator.rs @@ -1,8 +1,9 @@ -use crate::codec::{decode_execution_plan, encode_execution_plan}; +use crate::codec::roundtrip_pb; use crate::common::{TreeNodeExt, now_ns, task_ctx_with_extension}; use crate::config_extension_ext::get_config_extension_propagation_headers; -use crate::coordinator::MetricsStore; +use crate::coordinator::Store; use crate::coordinator::latency_metric::LatencyMetric; +use crate::dynamic_filtering::maybe_roundtrip_plan_to_sever_in_memory_dynamic_filter_relationships; use crate::events::{RouteTasksEvent, RouteTasksHandlers}; use crate::execution_plans::{ChildrenIsolatorUnionExec, DistributedLeafExec}; use crate::passthrough_headers::get_passthrough_headers; @@ -12,7 +13,8 @@ use crate::work_unit_feed::{build_work_unit_batch_msg, set_work_unit_send_time}; use crate::{ CoordinatorToWorkerMsg, DISTRIBUTED_DATAFUSION_TASK_ID_LABEL, DistributedTaskContext, DistributedWorkUnitFeedContext, LoadInfo, LocalWorkerContext, MaybeEncoded, SetPlanRequest, - TaskKey, WorkUnitFeedDeclaration, WorkerToCoordinatorMsg, get_distributed_channel_resolver, + TaskCompletedDynamicFilters, TaskKey, TaskMetrics, WorkUnitFeedDeclaration, + WorkerToCoordinatorMsg, get_distributed_channel_resolver, }; use datafusion::common::DataFusionError; use datafusion::common::instant::Instant; @@ -46,7 +48,8 @@ pub(super) struct QueryCoordinator { task_ctx: Arc, metrics: ExecutionPlanMetricsSet, coordinator_to_worker_metrics: CoordinatorToWorkerMetrics, - metrics_store: Option>, + metrics_store: Option>>, + completed_dynamic_filter_store: Option>>, end_stream_notifier: Arc, join_set: Mutex>>, } @@ -56,12 +59,14 @@ impl QueryCoordinator { pub(super) fn new( task_ctx: Arc, metrics_set: &ExecutionPlanMetricsSet, - metrics_store: Option>, + metrics_store: Option>>, + completed_dynamic_filter_store: Option>>, ) -> Self { Self { task_ctx, metrics: metrics_set.clone(), metrics_store, + completed_dynamic_filter_store, coordinator_to_worker_metrics: CoordinatorToWorkerMetrics::new(metrics_set), end_stream_notifier: Arc::new(Notify::new()), join_set: Mutex::new(JoinSet::new()), @@ -80,6 +85,7 @@ impl QueryCoordinator { metrics_set: &self.metrics, metrics: &self.coordinator_to_worker_metrics, metrics_store: &self.metrics_store, + completed_dynamic_filter_store: &self.completed_dynamic_filter_store, end_stream_notifier: &self.end_stream_notifier, join_set: &self.join_set, } @@ -124,7 +130,8 @@ pub(super) struct StageCoordinator<'a> { task_ctx: &'a Arc, metrics_set: &'a ExecutionPlanMetricsSet, metrics: &'a CoordinatorToWorkerMetrics, - metrics_store: &'a Option>, + metrics_store: &'a Option>>, + completed_dynamic_filter_store: &'a Option>>, end_stream_notifier: &'a Arc, join_set: &'a Mutex>>, } @@ -150,7 +157,6 @@ impl<'a> StageCoordinator<'a> { stage_id: self.stage_id, task_number: task_i, }; - let set_plan_request = SetPlanRequest { task_key, task_count: self.task_count, @@ -178,8 +184,8 @@ impl<'a> StageCoordinator<'a> { // 3. Here, `end_stream_notifier` fires and the coordinator->worker channel is // gracefully ended. // 4. The coordinator->worker channel EOS is received in `impl_coordinator_channel.rs`. - // 5. The metrics are send back in the worker->coordinator channel, and then that - // channel is closed. + // 5. The metrics and final dynamic filters are sent back in the + // worker->coordinator channel, and then that channel is closed. .chain(keep_stream_alive(Arc::clone(self.end_stream_notifier))) .boxed(); @@ -236,6 +242,7 @@ impl<'a> StageCoordinator<'a> { task_number: task_i, }; let task_metrics = self.metrics_store.clone(); + let completed_dynamic_filter_store = self.completed_dynamic_filter_store.clone(); let (load_info_tx, load_info_rx) = tokio::sync::mpsc::unbounded_channel(); let mut load_info_tx_opt = Some(load_info_tx); @@ -258,6 +265,11 @@ impl<'a> StageCoordinator<'a> { WorkerToCoordinatorMsg::LoadInfoEos => { let _ = load_info_tx_opt.take(); } + WorkerToCoordinatorMsg::TaskCompletedDynamicFilters(filters) => { + if let Some(store) = &completed_dynamic_filter_store { + store.insert(task_key, filters); + } + } } } }); @@ -368,9 +380,8 @@ impl<'a> StageCoordinator<'a> { // Right now, there's no other way for a WorkUnitFeed to be transitioned to // remote mode so that it can pull WorkUnits over the WorkerChannel. // - // Doing this roundtrip here is not super clean, but it actually does the trick in - // very few LOC, and performance overhead is negligible, as the roundtrip does not - // even imply serialization. + // Doing this roundtrip here is not super clean, but it transitions the feed with + // very little specialized code. let plan = roundtrip_pb(plan, self.task_ctx)?; return Ok(Transformed::yes(plan)); }; @@ -387,7 +398,11 @@ impl<'a> StageCoordinator<'a> { Ok(Transformed::no(plan)) })?; - Ok((transformed.data, work_unit_feed_declarations)) + let plan = maybe_roundtrip_plan_to_sever_in_memory_dynamic_filter_relationships( + Arc::clone(&transformed.data), + self.task_ctx, + )?; + Ok((plan, work_unit_feed_declarations)) } /// Returns as many URLs as the task count for the stage this [StageCoordinator] @@ -416,15 +431,6 @@ impl<'a> StageCoordinator<'a> { } } -/// Round-trips a plan converting it to a protobuf message and back to an [ExecutionPlan]. -fn roundtrip_pb( - plan: Arc, - ctx: &Arc, -) -> Result> { - let encoded = encode_execution_plan(plan, ctx)?; - decode_execution_plan(&encoded, ctx) -} - fn keep_stream_alive(notify: Arc) -> impl Stream + 'static { futures::stream::once(notify.notified_owned()).filter_map(|()| futures::future::ready(None)) } diff --git a/src/coordinator/store.rs b/src/coordinator/store.rs new file mode 100644 index 000000000..806240107 --- /dev/null +++ b/src/coordinator/store.rs @@ -0,0 +1,60 @@ +use crate::TaskKey; +use crate::distributed_planner::NetworkBoundaryExt; +use datafusion::common::HashMap; +use datafusion::common::tree_node::{TreeNode, TreeNodeRecursion}; +use datafusion::physical_plan::ExecutionPlan; +use std::sync::Arc; +use tokio::sync::watch; + +type StoreMap = HashMap; + +/// Stores task-scoped values and notifies waiters when entries change. +#[derive(Debug, Clone)] +pub(crate) struct Store { + tx: watch::Sender>, + rx: watch::Receiver>, +} + +impl Store { + pub(crate) fn new() -> Self { + let (tx, rx) = watch::channel(HashMap::new()); + Self { tx, rx } + } + + pub(crate) fn insert(&self, key: TaskKey, value: T) { + self.tx.send_modify(|map| { + map.insert(key, value); + }); + } + + pub(crate) async fn wait_for(&self, expected_keys: &[TaskKey]) -> StoreMap + where + T: Clone, + { + let mut rx = self.rx.clone(); + if !expected_keys.is_empty() { + let _ = rx + .wait_for(|map| expected_keys.iter().all(|key| map.contains_key(key))) + .await; + } + rx.borrow().clone() + } +} + +pub(crate) fn task_keys_for_plan(plan: &Arc) -> Vec { + let mut task_keys = Vec::new(); + let _ = plan.apply(|plan| { + if let Some(boundary) = plan.as_network_boundary() { + let stage = boundary.input_stage(); + for task_number in 0..stage.task_count() { + task_keys.push(TaskKey { + query_id: stage.query_id(), + stage_id: stage.num(), + task_number, + }); + } + } + Ok(TreeNodeRecursion::Continue) + }); + task_keys +} diff --git a/src/distributed_ext.rs b/src/distributed_ext.rs index d89f4b4e7..8c081e101 100644 --- a/src/distributed_ext.rs +++ b/src/distributed_ext.rs @@ -375,6 +375,20 @@ pub trait DistributedExt: Sized { /// Same as [DistributedExt::with_distributed_metrics_collection] but with an in-place mutation. fn set_distributed_metrics_collection(&mut self, enabled: bool) -> Result<(), DataFusionError>; + /// Collects completed dynamic filters from worker tasks so they can be displayed in the + /// distributed plan. This does not enable or disable dynamic filtering during execution. + fn with_distributed_dynamic_filter_collection( + self, + enabled: bool, + ) -> Result; + + /// Same as [`DistributedExt::with_distributed_dynamic_filter_collection`] but with an in-place + /// mutation. + fn set_distributed_dynamic_filter_collection( + &mut self, + enabled: bool, + ) -> Result<(), DataFusionError>; + /// Enables children isolator unions for distributing UNION operations across as many tasks as /// the sum of all the tasks required for each child. /// @@ -803,6 +817,15 @@ impl DistributedExt for SessionConfig { Ok(()) } + fn set_distributed_dynamic_filter_collection( + &mut self, + enabled: bool, + ) -> Result<(), DataFusionError> { + let d_cfg = DistributedConfig::from_config_options_mut(self.options_mut())?; + d_cfg.collect_dynamic_filters = enabled; + Ok(()) + } + fn set_distributed_children_isolator_unions( &mut self, enabled: bool, @@ -957,6 +980,10 @@ impl DistributedExt for SessionConfig { #[expr($?;Ok(self))] fn with_distributed_metrics_collection(mut self, enabled: bool) -> Result; + #[call(set_distributed_dynamic_filter_collection)] + #[expr($?;Ok(self))] + fn with_distributed_dynamic_filter_collection(mut self, enabled: bool) -> Result; + #[call(set_distributed_children_isolator_unions)] #[expr($?;Ok(self))] fn with_distributed_children_isolator_unions(mut self, enabled: bool) -> Result; @@ -1083,6 +1110,11 @@ impl DistributedExt for SessionStateBuilder { #[expr($?;Ok(self))] fn with_distributed_metrics_collection(mut self, enabled: bool) -> Result; + fn set_distributed_dynamic_filter_collection(&mut self, enabled: bool) -> Result<(), DataFusionError>; + #[call(set_distributed_dynamic_filter_collection)] + #[expr($?;Ok(self))] + fn with_distributed_dynamic_filter_collection(mut self, enabled: bool) -> Result; + fn set_distributed_children_isolator_unions(&mut self, enabled: bool) -> Result<(), DataFusionError>; #[call(set_distributed_children_isolator_unions)] #[expr($?;Ok(self))] @@ -1233,6 +1265,11 @@ impl DistributedExt for SessionState { #[expr($?;Ok(self))] fn with_distributed_metrics_collection(mut self, enabled: bool) -> Result; + fn set_distributed_dynamic_filter_collection(&mut self, enabled: bool) -> Result<(), DataFusionError>; + #[call(set_distributed_dynamic_filter_collection)] + #[expr($?;Ok(self))] + fn with_distributed_dynamic_filter_collection(mut self, enabled: bool) -> Result; + fn set_distributed_children_isolator_unions(&mut self, enabled: bool) -> Result<(), DataFusionError>; #[call(set_distributed_children_isolator_unions)] #[expr($?;Ok(self))] @@ -1376,6 +1413,11 @@ impl DistributedExt for SessionContext { #[expr($?;Ok(self))] fn with_distributed_metrics_collection(self, enabled: bool) -> Result; + fn set_distributed_dynamic_filter_collection(&mut self, enabled: bool) -> Result<(), DataFusionError>; + #[call(set_distributed_dynamic_filter_collection)] + #[expr($?;Ok(self))] + fn with_distributed_dynamic_filter_collection(self, enabled: bool) -> Result; + fn set_distributed_children_isolator_unions(&mut self, enabled: bool) -> Result<(), DataFusionError>; #[call(set_distributed_children_isolator_unions)] #[expr($?;Ok(self))] diff --git a/src/distributed_planner/distributed_config.rs b/src/distributed_planner/distributed_config.rs index fa330456b..e6f3583b5 100644 --- a/src/distributed_planner/distributed_config.rs +++ b/src/distributed_planner/distributed_config.rs @@ -30,6 +30,10 @@ extensions_options! { /// Propagate collected metrics from all nodes in the plan across network boundaries /// so that they can be reconstructed on the head node of the plan. pub collect_metrics: bool, default = true + /// Collect completed dynamic filters from worker tasks so that they can be displayed in + /// the distributed plan. This does not control whether dynamic filtering is used during + /// query execution. + pub collect_dynamic_filters: bool, default = true /// Enable broadcast joins for CollectLeft hash joins. When enabled, the build side of /// a CollectLeft join is broadcast to all consumer tasks. pub broadcast_joins: bool, default = true diff --git a/src/distributed_planner/distributed_query_planner.rs b/src/distributed_planner/distributed_query_planner.rs index d312b3293..c26f8854d 100644 --- a/src/distributed_planner/distributed_query_planner.rs +++ b/src/distributed_planner/distributed_query_planner.rs @@ -131,9 +131,7 @@ fn create_distributed_plan( return Ok(plan); } let plan = push_fetch_into_network_coalesce(plan)?; - return Ok(Arc::new( - DistributedExec::new(plan).with_metrics_collection(d_cfg.collect_metrics), - )); + return Ok(create_distributed_exec(Arc::clone(&plan), d_cfg)); } let mut plan = Arc::clone(&original_plan); @@ -150,9 +148,7 @@ fn create_distributed_plan( if d_cfg.dynamic_task_count { // The task count will be decided dynamically at execution time. - return Ok(Arc::new( - DistributedExec::new(plan).with_metrics_collection(d_cfg.collect_metrics), - )); + return Ok(create_distributed_exec(Arc::clone(&plan), d_cfg)); } // Compute per-node task counts and inject `Network*Exec` nodes at the stage boundaries. @@ -167,12 +163,21 @@ fn create_distributed_plan( let plan = partial_reduce_below_network_shuffles(plan, cfg)?; let plan = push_fetch_into_network_coalesce(plan)?; - Ok(Arc::new( - DistributedExec::new(plan).with_metrics_collection(d_cfg.collect_metrics), - )) + Ok(create_distributed_exec(Arc::clone(&plan), d_cfg)) }) } +fn create_distributed_exec( + plan: Arc, + d_cfg: &DistributedConfig, +) -> Arc { + Arc::new( + DistributedExec::new(plan) + .with_metrics_collection(d_cfg.collect_metrics) + .with_dynamic_filter_collection(d_cfg.collect_dynamic_filters), + ) +} + #[cfg(test)] mod tests { use crate::assert_snapshot; diff --git a/src/dynamic_filtering/discovery.rs b/src/dynamic_filtering/discovery.rs new file mode 100644 index 000000000..e07b052ad --- /dev/null +++ b/src/dynamic_filtering/discovery.rs @@ -0,0 +1,278 @@ +use datafusion::arrow::datatypes::SchemaRef; +use datafusion::common::tree_node::{TreeNode, TreeNodeRecursion}; +use datafusion::common::{HashMap, HashSet, Result, internal_err}; +use datafusion::physical_expr::PhysicalExpr; +use datafusion::physical_expr::expressions::DynamicFilterPhysicalExpr; +use datafusion::physical_plan::ExecutionPlan; +use std::sync::Arc; + +/// A dynamic-filter consumer discovered in an execution plan along with the schema it is evaluated +/// against. +#[derive(Clone)] +pub(crate) struct DiscoveredDynamicFilter { + pub(crate) id: u64, + pub(crate) expression: Arc, + pub(crate) input_schema: SchemaRef, +} + +/// Finds dynamic-filter consumers in `plan`, deduplicated by expression ID. +pub(crate) fn discover_dynamic_filter_consumers( + plan: &Arc, +) -> Result> { + let mut consumers = HashMap::new(); + + plan.apply(|node| { + let produced_ids: HashSet<_> = node + .dynamic_expressions_produced() + .into_iter() + .map(|produced| { + let Some(id) = produced.expression_id() else { + return internal_err!( + "{}::dynamic_expressions_produced returned an expression without an expression ID", + node.name() + ); + }; + Ok(id) + }) + .collect::>()?; + let input_schema = node + .children() + .first() + .map(|child| child.schema()) + .unwrap_or_else(|| node.schema()); + + node.apply_expressions(&mut |root| { + root.apply(|expression| { + let expression = Arc::clone(expression); + let Ok(expression) = Arc::downcast::(expression) else { + return Ok(TreeNodeRecursion::Continue); + }; + + let Some(id) = expression.expression_id() else { + return internal_err!( + "DynamicFilterPhysicalExpr did not have an expression ID" + ); + }; + let is_producer_occurrence = produced_ids.contains(&id); + if !is_producer_occurrence { + consumers + .entry(id) + .or_insert_with(|| DiscoveredDynamicFilter { + id, + expression, + input_schema: Arc::clone(&input_schema), + }); + } + + Ok(TreeNodeRecursion::Continue) + }) + })?; + Ok(TreeNodeRecursion::Continue) + })?; + + let mut consumers: Vec<_> = consumers.into_values().collect(); + consumers.sort_unstable_by_key(|consumer| consumer.id); + Ok(consumers) +} + +/// Returns whether `plan` contains only the consumer side of a dynamic filter +/// relationship. +pub(crate) fn has_nonlocal_dynamic_filter_relationships( + plan: &Arc, +) -> Result { + let consumer_ids: HashSet<_> = discover_dynamic_filter_consumers(plan)? + .into_iter() + .map(|consumer| consumer.id) + .collect(); + + let mut producer_ids = HashSet::new(); + plan.apply(|node| { + for produced in node.dynamic_expressions_produced() { + let Some(id) = produced.expression_id() else { + return internal_err!( + "{}::dynamic_expressions_produced returned an expression without an expression ID", + node.name() + ); + }; + producer_ids.insert(id); + } + Ok(TreeNodeRecursion::Continue) + })?; + + Ok(consumer_ids != producer_ids) +} + +#[cfg(test)] +mod tests { + use super::*; + use datafusion::arrow::datatypes::{DataType, Field, Schema}; + use datafusion::common::Result; + use datafusion::execution::{SendableRecordBatchStream, TaskContext}; + use datafusion::logical_expr::Operator; + use datafusion::physical_expr::expressions::{BinaryExpr, Column, lit}; + use datafusion::physical_plan::empty::EmptyExec; + use datafusion::physical_plan::union::UnionExec; + use datafusion::physical_plan::{ + DisplayAs, DisplayFormatType, PlanProperties, apply_expression_roots, + }; + use std::fmt::Formatter; + + #[tokio::test] + async fn discovers_nested_consumer_but_not_its_producer_occurrence() -> Result<()> { + let schema = Arc::new(Schema::new(vec![Field::new("a", DataType::Int32, false)])); + let input = Arc::new(EmptyExec::new(Arc::clone(&schema))) as Arc; + let column = Arc::new(Column::new("a", 0)) as Arc; + let dynamic_filter = Arc::new(DynamicFilterPhysicalExpr::new( + vec![Arc::clone(&column)], + lit(true), + )) as Arc; + let nested = Arc::new(BinaryExpr::new( + Arc::clone(&dynamic_filter), + Operator::And, + lit(true), + )) as Arc; + + let consumer = + Arc::new(ExpressionExec::new(input, nested, false)) as Arc; + let plan = Arc::new(ExpressionExec::new( + consumer, + Arc::clone(&dynamic_filter), + true, + )) as Arc; + + let discovered = discover_dynamic_filter_consumers(&plan)?; + assert_eq!(discovered.len(), 1); + assert_eq!(discovered[0].id, dynamic_filter.expression_id().unwrap()); + assert!(!has_nonlocal_dynamic_filter_relationships(&plan)?); + + dynamic_filter + .downcast_ref::() + .unwrap() + .update(Arc::new(BinaryExpr::new(column, Operator::Gt, lit(10_i32))))?; + dynamic_filter + .downcast_ref::() + .unwrap() + .mark_complete(); + + let current = discovered[0].expression.current()?; + assert_eq!(current.to_string(), "a@0 > 10"); + Ok(()) + } + + #[test] + fn deduplicates_consumers_with_the_same_expression_id() -> Result<()> { + let schema = Arc::new(Schema::new(vec![Field::new("a", DataType::Int32, false)])); + let dynamic_filter = Arc::new(DynamicFilterPhysicalExpr::new( + vec![Arc::new(Column::new("a", 0))], + lit(true), + )) as Arc; + let consumers = (0..2) + .map(|_| { + Arc::new(ExpressionExec::new( + Arc::new(EmptyExec::new(Arc::clone(&schema))), + Arc::clone(&dynamic_filter), + false, + )) as Arc + }) + .collect(); + let plan = UnionExec::try_new(consumers)?; + + let discovered = discover_dynamic_filter_consumers(&plan)?; + + assert_eq!(discovered.len(), 1); + assert_eq!(discovered[0].id, dynamic_filter.expression_id().unwrap()); + assert!(has_nonlocal_dynamic_filter_relationships(&plan)?); + Ok(()) + } + + #[test] + fn identifies_a_producer_without_a_local_consumer() -> Result<()> { + let schema = Arc::new(Schema::new(vec![Field::new("a", DataType::Int32, false)])); + let dynamic_filter = Arc::new(DynamicFilterPhysicalExpr::new( + vec![Arc::new(Column::new("a", 0))], + lit(true), + )) as Arc; + let plan = Arc::new(ExpressionExec::new( + Arc::new(EmptyExec::new(schema)), + dynamic_filter, + true, + )) as Arc; + + assert!(has_nonlocal_dynamic_filter_relationships(&plan)?); + Ok(()) + } + + #[derive(Debug)] + struct ExpressionExec { + input: Arc, + expression: Arc, + produces_expression: bool, + } + + impl ExpressionExec { + fn new( + input: Arc, + expression: Arc, + produces_expression: bool, + ) -> Self { + Self { + input, + expression, + produces_expression, + } + } + } + + impl DisplayAs for ExpressionExec { + fn fmt_as(&self, _: DisplayFormatType, f: &mut Formatter) -> std::fmt::Result { + write!(f, "ExpressionExec") + } + } + + impl ExecutionPlan for ExpressionExec { + fn name(&self) -> &str { + "ExpressionExec" + } + + fn properties(&self) -> &Arc { + self.input.properties() + } + + fn children(&self) -> Vec<&Arc> { + vec![&self.input] + } + + fn dynamic_expressions_produced(&self) -> Vec> { + self.produces_expression + .then(|| Arc::clone(&self.expression)) + .into_iter() + .collect() + } + + fn apply_expressions( + &self, + f: &mut dyn FnMut(&Arc) -> Result, + ) -> Result { + apply_expression_roots([&self.expression], f) + } + + fn with_new_children( + self: Arc, + mut children: Vec>, + ) -> Result> { + Ok(Arc::new(Self::new( + children.remove(0), + Arc::clone(&self.expression), + self.produces_expression, + ))) + } + + fn execute( + &self, + partition: usize, + context: Arc, + ) -> Result { + self.input.execute(partition, context) + } + } +} diff --git a/src/dynamic_filtering/display.rs b/src/dynamic_filtering/display.rs new file mode 100644 index 000000000..8075020d6 --- /dev/null +++ b/src/dynamic_filtering/display.rs @@ -0,0 +1,337 @@ +use crate::codec::{decode_physical_expr, roundtrip_pb}; +use crate::coordinator::DistributedExec; +use crate::dynamic_filtering::discover_dynamic_filter_consumers; +use crate::execution_plans::DistributedLeafExec; +use crate::{TaskCompletedDynamicFilters, TaskKey}; +use datafusion::common::tree_node::{Transformed, TreeNode, TreeNodeRecursion}; +use datafusion::common::{HashMap, Result, internal_err}; +use datafusion::execution::TaskContext; +use datafusion::physical_expr::expressions::DynamicFilterPhysicalExpr; +use datafusion::physical_plan::empty::EmptyExec; +use datafusion::physical_plan::sorts::sort::SortExec; +use datafusion::physical_plan::{ + ChildrenPropertiesMode, ExecutionPlan, ExecutionPlanProperties, ReplaceChildrenOptions, +}; +use datafusion_proto::protobuf::physical_expr_node::ExprType; +use std::sync::Arc; + +/// Rewrites an executed distributed plan with the dynamic filters reported by its completed +/// worker tasks. +/// +/// When composing this with [`crate::rewrite_distributed_plan_with_metrics`], dynamic filters must +/// be rewritten first. +/// `task_ctx` must have the same session configuration and codecs as the context used to execute +/// the plan. +pub async fn rewrite_distributed_plan_with_dynamic_filters( + plan: Arc, + task_ctx: &Arc, +) -> Result> { + let Some(distributed_exec) = plan.downcast_ref::() else { + return Ok(plan); + }; + + if distributed_exec.completed_dynamic_filter_store.is_none() { + return Ok(plan); + } + let plan_for_viz = distributed_exec.plan_for_viz()?; + let Some(reports) = distributed_exec.wait_for_dynamic_filters().await else { + return internal_err!("dynamic filters were enabled but the execution was not prepared"); + }; + // Avoids mutating the `plan_for_viz` of the incoming DistributedExec. + let plan_for_viz = + sever_dynamic_filter_relationships_in_plan_for_display(plan_for_viz, task_ctx)?; + apply_reports_to_distributed_leaves(&plan_for_viz, &reports, task_ctx); + let plan = distributed_exec.with_plan_for_viz(Arc::clone(&plan_for_viz))?; + plan.replace_children( + vec![plan_for_viz], + ReplaceChildrenOptions::new(ChildrenPropertiesMode::Recompute), + ) +} + +/// Severs dynamic filter connections so we can update filter values for +/// display purposes without having an update in one node propagate to another. +/// +/// For example, in this plan, we would like to be able to [`update()`] every variant independently +/// without mutating the producer or other variants +/// +/// ```text +/// RepartitionExec: +/// AggregateExec: mode=Partial +/// HashJoinExec: mode=Partitioned +/// DistributedLeafExec: +/// t0: DataSourceExec: ... +/// t1: DataSourceExec: ... +/// DistributedLeafExec: +/// t0: DataSourceExec: predicate=DynamicFilter [ f_dkey@2 >= A AND f_dkey@2 <= A AND f_dkey@2 IN (SET) ([]) ] <- unique filter +/// t1: DataSourceExec: predicate=DynamicFilter [ f_dkey@2 >= B AND f_dkey@2 <= B AND f_dkey@2 IN (SET) ([]) ] <- unique filter +/// ``` +/// +/// This is done by deep-copying every leaf variant so we don't have to +/// worry about any shared state. +/// +/// [`update()`]: DynamicFilterPhysicalExpr::update() +pub(crate) fn sever_dynamic_filter_relationships_in_plan_for_display( + plan: Arc, + task_ctx: &Arc, +) -> Result> { + plan.transform_up(|node| { + let Some(leaf) = node.downcast_ref::() else { + return Ok(Transformed::no(node)); + }; + + let variants = leaf + .variants() + .iter() + .map(|variant| { + let variant = roundtrip_pb(Arc::clone(variant), task_ctx)?; + // The variant can have a SortExec + isolate_sort_dynamic_filters_for_display(variant, task_ctx) + }) + .collect::>>()?; + + Ok(Transformed::yes(Arc::new(DistributedLeafExec::try_new( + Arc::clone(leaf.original()), + variants, + )?) as Arc)) + }) + // Handle SortExec nodes not inside variants. + .and_then(|transformed| isolate_sort_dynamic_filters_for_display(transformed.data, task_ctx)) +} + +/// Deep-copies dynamic-filter-producing [`SortExec`]s by doing a proto roundtrip. Some producers +/// like [`SortExec`] display their dynamic filters, so we need to explicitly handle +/// displaying different dynamic filters for each task containing a [`SortExec`]. For now, we +/// just clear the dynamic filter and don't display it. +/// +/// To avoid serializing an entire subtree, we swap in an [`EmptyExec`]: +/// +/// ```text +/// SortExec SortExec SortExec +/// ...children -> EmptyExec -> ...children +/// ``` +/// +/// TODO(#677): display producer dynamic filters +fn isolate_sort_dynamic_filters_for_display( + plan: Arc, + task_ctx: &Arc, +) -> Result> { + plan.transform_up(|node| { + let Some(sort) = node.downcast_ref::() else { + return Ok(Transformed::no(node)); + }; + if node.dynamic_expressions_produced().is_empty() { + return Ok(Transformed::no(node)); + } + + let input = Arc::clone(sort.input()); + let placeholder = Arc::new( + EmptyExec::new(input.schema()) + .with_partitions(input.output_partitioning().partition_count()), + ) as Arc; + let recompute = ReplaceChildrenOptions::new(ChildrenPropertiesMode::Recompute); + let sort_with_placeholder = node.replace_children(vec![placeholder], recompute)?; + + let isolated = roundtrip_pb(sort_with_placeholder, task_ctx)?; + let isolated = isolated.replace_children(vec![input], recompute)?; + + Ok(Transformed::yes(isolated)) + }) + .map(|transformed| transformed.data) +} + +/// Applies successful worker reports only to the matching task-local visualization variants. +fn apply_reports_to_distributed_leaves( + plan: &Arc, + reports: &HashMap, + task_ctx: &Arc, +) { + let _ = plan.apply(|node| { + let Some(leaf) = node.downcast_ref::() else { + return Ok(TreeNodeRecursion::Continue); + }; + + for (task_key, report) in reports { + let Some(variant) = leaf.variants().get(task_key.task_number) else { + continue; + }; + let updates: HashMap<_, _> = report + .filters + .iter() + .map(|filter| (filter.expression_id, &filter.expression)) + .collect(); + let Ok(consumers) = discover_dynamic_filter_consumers(variant) else { + continue; + }; + for consumer in consumers { + let Some(expression) = updates.get(&consumer.id).copied() else { + continue; + }; + let Ok(proto) = expression.to_proto(task_ctx) else { + continue; + }; + let Some(ExprType::DynamicFilter(dynamic_filter_proto)) = proto.expr_type.as_ref() + else { + continue; + }; + let Ok(reported_expression) = + decode_physical_expr(&proto, consumer.input_schema.as_ref(), task_ctx) + else { + continue; + }; + let Some(reported_dynamic_filter) = + reported_expression.downcast_ref::() + else { + continue; + }; + let Ok(expression) = reported_dynamic_filter.current() else { + continue; + }; + if dynamic_filter_proto.generation > 1 { + let _ = consumer.expression.update(expression); + } + } + } + + Ok(TreeNodeRecursion::Continue) + }); +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::test_utils::mock_exec::MockExec; + use crate::{MaybeEncoded, TaskDynamicFilter}; + use datafusion::arrow::datatypes::{DataType, Field, Schema}; + use datafusion::logical_expr::Operator; + use datafusion::physical_expr::expressions::{ + BinaryExpr, Column, DynamicFilterPhysicalExpr, lit, + }; + use datafusion::physical_expr::{LexOrdering, PhysicalExpr, PhysicalSortExpr}; + use datafusion::physical_plan::displayable; + use datafusion::physical_plan::empty::EmptyExec; + use datafusion::physical_plan::filter::FilterExec; + use datafusion::physical_plan::sorts::sort::SortExec; + use datafusion::prelude::SessionContext; + use insta::assert_snapshot; + use uuid::Uuid; + + #[test] + fn visualization_variants_do_not_share_dynamic_filter_state() -> Result<()> { + let schema = Arc::new(Schema::new(vec![Field::new("a", DataType::Int32, false)])); + let column = Arc::new(Column::new("a", 0)) as Arc; + let dynamic_filter = Arc::new(DynamicFilterPhysicalExpr::new( + vec![Arc::clone(&column)], + lit(true), + )) as Arc; + let input = Arc::new(EmptyExec::new(schema)) as Arc; + let variant = Arc::new(FilterExec::try_new( + Arc::clone(&dynamic_filter), + Arc::clone(&input), + )?) as Arc; + let leaf = Arc::new(DistributedLeafExec::try_new( + Arc::clone(&variant), + [Arc::clone(&variant), variant], + )?) as Arc; + + let task_ctx = SessionContext::new().task_ctx(); + let isolated = sever_dynamic_filter_relationships_in_plan_for_display(leaf, &task_ctx)?; + let expression = + Arc::new(BinaryExpr::new(column, Operator::Gt, lit(10_i32))) as Arc; + dynamic_filter + .downcast_ref::() + .unwrap() + .update(expression)?; + dynamic_filter + .downcast_ref::() + .unwrap() + .mark_complete(); + let report = TaskCompletedDynamicFilters { + filters: vec![TaskDynamicFilter { + expression_id: dynamic_filter.expression_id().unwrap(), + expression: MaybeEncoded::Decoded(Arc::clone(&dynamic_filter)), + }], + }; + let reports = HashMap::from_iter([( + TaskKey { + query_id: Uuid::nil(), + stage_id: 1, + task_number: 0, + }, + report, + )]); + + apply_reports_to_distributed_leaves(&isolated, &reports, &task_ctx); + let leaf = isolated.downcast_ref::().unwrap(); + let task_0 = displayable(leaf.variants()[0].as_ref()) + .one_line() + .to_string(); + let task_1 = displayable(leaf.variants()[1].as_ref()) + .one_line() + .to_string(); + assert_snapshot!(task_0, @"FilterExec: DynamicFilter [ a@0 > 10 ]"); + assert_snapshot!(task_1, @"FilterExec: DynamicFilter [ empty ]"); + Ok(()) + } + + #[test] + fn visualization_isolates_sort_producer_filter_without_mutating_original() -> Result<()> { + let schema = Arc::new(Schema::new(vec![Field::new("a", DataType::Int32, false)])); + // MockExec has no protobuf representation. A successful isolation therefore proves the + // real child was replaced before serializing the SortExec. + let input = Arc::new(MockExec::new_partitioned(vec![vec![], vec![]], schema)) + as Arc; + let ordering = + LexOrdering::new([PhysicalSortExpr::new_default(Arc::new(Column::new("a", 0)))]) + .unwrap(); + let sort = Arc::new( + SortExec::new(ordering, Arc::clone(&input)) + .with_fetch(Some(10)) + .with_preserve_partitioning(true), + ) as Arc; + let produced = sort.dynamic_expressions_produced(); + let dynamic_filter = produced[0] + .downcast_ref::() + .unwrap(); + let expression_id = dynamic_filter.expression_id(); + dynamic_filter.update(Arc::new(BinaryExpr::new( + Arc::new(Column::new("a", 0)), + Operator::Gt, + lit(10_i32), + )))?; + + let task_ctx = SessionContext::new().task_ctx(); + let isolated = isolate_sort_dynamic_filters_for_display(Arc::clone(&sort), &task_ctx)?; + let isolated_sort = isolated.downcast_ref::().unwrap(); + assert!(Arc::ptr_eq(isolated_sort.input(), &input)); + assert_eq!(isolated_sort.fetch(), Some(10)); + assert!(isolated_sort.preserve_partitioning()); + assert_eq!(isolated.schema(), sort.schema()); + assert_eq!( + isolated.output_partitioning().partition_count(), + sort.output_partitioning().partition_count() + ); + assert_eq!( + isolated.dynamic_expressions_produced()[0].expression_id(), + expression_id + ); + assert_snapshot!( + displayable(isolated.as_ref()).one_line().to_string(), + @"SortExec: TopK(fetch=10), expr=[a@0 ASC], preserve_partitioning=[true], filter=[a@0 > 10]" + ); + + dynamic_filter.update(Arc::new(BinaryExpr::new( + Arc::new(Column::new("a", 0)), + Operator::Gt, + lit(20_i32), + )))?; + assert_snapshot!( + displayable(sort.as_ref()).one_line().to_string(), + @"SortExec: TopK(fetch=10), expr=[a@0 ASC], preserve_partitioning=[true], filter=[a@0 > 20]" + ); + assert_snapshot!( + displayable(isolated.as_ref()).one_line().to_string(), + @"SortExec: TopK(fetch=10), expr=[a@0 ASC], preserve_partitioning=[true], filter=[a@0 > 10]" + ); + Ok(()) + } +} diff --git a/src/dynamic_filtering/mod.rs b/src/dynamic_filtering/mod.rs new file mode 100644 index 000000000..5fa38c2e9 --- /dev/null +++ b/src/dynamic_filtering/mod.rs @@ -0,0 +1,49 @@ +mod discovery; +mod display; + +use crate::codec::roundtrip_pb; +use datafusion::common::Result; +use datafusion::execution::TaskContext; +use datafusion::physical_plan::ExecutionPlan; +use std::sync::Arc; + +pub(crate) use discovery::*; +pub use display::rewrite_distributed_plan_with_dynamic_filters; +pub(crate) use display::sever_dynamic_filter_relationships_in_plan_for_display; + +// We must take care to avoid partial dynamic filter updates when sending an +// in-memory plan. +// +// Consider this partitioned hash join topology where the consumer task is +// collocated with one producer on worker A: +// ```text +// Worker A +// +// Stage 2 Task 0 +// HashJoinExec <- Dynamic Filter Produced: (foo > 100) +// +// Stage 1 Task 0 +// DataSourceExec <- consumer +// +// Worker B +// Stage 2 Task 1 +// HashJoinExec <- Dynamic Filter Produced: (foo != 150) +// ``` +// +// The in-process transport allows the Worker A join to propagate its filter to +// the consumer and mark it as completed, so the consumer incorrectly applies +// (foo > 100) instead of (foo > 100 OR foo != 150). +// +// In this situation, we roundtrip Stage 1 Task 0 to sever the in-memory +// relationship. The dynamic filter update from the producer must reach the +// coordinator for merging prior to being forwarded to the consumer. +pub(crate) fn maybe_roundtrip_plan_to_sever_in_memory_dynamic_filter_relationships( + plan: Arc, + task_ctx: &Arc, +) -> Result> { + if has_nonlocal_dynamic_filter_relationships(&plan)? { + roundtrip_pb(plan, task_ctx) + } else { + Ok(plan) + } +} diff --git a/src/lib.rs b/src/lib.rs index 3ad27ddd9..87fd144a4 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -6,6 +6,7 @@ mod config_extension_ext; mod coordinator; mod distributed_ext; mod distributed_planner; +mod dynamic_filtering; mod execution_plans; mod explain_analyze; mod metrics; @@ -23,6 +24,7 @@ pub use distributed_ext::{DistributedExt, DistributedGetterExt}; pub use distributed_planner::{ DistributedConfig, NetworkBoundary, NetworkBoundaryExt, ProducerHead, SessionStateBuilderExt, }; +pub use dynamic_filtering::rewrite_distributed_plan_with_dynamic_filters; pub use events::{ DesiredTaskCountEvent, DesiredTaskCountEventResponse, DesiredTaskCountHandler, RouteTasksEvent, RouteTasksEventResponse, RouteTasksHandler, ScaleUpLeafNodeEvent, ScaleUpLeafNodeEventResponse, @@ -54,9 +56,9 @@ pub use worker_resolver::{WorkerResolver, get_distributed_worker_resolver}; pub use protocol::{ ChannelResolver, CoordinatorToWorkerMsg, ExecuteTaskRequest, GetWorkerInfoRequest, - GetWorkerInfoResponse, LoadInfo, SetPlanRequest, TaskKey, TaskMetrics, WorkUnitBatch, - WorkUnitFeedDeclaration, WorkUnitMsg, WorkerChannel, WorkerToCoordinatorMsg, - get_distributed_channel_resolver, + GetWorkerInfoResponse, LoadInfo, SetPlanRequest, TaskCompletedDynamicFilters, + TaskDynamicFilter, TaskKey, TaskMetrics, WorkUnitBatch, WorkUnitFeedDeclaration, WorkUnitMsg, + WorkerChannel, WorkerToCoordinatorMsg, get_distributed_channel_resolver, }; pub use stage::{ DistributedTaskContext, Stage, display_plan_ascii, display_plan_graphviz, explain_analyze, diff --git a/src/metrics/task_metrics_collector.rs b/src/metrics/task_metrics_collector.rs index ba6e68556..97429cb0f 100644 --- a/src/metrics/task_metrics_collector.rs +++ b/src/metrics/task_metrics_collector.rs @@ -174,13 +174,14 @@ mod tests { // Per-task metrics are delivered asynchronously over the `WorkerToCoordinator` side // channel after execution completes; await that delivery instead of racing it (see #487). - dist_exec.wait_for_metrics().await; - - let metrics_store = dist_exec.metrics_store.as_ref().unwrap(); + let task_metrics = dist_exec + .wait_for_metrics() + .await + .expect("metrics collection is enabled"); // Ensure that there's metrics for each node for each task for each stage. for expected_task_key in expected_task_keys { - let actual_metrics = metrics_store.get(&expected_task_key).unwrap(); + let actual_metrics = task_metrics.get(&expected_task_key).unwrap(); // Verify that metrics were collected for all nodes. Some nodes may legitimately have // empty metrics (e.g., custom execution plans without metrics), which is fine - we @@ -295,11 +296,13 @@ mod tests { // Metrics are delivered via the WorkerToCoordinator side channel in a background task. // Wait for that delivery to complete before asserting, rather than racing it. - dist_exec.wait_for_metrics().await; - let metrics_store = dist_exec.metrics_store.as_ref().unwrap(); + let task_metrics = dist_exec + .wait_for_metrics() + .await + .expect("metrics collection is enabled"); for expected_task_key in &expected_task_keys { - let actual_metrics = metrics_store.get(expected_task_key).unwrap_or_else(|| { + let actual_metrics = task_metrics.get(expected_task_key).unwrap_or_else(|| { panic!( "Missing metrics for task key {expected_task_key:?}. \ The LIMIT caused the stream to be dropped before the worker \ diff --git a/src/metrics/task_metrics_rewriter.rs b/src/metrics/task_metrics_rewriter.rs index a3633203f..2344c21e2 100644 --- a/src/metrics/task_metrics_rewriter.rs +++ b/src/metrics/task_metrics_rewriter.rs @@ -1,11 +1,11 @@ use crate::common::TreeNodeExt; -use crate::coordinator::{DistributedExec, MetricsStore}; +use crate::coordinator::DistributedExec; use crate::distributed_planner::NetworkBoundaryExt; use crate::execution_plans::MetricsWrapperExec; use crate::metrics::DISTRIBUTED_DATAFUSION_TASK_ID_LABEL; use crate::metrics::collect_plan_metrics; use crate::stage::{LocalStage, Stage}; -use crate::{DistributedTaskContext, TaskKey}; +use crate::{DistributedTaskContext, TaskKey, TaskMetrics}; use datafusion::common::HashMap; use datafusion::common::plan_err; use datafusion::common::tree_node::Transformed; @@ -49,13 +49,14 @@ pub async fn rewrite_distributed_plan_with_metrics( return Ok(plan); }; - distributed_exec.wait_for_metrics().await; - - let Some(metrics_collection) = distributed_exec.metrics_store.clone() else { + if distributed_exec.metrics_store.is_none() { return Ok(plan); - }; + } let head_stage = distributed_exec.head_stage()?; + let Some(metrics_collection) = distributed_exec.wait_for_metrics().await else { + return internal_err!("metrics were enabled but the execution was not prepared"); + }; let task_metrics = collect_plan_metrics(&head_stage)?; // Rewrite the DistributedExec's child plan with metrics. @@ -73,14 +74,13 @@ pub async fn rewrite_distributed_plan_with_metrics( }; // This transform is a bit inefficient because we traverse the plan nodes twice // For now, we are okay with trading off performance for simplicity. - let plan_with_metrics = - stage_metrics_rewriter(stage, Arc::clone(&metrics_collection), format)?; + let plan_with_metrics = stage_metrics_rewriter(stage, &metrics_collection, format)?; let network_boundary = network_boundary.with_input_stage(Stage::Local(LocalStage { query_id: stage.query_id, num: stage.num, plan: plan_with_metrics, tasks: stage.tasks, - metrics_set: stage.metrics_set.clone(), + metrics_set: stage_metrics(stage, &metrics_collection)?, }))?; let network_boundary = MetricsWrapperExec::new(network_boundary, plan.metrics().unwrap_or_default()); @@ -89,12 +89,48 @@ pub async fn rewrite_distributed_plan_with_metrics( Ok(Transformed::no(plan)) })?; + let plan = distributed_exec.with_plan_for_viz(Arc::clone(&transformed.data))?; plan.replace_children( vec![transformed.data], ReplaceChildrenOptions::new(ChildrenPropertiesMode::Recompute), ) } +/// Gathers metrics that belong to a task as a whole rather than to an execution-plan node. +fn stage_metrics( + stage: &LocalStage, + metrics_collection: &HashMap, +) -> Result { + let mut all_metrics = stage.metrics_set.clone(); + for task_number in 0..stage.tasks { + let task_key = TaskKey { + query_id: stage.query_id, + stage_id: stage.num, + task_number, + }; + let Some(task_metrics) = metrics_collection.get(&task_key) else { + return internal_err!( + "not enough metrics provided to rewrite task: missing metrics for task {} in stage {}", + task_number, + stage.num + ); + }; + for metric in task_metrics.task_metrics.iter() { + let mut labels = metric.labels().to_vec(); + labels.push(Label::new( + DISTRIBUTED_DATAFUSION_TASK_ID_LABEL, + task_number.to_string(), + )); + all_metrics.push(Arc::new(Metric::new_with_labels( + metric.value().clone(), + metric.partition(), + labels, + ))); + } + } + Ok(all_metrics) +} + /// Extra information for rewriting local plans. #[derive(Default)] pub struct RewriteCtx { @@ -210,7 +246,7 @@ pub fn rewrite_local_plan_with_metrics( /// Note: Metrics may be aggregated by name (ex. output_rows) automatically by various datafusion utils. pub fn stage_metrics_rewriter( stage: &LocalStage, - metrics_collection: Arc, + metrics_collection: &HashMap, format: DistributedMetricsFormat, ) -> Result> { // Phase 1 — accumulate per-task metrics into a map keyed by node identity. @@ -286,7 +322,6 @@ pub fn stage_metrics_rewriter( #[cfg(test)] mod tests { use crate::DistributedExt; - use crate::coordinator::MetricsStore; use crate::metrics::DISTRIBUTED_DATAFUSION_TASK_ID_LABEL; use crate::metrics::task_metrics_rewriter::MetricsWrapperExec; use crate::metrics::task_metrics_rewriter::{ @@ -304,6 +339,7 @@ mod tests { use datafusion::arrow::array::{Int32Array, StringArray}; use datafusion::arrow::datatypes::{DataType, Field, Schema}; use datafusion::arrow::record_batch::RecordBatch; + use datafusion::common::HashMap; use datafusion::execution::SessionStateBuilder; use datafusion::physical_plan::empty::EmptyExec; use datafusion::physical_plan::metrics::{Count, Label, Metric, MetricValue, MetricsSet}; @@ -452,7 +488,7 @@ mod tests { let num_metrics_per_task_per_node = 4; // Generate metrics for each task and store them in the map. - let metrics_collection = MetricsStore::from_entries((0..stage.tasks).map(|task_id| { + let metrics_collection = HashMap::from_iter((0..stage.tasks).map(|task_id| { let task_key = TaskKey { query_id: stage.query_id, stage_id: stage.num, @@ -472,11 +508,8 @@ mod tests { }; (task_key, task_metrics) })); - let metrics_collection = Arc::new(metrics_collection); - // Rewrite the plan. - let rewritten_plan = - stage_metrics_rewriter(&stage, metrics_collection.clone(), format).unwrap(); + let rewritten_plan = stage_metrics_rewriter(&stage, &metrics_collection, format).unwrap(); // Collect metrics from the plan. let mut actual_metrics = vec![]; @@ -600,7 +633,6 @@ mod tests { .await .unwrap(); collect(plan.clone(), ctx.task_ctx()).await.unwrap(); - assert!(plan.is::()); let rewritten_plan = rewrite_distributed_plan_with_metrics(plan, DistributedMetricsFormat::Aggregated) .await diff --git a/src/protocol/grpc/generated/worker.rs b/src/protocol/grpc/generated/worker.rs index 6bf8beb89..78bbd1058 100644 --- a/src/protocol/grpc/generated/worker.rs +++ b/src/protocol/grpc/generated/worker.rs @@ -24,7 +24,7 @@ pub mod coordinator_to_worker_msg { } #[derive(Clone, PartialEq, ::prost::Message)] pub struct WorkerToCoordinatorMsg { - #[prost(oneof = "worker_to_coordinator_msg::Inner", tags = "1, 2, 3")] + #[prost(oneof = "worker_to_coordinator_msg::Inner", tags = "1, 2, 3, 4")] pub inner: ::core::option::Option, } /// Nested message and enum types in `WorkerToCoordinatorMsg`. @@ -43,8 +43,35 @@ pub mod worker_to_coordinator_msg { LoadInfo(super::LoadInfo), #[prost(bool, tag = "3")] LoadInfoEos(bool), + /// Final dynamic filters used by dynamic-filter consumer execution-plan nodes. + /// + /// Filters are deduplicated by expression_id because consumers with the same ID share + /// logical filter state within a task. For example, this plan includes one entry: + /// + /// HashJoin producer: expression_id=10 + /// ├── DataSourceExec build side + /// └── UnionExec probe side + /// ├── DataSourceExec A consumer: expression_id=10 + /// └── DataSourceExec B consumer: expression_id=10 + /// + /// Another task in the same stage may report a different value for expression_id=10. + #[prost(message, tag = "4")] + TaskCompletedDynamicFilters(super::TaskCompletedDynamicFilters), } } +#[derive(Clone, PartialEq, Eq, Hash, ::prost::Message)] +pub struct DynamicFilter { + #[prost(uint64, tag = "1")] + pub expression_id: u64, + /// Serialized datafusion.proto.PhysicalExprNode. + #[prost(bytes = "vec", tag = "2")] + pub expression_proto: ::prost::alloc::vec::Vec, +} +#[derive(Clone, PartialEq, ::prost::Message)] +pub struct TaskCompletedDynamicFilters { + #[prost(message, repeated, tag = "1")] + pub filters: ::prost::alloc::vec::Vec, +} #[derive(Clone, PartialEq, ::prost::Message)] pub struct TaskMetrics { /// Metrics for a single task's plan nodes in pre-order traversal order. @@ -80,8 +107,8 @@ pub struct LoadInfo { /// The amount of rows that were pulled from leaf nodes while this partition was sampling data. #[prost(uint64, tag = "8")] pub rows_pulled_from_leaf: u64, - /// Whether the sampled partition stream reached end-of-stream by the time this LoadInfo was - /// captured. + /// Whether the sampled partition stream reached end-of-stream (i.e. the partition finished + /// producing all of its output) by the time this LoadInfo was captured. #[prost(bool, tag = "9")] pub reached_eos: bool, } diff --git a/src/protocol/grpc/worker.proto b/src/protocol/grpc/worker.proto index 2affe1187..61319f5f8 100644 --- a/src/protocol/grpc/worker.proto +++ b/src/protocol/grpc/worker.proto @@ -39,9 +39,33 @@ message WorkerToCoordinatorMsg { LoadInfo load_info = 2; bool load_info_eos = 3; + + // Final dynamic filters used by dynamic-filter consumer execution-plan nodes. + // + // Filters are deduplicated by expression_id because consumers with the same ID share + // logical filter state within a task. For example, this plan includes one entry: + // + // HashJoin producer: expression_id=10 + // ├── DataSourceExec build side + // └── UnionExec probe side + // ├── DataSourceExec A consumer: expression_id=10 + // └── DataSourceExec B consumer: expression_id=10 + // + // Another task in the same stage may report a different value for expression_id=10. + TaskCompletedDynamicFilters task_completed_dynamic_filters = 4; } } +message DynamicFilter { + uint64 expression_id = 1; + // Serialized datafusion.proto.PhysicalExprNode. + bytes expression_proto = 2; +} + +message TaskCompletedDynamicFilters { + repeated DynamicFilter filters = 1; +} + message TaskMetrics { // Metrics for a single task's plan nodes in pre-order traversal order. // The TaskKey is implicit — it is determined by the SetPlanRequest that diff --git a/src/protocol/grpc/worker_client.rs b/src/protocol/grpc/worker_client.rs index 060d8e875..81af88ab7 100644 --- a/src/protocol/grpc/worker_client.rs +++ b/src/protocol/grpc/worker_client.rs @@ -9,9 +9,9 @@ use crate::{ BytesMetricExt, CoordinatorToWorkerMsg, DISTRIBUTED_DATAFUSION_TASK_ID_LABEL, DistributedConfig, ExecuteTaskRequest, FirstLatencyMetric, GetWorkerInfoRequest, GetWorkerInfoResponse, LatencyMetricExt, LoadInfo, MaxLatencyMetric, MaybeEncoded, - MinLatencyMetric, P50LatencyMetric, P95LatencyMetric, ProducerHead, SetPlanRequest, TaskKey, - TaskMetrics, WorkUnitBatch, WorkUnitFeedDeclaration, WorkUnitMsg, WorkerChannel, - WorkerToCoordinatorMsg, + MinLatencyMetric, P50LatencyMetric, P95LatencyMetric, ProducerHead, SetPlanRequest, + TaskCompletedDynamicFilters, TaskDynamicFilter, TaskKey, TaskMetrics, WorkUnitBatch, + WorkUnitFeedDeclaration, WorkUnitMsg, WorkerChannel, WorkerToCoordinatorMsg, }; use arrow_flight::FlightData; use arrow_flight::decode::FlightRecordBatchStream; @@ -508,10 +508,30 @@ fn decode_worker_to_coordinator_msg( pb::worker_to_coordinator_msg::Inner::LoadInfoEos(_) => { WorkerToCoordinatorMsg::LoadInfoEos } + pb::worker_to_coordinator_msg::Inner::TaskCompletedDynamicFilters(filters) => { + WorkerToCoordinatorMsg::TaskCompletedDynamicFilters( + decode_task_completed_dynamic_filters(filters)?, + ) + } }, ) } +fn decode_task_completed_dynamic_filters( + filters: pb::TaskCompletedDynamicFilters, +) -> Result { + Ok(TaskCompletedDynamicFilters { + filters: filters + .filters + .into_iter() + .map(|filter| TaskDynamicFilter { + expression_id: filter.expression_id, + expression: MaybeEncoded::Encoded(filter.expression_proto), + }) + .collect(), + }) +} + fn decode_task_metrics(task_metrics: pb::TaskMetrics) -> Result { Ok(TaskMetrics { pre_order_plan_metrics: task_metrics diff --git a/src/protocol/grpc/worker_service.rs b/src/protocol/grpc/worker_service.rs index 1f0fdf0a6..f0231a635 100644 --- a/src/protocol/grpc/worker_service.rs +++ b/src/protocol/grpc/worker_service.rs @@ -7,8 +7,8 @@ use crate::common::{deserialize_uuid, now_ns}; use crate::protocol::grpc::{ObservabilityServiceImpl, ObservabilityServiceServer}; use crate::{ CoordinatorToWorkerMsg, DistributedConfig, ExecuteTaskRequest, LoadInfo, MaybeEncoded, - ProducerHead, SetPlanRequest, TaskKey, TaskMetrics, WorkUnitBatch, WorkUnitFeedDeclaration, - WorkUnitMsg, Worker, WorkerResolver, WorkerToCoordinatorMsg, + ProducerHead, SetPlanRequest, TaskCompletedDynamicFilters, TaskKey, TaskMetrics, WorkUnitBatch, + WorkUnitFeedDeclaration, WorkUnitMsg, Worker, WorkerResolver, WorkerToCoordinatorMsg, }; use arrow_flight::FlightData; @@ -20,7 +20,7 @@ use datafusion::arrow::array::{Array, AsArray, RecordBatch, RecordBatchOptions}; use datafusion::arrow::ipc::CompressionType; use datafusion::arrow::ipc::writer::IpcWriteOptions; use datafusion::common::DataFusionError; -use datafusion::execution::SendableRecordBatchStream; +use datafusion::execution::{SendableRecordBatchStream, TaskContext}; use futures::stream::BoxStream; use futures::{StreamExt, TryStreamExt}; use prost::Message; @@ -107,6 +107,14 @@ impl pb::worker_service_server::WorkerService for Worker { }; let set_plan_request = decode_set_plan_request(set_plan_request)?; + let task_key = set_plan_request.task_key; + // Dynamic-filter reports may carry decoded physical expressions. Retain the worker's + // task data so the gRPC boundary can encode them with its configured codecs, even after + // the completed task has been removed from the worker cache. + let task_data_entry = self + .task_data_entries + .get_with(task_key, async { Default::default() }) + .await; let input_stream = body .map_err(map_status_to_datafusion_error) @@ -118,9 +126,15 @@ impl pb::worker_service_server::WorkerService for Worker { let output_stream = self .coordinator_channel(metadata.into_headers(), set_plan_request, input_stream) .await - .map_err(datafusion_error_to_tonic_status)? - .map(|msg| match msg { - Ok(msg) => encode_worker_to_coordinator_msg(msg), + .map_err(datafusion_error_to_tonic_status)?; + let task_data = task_data_entry + .read_now() + .ok_or_else(|| Status::internal("worker task data was not initialized"))? + .map_err(|error| datafusion_error_to_tonic_status(DataFusionError::Shared(error)))?; + let task_ctx = Arc::clone(&task_data.task_ctx); + let output_stream = output_stream + .map(move |msg| match msg { + Ok(msg) => encode_worker_to_coordinator_msg(msg, &task_ctx), Err(err) => Err(datafusion_error_to_tonic_status(err)), }) .boxed(); @@ -258,6 +272,7 @@ pub(super) fn decode_producer_head(proto: pb::execute_task_request::ProducerHead fn encode_worker_to_coordinator_msg( msg: WorkerToCoordinatorMsg, + task_ctx: &Arc, ) -> Result { Ok(pb::WorkerToCoordinatorMsg { inner: Some(match msg { @@ -272,10 +287,36 @@ fn encode_worker_to_coordinator_msg( WorkerToCoordinatorMsg::LoadInfoEos => { pb::worker_to_coordinator_msg::Inner::LoadInfoEos(true) } + WorkerToCoordinatorMsg::TaskCompletedDynamicFilters(filters) => { + pb::worker_to_coordinator_msg::Inner::TaskCompletedDynamicFilters( + encode_task_completed_dynamic_filters(filters, task_ctx)?, + ) + } }), }) } +fn encode_task_completed_dynamic_filters( + filters: TaskCompletedDynamicFilters, + task_ctx: &Arc, +) -> Result { + Ok(pb::TaskCompletedDynamicFilters { + filters: filters + .filters + .into_iter() + .map(|filter| { + Ok(pb::DynamicFilter { + expression_id: filter.expression_id, + expression_proto: filter + .expression + .encode(task_ctx) + .map_err(datafusion_error_to_tonic_status)?, + }) + }) + .collect::>()?, + }) +} + fn encode_task_metrics(task_metrics: TaskMetrics) -> Result { Ok(pb::TaskMetrics { pre_order_plan_metrics: task_metrics diff --git a/src/protocol/mod.rs b/src/protocol/mod.rs index 4bd507dfe..1ee655e3a 100644 --- a/src/protocol/mod.rs +++ b/src/protocol/mod.rs @@ -11,6 +11,6 @@ pub use channel_resolver::{ChannelResolver, get_distributed_channel_resolver}; pub use in_process::LocalWorkerContext; pub use worker_channel::{ CoordinatorToWorkerMsg, ExecuteTaskRequest, GetWorkerInfoRequest, GetWorkerInfoResponse, - LoadInfo, SetPlanRequest, TaskKey, TaskMetrics, WorkUnitBatch, WorkUnitFeedDeclaration, - WorkUnitMsg, WorkerChannel, WorkerToCoordinatorMsg, + LoadInfo, SetPlanRequest, TaskCompletedDynamicFilters, TaskDynamicFilter, TaskKey, TaskMetrics, + WorkUnitBatch, WorkUnitFeedDeclaration, WorkUnitMsg, WorkerChannel, WorkerToCoordinatorMsg, }; diff --git a/src/protocol/worker_channel.rs b/src/protocol/worker_channel.rs index 0bea0bbe2..d702818f2 100644 --- a/src/protocol/worker_channel.rs +++ b/src/protocol/worker_channel.rs @@ -3,6 +3,7 @@ use async_trait::async_trait; use datafusion::arrow::record_batch::RecordBatch; use datafusion::common::Result; use datafusion::execution::TaskContext; +use datafusion::physical_expr::PhysicalExpr; use datafusion::physical_plan::ExecutionPlan; use datafusion::physical_plan::metrics::{ExecutionPlanMetricsSet, MetricsSet}; use futures::stream::BoxStream; @@ -120,12 +121,29 @@ pub enum WorkerToCoordinatorMsg { /// ensuring metrics are never lost due to early stream termination. /// metrics[i] is the set of metrics for plan node i in pre-order traversal order. TaskMetrics(TaskMetrics), + /// Sends the final dynamic filters used by dynamic filter consumers back to the coorindator + /// for displaying. + TaskCompletedDynamicFilters(TaskCompletedDynamicFilters), /// Load information reported by a task. This information is used for dynamically /// sizing the number of workers involved in a query. LoadInfo(LoadInfo), LoadInfoEos, } +#[derive(Clone, Debug, Default)] +pub struct TaskCompletedDynamicFilters { + /// Final expressions keyed by their DataFusion physical-expression ID. The TaskKey is + /// implicit from the coordinator channel that carried this message. + pub filters: Vec, +} + +#[derive(Clone, Debug)] +pub struct TaskDynamicFilter { + pub expression_id: u64, + /// A `DynamicFilterPhysicalExpr` containing its final predicate and completion state. + pub expression: MaybeEncoded>, +} + #[derive(Clone, Debug)] pub struct TaskMetrics { /// Metrics for a single task's plan nodes in pre-order traversal order. diff --git a/src/stage.rs b/src/stage.rs index cd054c575..5a1389245 100644 --- a/src/stage.rs +++ b/src/stage.rs @@ -1,4 +1,4 @@ -use crate::coordinator::{DistributedExec, MetricsStore}; +use crate::coordinator::DistributedExec; use crate::execution_plans::{DistributedLeafExec, NetworkCoalesceExec}; use crate::metrics::DISTRIBUTED_DATAFUSION_TASK_ID_LABEL; use datafusion::common::{HashMap, Statistics, config_err}; @@ -223,9 +223,7 @@ impl DistributedTaskContext { } } -use crate::{ - DistributedMetricsFormat, NetworkShuffleExec, TaskKey, rewrite_distributed_plan_with_metrics, -}; +use crate::{DistributedMetricsFormat, NetworkShuffleExec, rewrite_distributed_plan_with_metrics}; use crate::{NetworkBoundary, NetworkBoundaryExt}; use datafusion::arrow::datatypes::SchemaRef; use datafusion::common::DataFusionError; @@ -269,7 +267,7 @@ const HORIZONTAL: &str = "─"; // Horizontal line pub fn display_plan_ascii(plan: &dyn ExecutionPlan, show_metrics: bool) -> String { if let Some(plan) = plan.downcast_ref::() { let mut f = String::new(); - display_ascii(plan, Either::Left(plan), 0, show_metrics, &mut f).unwrap(); + display_ascii(Either::Left(plan), 0, show_metrics, &mut f).unwrap(); f } else { match show_metrics { @@ -282,19 +280,18 @@ pub fn display_plan_ascii(plan: &dyn ExecutionPlan, show_metrics: bool) -> Strin } fn display_ascii( - root: &DistributedExec, stage: Either<&DistributedExec, &Stage>, depth: usize, show_metrics: bool, f: &mut String, ) -> std::fmt::Result { let plan = match stage { - Either::Left(distributed_exec) => distributed_exec.children().first().unwrap(), + Either::Left(distributed_exec) => distributed_exec.plan_for_viz_or_base_plan(), Either::Right(stage) => { let Some(plan) = stage.local_plan() else { return write!(f, "StageExec: encoded input plan"); }; - plan + Arc::clone(plan) } }; match stage { @@ -328,10 +325,10 @@ fn display_ascii( HORIZONTAL.repeat(5), stage.num(), HORIZONTAL.repeat(2), - format_tasks_for_stage(stage.task_count(), plan) + format_tasks_for_stage(stage.task_count(), &plan) )?; - if show_metrics && let Some(metrics_store) = &root.metrics_store { - let metrics = gather_stage_header_metrics(stage, metrics_store); + let metrics = stage.metrics(); + if show_metrics && metrics.iter().next().is_some() { write!(f, " ")?; writeln!(f, "{}", format_metrics_by_task(&metrics))?; } else { @@ -341,7 +338,7 @@ fn display_ascii( } let mut plan_str = String::new(); - display_inner_ascii(plan, 0, show_metrics, &mut plan_str)?; + display_inner_ascii(&plan, 0, show_metrics, &mut plan_str)?; let plan_str = plan_str .split('\n') .filter(|v| !v.is_empty()) @@ -356,7 +353,7 @@ fn display_ascii( HORIZONTAL.repeat(50) )?; for input_stage in find_input_stages(plan.as_ref()) { - display_ascii(root, Either::Right(input_stage), depth + 1, show_metrics, f)?; + display_ascii(Either::Right(input_stage), depth + 1, show_metrics, f)?; } Ok(()) } @@ -437,33 +434,6 @@ fn display_inner_distributed_leaf( Ok(()) } -/// Gathers the metrics global to a stage. These metrics are not specific to any plan node, and -/// are instead global to a whole stage. -fn gather_stage_header_metrics(stage: &Stage, metrics_store: &MetricsStore) -> MetricsSet { - let mut task_key = TaskKey { - query_id: stage.query_id(), - stage_id: stage.num(), - task_number: 0, - }; - let mut all_metrics = stage.metrics(); - while let Some(metrics_set) = metrics_store.get(&task_key).map(|v| v.task_metrics) { - for metric in metrics_set.iter() { - let mut labels = metric.labels().to_vec(); - labels.push(Label::new( - DISTRIBUTED_DATAFUSION_TASK_ID_LABEL, - task_key.task_number.to_string(), - )); - all_metrics.push(Arc::new(Metric::new_with_labels( - metric.value().clone(), - metric.partition(), - labels, - ))); - } - task_key.task_number += 1; - } - all_metrics -} - /// Aggregates metrics by (name, task_id), preserving the [DISTRIBUTED_DATAFUSION_TASK_ID_LABEL] /// only. Metrics without a task_id label (ie. non distributed metrics) are aggregated together. /// diff --git a/src/test_utils/metrics.rs b/src/test_utils/metrics.rs index e8e5fe46a..30e889230 100644 --- a/src/test_utils/metrics.rs +++ b/src/test_utils/metrics.rs @@ -7,7 +7,7 @@ use std::sync::Arc; /// Waits until all worker tasks have reported their metrics back via the coordinator channel. pub async fn wait_for_all_metrics(plan: &Arc) { if let Some(dist_exec) = plan.downcast_ref::() { - dist_exec.wait_for_metrics().await; + let _ = dist_exec.wait_for_metrics().await; } } diff --git a/src/worker/impl_coordinator_channel.rs b/src/worker/impl_coordinator_channel.rs index efba61c28..f40272f96 100644 --- a/src/worker/impl_coordinator_channel.rs +++ b/src/worker/impl_coordinator_channel.rs @@ -1,4 +1,5 @@ use crate::common::TreeNodeExt; +use crate::dynamic_filtering::discover_dynamic_filter_consumers; use crate::events::{WorkerPlanRewriteEvent, WorkerPlanRewriteHandlers}; use crate::execution_plans::SamplerExec; use crate::protocol::LocalWorkerContext; @@ -6,14 +7,15 @@ use crate::work_unit_feed::{RemoteWorkUnitFeedRegistry, set_work_unit_received_t use crate::worker::task_data::TaskDataMetrics; use crate::{ CoordinatorToWorkerMsg, DistributedConfig, DistributedExt, DistributedTaskContext, - SetPlanRequest, TaskData, TaskMetrics, Worker, WorkerQueryContext, WorkerToCoordinatorMsg, + MaybeEncoded, SetPlanRequest, TaskCompletedDynamicFilters, TaskData, TaskDynamicFilter, + TaskMetrics, Worker, WorkerQueryContext, WorkerToCoordinatorMsg, }; use datafusion::common::tree_node::TreeNodeRecursion; use datafusion::common::{DataFusionError, Result, exec_datafusion_err}; use datafusion::execution::SessionStateBuilder; use datafusion::physical_plan::ExecutionPlan; use datafusion::prelude::SessionConfig; -use futures::stream::{BoxStream, FuturesUnordered}; +use futures::stream::{BoxStream, FuturesUnordered, select_all}; use futures::{FutureExt, StreamExt, TryStreamExt}; use http::HeaderMap; use std::sync::{Arc, OnceLock}; @@ -40,6 +42,7 @@ impl Worker { } let (metrics_tx, metrics_rx) = oneshot::channel(); + let (dynamic_filters_tx, dynamic_filters_rx) = oneshot::channel(); let mut load_info_rxs = vec![]; let task_data = || async { @@ -58,6 +61,7 @@ impl Worker { let d_cfg = DistributedConfig::from_config_options(cfg.options())?; let shuffle_batch_size = d_cfg.shuffle_batch_size; let collect_metrics = d_cfg.collect_metrics; + let collect_dynamic_filters = d_cfg.collect_dynamic_filters; if shuffle_batch_size != 0 { cfg = cfg.with_batch_size(shuffle_batch_size); } @@ -93,6 +97,10 @@ impl Worker { true => Arc::new(std::sync::Mutex::new(Some(metrics_tx))), false => Arc::new(std::sync::Mutex::new(None)), }, + completed_dynamic_filters_tx: match collect_dynamic_filters { + true => Arc::new(std::sync::Mutex::new(Some(dynamic_filters_tx))), + false => Arc::new(std::sync::Mutex::new(None)), + }, task_data_metrics: Arc::new(TaskDataMetrics::new(request.query_start_time_ns)), }) }; @@ -109,17 +117,18 @@ impl Worker { let mut work_unit_senders = Some(remote_work_unit_feed_registry.senders); let task_data_entries = Arc::clone(&self.task_data_entries); - // This tokio task takes ownership of the `oneshot::Sender` that keeps - // alive the worker->coordinator stream. as soon as this task ends, the runtime metrics - // are send back and the worker->coordinator stream ends. The flow is the following: + // This tokio task takes ownership of the final-report senders that keep the + // worker->coordinator stream alive. As soon as this task ends, the runtime metrics and + // final dynamic filters are sent back and the worker->coordinator stream ends. The flow + // is the following: // 1. The query ends normally, as all Arrow RecordBatches are already streamed. // 2. In DistributedExec::execute(), the end query guard is dropped. // 3. In StageCoordinator::send_plan_task(), `end_stream_notifier` fires and the // coordinator->worker channel is gracefully ended. // 4. The coordinator->worker channel EOS is received by this same function, ending the // while loop inside this `tokio::spawn` below. - // 5. The metrics are send back in the worker->coordinator channel, and then that channel - // is closed. + // 5. The metrics and final dynamic filters are sent back in the worker->coordinator + // channel, and then that channel is closed. #[allow(clippy::disallowed_methods)] tokio::spawn(async move { let mut stream = stream.map_ok(set_work_unit_received_time); @@ -156,7 +165,14 @@ impl Worker { } } + // Send metrics and completed dynamic filters if enabled. + // TODO(#686): handle errors let metrics_tx = task_data.metrics_tx.lock().unwrap().take(); + let dynamic_filters_tx = task_data + .completed_dynamic_filters_tx + .lock() + .unwrap() + .take(); if let Some(Ok(plan)) = task_data.final_plan.get() { let d_ctx = DistributedTaskContext { task_index: key.task_number, @@ -167,6 +183,12 @@ impl Worker { if let Some(metrics_tx) = metrics_tx { send_metrics_via_channel(metrics_tx, plan, d_ctx, task_data_metrics); } + if let Some(dynamic_filters_tx) = dynamic_filters_tx { + // TODO(#686): handle error + let dynamic_filters = + build_task_completed_dynamic_filters(plan).unwrap_or_default(); + let _ = dynamic_filters_tx.send(dynamic_filters); + } } task_data_entries.invalidate(&key).await }); @@ -190,10 +212,44 @@ impl Worker { Some(WorkerToCoordinatorMsg::TaskMetrics(task_metrics)) }); - Ok(futures::stream::select(load_info_stream, metrics_stream) - .map(Ok) - .boxed()) + let dynamic_filters_stream = dynamic_filters_rx.into_stream().filter_map( + async |dynamic_filters_or_channel_dropped| { + let dynamic_filters = dynamic_filters_or_channel_dropped.ok()?; + Some(WorkerToCoordinatorMsg::TaskCompletedDynamicFilters( + dynamic_filters, + )) + }, + ); + + Ok(select_all([ + load_info_stream.boxed(), + metrics_stream.boxed(), + dynamic_filters_stream.boxed(), + ]) + .map(Ok) + .boxed()) + } +} + +/// Finds all consumed dynamic filters for the completed task report. +/// +/// Note that it's possible that a dynamic filter is consumed by the leaf, updated by +/// the producer, then read here, meaning the observed dynamic filter was not +/// necessarily the one applied. This may happen in upstream datafusion as well. +/// Generally this happens because dynamic filter updates happen asynchronously to execution, +/// meaning consumers do not necessarily have to wait for dynamic filters to update / complete +/// before executing. +fn build_task_completed_dynamic_filters( + plan: &Arc, +) -> Result { + let mut filters = vec![]; + for consumer in discover_dynamic_filter_consumers(plan)? { + filters.push(TaskDynamicFilter { + expression_id: consumer.id, + expression: MaybeEncoded::Decoded(consumer.expression), + }); } + Ok(TaskCompletedDynamicFilters { filters }) } /// Collects metrics from the plan in pre-order traversal order and sends them via the diff --git a/src/worker/task_data.rs b/src/worker/task_data.rs index 2ff377173..81ad44617 100644 --- a/src/worker/task_data.rs +++ b/src/worker/task_data.rs @@ -1,6 +1,6 @@ use crate::common::OnceLockResult; use crate::common::now_ns; -use crate::{MaxLatencyMetric, ProducerHead, TaskMetrics}; +use crate::{MaxLatencyMetric, ProducerHead, TaskCompletedDynamicFilters, TaskMetrics}; use datafusion::common::{DataFusionError, Result}; use datafusion::execution::TaskContext; use datafusion::physical_plan::ExecutionPlan; @@ -22,6 +22,10 @@ pub struct TaskData { /// `Option::take`) when the coordinator channel reaches EOS, sending the collected metrics /// back to the coordinator through the `CoordinatorChannel` side channel. pub(super) metrics_tx: Arc>>>, + /// Sender half of the completed dynamic-filter channel. It is absent when the user does not + /// want to display dynamic filters. + pub(super) completed_dynamic_filters_tx: + Arc>>>, /// Metrics related to the execution of a task within a stage. This metrics, instead of being /// associated to a specific node, they are global to the task, like the time at which the plan /// was fed by the coordinator to the worker. diff --git a/src/worker/test_utils/worker_handles.rs b/src/worker/test_utils/worker_handles.rs index 4fe19c5c8..13e8b7eb7 100644 --- a/src/worker/test_utils/worker_handles.rs +++ b/src/worker/test_utils/worker_handles.rs @@ -207,12 +207,14 @@ pub async fn register_plan_on_worker( .get_with(task_key, async { Default::default() }) .await; let (metrics_tx, _metrics_rx) = tokio::sync::oneshot::channel(); + let (dynamic_filters_tx, _dynamic_filters_rx) = tokio::sync::oneshot::channel(); swmr_task_data .write(Ok(TaskData { task_ctx, base_plan: plan, final_plan: Default::default(), metrics_tx: Arc::new(std::sync::Mutex::new(Some(metrics_tx))), + completed_dynamic_filters_tx: Arc::new(std::sync::Mutex::new(Some(dynamic_filters_tx))), task_data_metrics: Arc::new(TaskDataMetrics::new(0)), })) .expect("failed to write to task data"); diff --git a/tests/dynamic_filtering/aggregates.rs b/tests/dynamic_filtering/aggregates.rs new file mode 100644 index 000000000..41c3cc4f6 --- /dev/null +++ b/tests/dynamic_filtering/aggregates.rs @@ -0,0 +1,71 @@ +#[cfg(test)] +mod tests { + use crate::common::TestQuery; + use datafusion::common::Result; + use datafusion_distributed::assert_snapshot; + + /// A partial aggregate updates a data source in the same stage. + #[tokio::test] + async fn local_dynamic_filters() -> Result<()> { + let display = TestQuery::new( + r#" + SELECT MIN("MinTemp") + FROM weather + WHERE "RainToday" = 'Yes' + "#, + ) + .execute() + .await?; + assert_snapshot!(display, @" + ┌───── DistributedExec + │ AggregateExec: mode=Final, gby=[], aggr=[min(weather.MinTemp)] + │ CoalescePartitionsExec + │ [Stage 1] => NetworkCoalesceExec: output_partitions=6, input_tasks=2 + └────────────────────────────────────────────────── + ┌───── Stage 1 ── tasks=2, partitions=6 + │ AggregateExec: mode=Partial, gby=[], aggr=[min(weather.MinTemp)] + │ FilterExec: RainToday@1 = Yes, projection=[MinTemp@0] + │ DistributedLeafExec: + │ t0: DataSourceExec: file_groups={3 groups: [[/testdata/weather/result-000000.parquet:..], [/testdata/weather/result-000000.parquet:.., /testdata/weather/result-000001.parquet:..], [/testdata/weather/result-000002.parquet:..]]}, projection=[MinTemp, RainToday], file_type=parquet, predicate=RainToday@19 = Yes AND DynamicFilter [ MinTemp@0 < 4.3 ], dynamic_rg_pruning=eligible, pruning_predicate=RainToday_null_count@2 != row_count@3 AND RainToday_min@0 <= Yes AND Yes <= RainToday_max@1 AND MinTemp_null_count@5 != row_count@3 AND MinTemp_min@4 < 4.3, required_guarantees=[RainToday in (Yes)] + │ t1: DataSourceExec: file_groups={3 groups: [[/testdata/weather/result-000000.parquet:..], [/testdata/weather/result-000001.parquet:.., /testdata/weather/result-000002.parquet:..], [/testdata/weather/result-000002.parquet:..]]}, projection=[MinTemp, RainToday], file_type=parquet, predicate=RainToday@19 = Yes AND DynamicFilter [ MinTemp@0 < -1.6 ], dynamic_rg_pruning=eligible, pruning_predicate=RainToday_null_count@2 != row_count@3 AND RainToday_min@0 <= Yes AND Yes <= RainToday_max@1 AND MinTemp_null_count@5 != row_count@3 AND MinTemp_min@4 < -1.6, required_guarantees=[RainToday in (Yes)] + └────────────────────────────────────────────────── + "); + Ok(()) + } + + /// A partial aggregate does not yet update a data source across a shuffle. + #[tokio::test] + async fn remote_dynamic_filters() -> Result<()> { + let display = TestQuery::new( + r#" + SELECT MIN(key) + FROM ( + SELECT DISTINCT "MinTemp" AS key + FROM weather + ) + "#, + ) + .execute() + .await?; + assert_snapshot!(display, @" + ┌───── DistributedExec + │ AggregateExec: mode=Final, gby=[], aggr=[min(key)] + │ CoalescePartitionsExec + │ [Stage 2] => NetworkCoalesceExec: output_partitions=6, input_tasks=2 + └────────────────────────────────────────────────── + ┌───── Stage 2 ── tasks=2, partitions=3 + │ AggregateExec: mode=Partial, gby=[], aggr=[min(key)] + │ AggregateExec: mode=FinalPartitioned, gby=[key@0 as key], aggr=[] + │ [Stage 1] => NetworkShuffleExec: output_partitions=3, input_tasks=2 + └────────────────────────────────────────────────── + ┌───── Stage 1 ── tasks=2, partitions=6 + │ RepartitionExec: partitioning=Hash([key@0], 6), input_partitions=3 + │ AggregateExec: mode=Partial, gby=[key@0 as key], aggr=[] + │ DistributedLeafExec: + │ t0: DataSourceExec: file_groups={3 groups: [[/testdata/weather/result-000000.parquet:..], [/testdata/weather/result-000000.parquet:.., /testdata/weather/result-000001.parquet:..], [/testdata/weather/result-000002.parquet:..]]}, projection=[MinTemp@0 as key], file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible + │ t1: DataSourceExec: file_groups={3 groups: [[/testdata/weather/result-000000.parquet:..], [/testdata/weather/result-000001.parquet:.., /testdata/weather/result-000002.parquet:..], [/testdata/weather/result-000002.parquet:..]]}, projection=[MinTemp@0 as key], file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible + └────────────────────────────────────────────────── + "); + Ok(()) + } +} diff --git a/tests/dynamic_filtering/collect_left_join.rs b/tests/dynamic_filtering/collect_left_join.rs new file mode 100644 index 000000000..00105d363 --- /dev/null +++ b/tests/dynamic_filtering/collect_left_join.rs @@ -0,0 +1,177 @@ +#[cfg(test)] +mod tests { + use crate::common::TestQuery; + use datafusion::common::Result; + use datafusion_distributed::assert_snapshot; + + /// A CollectLeft HashJoinExec producer propagates identical updates to local consumers. + #[tokio::test] + async fn local_dynamic_filters() -> Result<()> { + let display = TestQuery::new( + r#" + SELECT COUNT(*) + FROM ( + SELECT DISTINCT "RainToday" AS key + FROM weather + ) build + JOIN weather probe ON build.key = probe."RainToday" + "#, + ) + .with_broadcast_joins() + .execute() + .await?; + assert_snapshot!(display, @r" + ┌───── DistributedExec + │ ProjectionExec: expr=[count(Int64(1))@0 as count(*)] + │ AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] + │ CoalescePartitionsExec + │ [Stage 3] => NetworkCoalesceExec: output_partitions=6, input_tasks=2 + └────────────────────────────────────────────────── + ┌───── Stage 3 ── tasks=2, partitions=6 + │ AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] + │ HashJoinExec: mode=CollectLeft, join_type=RightSemi, on=[(key@0, RainToday@0)], projection=[] + │ CoalescePartitionsExec + │ [Stage 2] => NetworkBroadcastExec: partitions_per_consumer=3, stage_partitions=6, input_tasks=2 + │ DistributedLeafExec: + │ t0: DataSourceExec: file_groups={3 groups: [[/testdata/weather/result-000000.parquet:..], [/testdata/weather/result-000000.parquet:.., /testdata/weather/result-000001.parquet:..], [/testdata/weather/result-000002.parquet:..]]}, projection=[RainToday], file_type=parquet, predicate=DynamicFilter [ RainToday@19 >= No AND RainToday@19 <= Yes AND RainToday@19 IN (SET) ([]) ], dynamic_rg_pruning=eligible, pruning_predicate=RainToday_null_count@1 != row_count@2 AND RainToday_max@0 >= No AND RainToday_null_count@1 != row_count@2 AND RainToday_min@3 <= Yes AND (RainToday_null_count@1 != row_count@2 AND RainToday_min@3 <= Yes AND Yes <= RainToday_max@0 OR RainToday_null_count@1 != row_count@2 AND RainToday_min@3 <= No AND No <= RainToday_max@0), required_guarantees=[RainToday in (No, Yes)] + │ t1: DataSourceExec: file_groups={3 groups: [[/testdata/weather/result-000000.parquet:..], [/testdata/weather/result-000001.parquet:.., /testdata/weather/result-000002.parquet:..], [/testdata/weather/result-000002.parquet:..]]}, projection=[RainToday], file_type=parquet, predicate=DynamicFilter [ RainToday@19 >= No AND RainToday@19 <= Yes AND RainToday@19 IN (SET) ([]) ], dynamic_rg_pruning=eligible, pruning_predicate=RainToday_null_count@1 != row_count@2 AND RainToday_max@0 >= No AND RainToday_null_count@1 != row_count@2 AND RainToday_min@3 <= Yes AND (RainToday_null_count@1 != row_count@2 AND RainToday_min@3 <= Yes AND Yes <= RainToday_max@0 OR RainToday_null_count@1 != row_count@2 AND RainToday_min@3 <= No AND No <= RainToday_max@0), required_guarantees=[RainToday in (No, Yes)] + └────────────────────────────────────────────────── + ┌───── Stage 2 ── tasks=2, partitions=12 + │ BroadcastExec: input_partitions=3, consumer_tasks=2, output_partitions=6 + │ AggregateExec: mode=FinalPartitioned, gby=[key@0 as key], aggr=[] + │ [Stage 1] => NetworkShuffleExec: output_partitions=3, input_tasks=2 + └────────────────────────────────────────────────── + ┌───── Stage 1 ── tasks=2, partitions=6 + │ RepartitionExec: partitioning=Hash([key@0], 6), input_partitions=3 + │ AggregateExec: mode=Partial, gby=[key@0 as key], aggr=[] + │ DistributedLeafExec: + │ t0: DataSourceExec: file_groups={3 groups: [[/testdata/weather/result-000000.parquet:..], [/testdata/weather/result-000000.parquet:.., /testdata/weather/result-000001.parquet:..], [/testdata/weather/result-000002.parquet:..]]}, projection=[RainToday@19 as key], file_type=parquet + │ t1: DataSourceExec: file_groups={3 groups: [[/testdata/weather/result-000000.parquet:..], [/testdata/weather/result-000001.parquet:.., /testdata/weather/result-000002.parquet:..], [/testdata/weather/result-000002.parquet:..]]}, projection=[RainToday@19 as key], file_type=parquet + └────────────────────────────────────────────────── + "); + Ok(()) + } + + /// A CollectLeft HashJoinExec does not propagate dynamic filters to a remote consumer. + #[tokio::test] + async fn remote_dynamic_filters() -> Result<()> { + let display = TestQuery::new( + r#" + SELECT COUNT(*) + FROM ( + SELECT DISTINCT "RainToday" AS key + FROM weather + ) build + RIGHT SEMI JOIN ( + SELECT DISTINCT "RainToday" AS key + FROM weather + ) probe ON build.key = probe.key + "#, + ) + .with_broadcast_joins() + .execute() + .await?; + assert_snapshot!(display, @r" + ┌───── DistributedExec + │ ProjectionExec: expr=[count(Int64(1))@0 as count(*)] + │ AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] + │ CoalescePartitionsExec + │ [Stage 4] => NetworkCoalesceExec: output_partitions=6, input_tasks=2 + └────────────────────────────────────────────────── + ┌───── Stage 4 ── tasks=2, partitions=3 + │ AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] + │ HashJoinExec: mode=CollectLeft, join_type=RightSemi, on=[(key@0, key@0)], projection=[] + │ CoalescePartitionsExec + │ [Stage 2] => NetworkBroadcastExec: partitions_per_consumer=3, stage_partitions=6, input_tasks=2 + │ AggregateExec: mode=FinalPartitioned, gby=[key@0 as key], aggr=[] + │ [Stage 3] => NetworkShuffleExec: output_partitions=3, input_tasks=2 + └────────────────────────────────────────────────── + ┌───── Stage 2 ── tasks=2, partitions=12 + │ BroadcastExec: input_partitions=3, consumer_tasks=2, output_partitions=6 + │ AggregateExec: mode=FinalPartitioned, gby=[key@0 as key], aggr=[] + │ [Stage 1] => NetworkShuffleExec: output_partitions=3, input_tasks=2 + └────────────────────────────────────────────────── + ┌───── Stage 1 ── tasks=2, partitions=6 + │ RepartitionExec: partitioning=Hash([key@0], 6), input_partitions=3 + │ AggregateExec: mode=Partial, gby=[key@0 as key], aggr=[] + │ DistributedLeafExec: + │ t0: DataSourceExec: file_groups={3 groups: [[/testdata/weather/result-000000.parquet:..], [/testdata/weather/result-000000.parquet:.., /testdata/weather/result-000001.parquet:..], [/testdata/weather/result-000002.parquet:..]]}, projection=[RainToday@19 as key], file_type=parquet + │ t1: DataSourceExec: file_groups={3 groups: [[/testdata/weather/result-000000.parquet:..], [/testdata/weather/result-000001.parquet:.., /testdata/weather/result-000002.parquet:..], [/testdata/weather/result-000002.parquet:..]]}, projection=[RainToday@19 as key], file_type=parquet + └────────────────────────────────────────────────── + ┌───── Stage 3 ── tasks=2, partitions=6 + │ RepartitionExec: partitioning=Hash([key@0], 6), input_partitions=3 + │ AggregateExec: mode=Partial, gby=[key@0 as key], aggr=[] + │ DistributedLeafExec: + │ t0: DataSourceExec: file_groups={3 groups: [[/testdata/weather/result-000000.parquet:..], [/testdata/weather/result-000000.parquet:.., /testdata/weather/result-000001.parquet:..], [/testdata/weather/result-000002.parquet:..]]}, projection=[RainToday@19 as key], file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible + │ t1: DataSourceExec: file_groups={3 groups: [[/testdata/weather/result-000000.parquet:..], [/testdata/weather/result-000001.parquet:.., /testdata/weather/result-000002.parquet:..], [/testdata/weather/result-000002.parquet:..]]}, projection=[RainToday@19 as key], file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible + └────────────────────────────────────────────────── + "); + Ok(()) + } + + #[tokio::test] + async fn union_probe_deduplicates_dynamic_filters() -> Result<()> { + let display = TestQuery::new( + r#" + SELECT COUNT(*) + FROM ( + SELECT DISTINCT "MinTemp" AS key + FROM weather + ) build + JOIN ( + SELECT "MinTemp" FROM weather + UNION ALL + SELECT CAST(-1000.0 AS DOUBLE) AS "MinTemp" + UNION ALL + SELECT "MinTemp" FROM weather + UNION ALL + SELECT CAST(-1001.0 AS DOUBLE) AS "MinTemp" + UNION ALL + SELECT CAST(-1002.0 AS DOUBLE) AS "MinTemp" + ) probe ON build.key = probe."MinTemp" + "#, + ) + .with_broadcast_joins() + // Indirectly forces the union to put c0 and c2 on the same task. We would like to test + // that consumers of a dynamic filter on the same node get the same filter. + // + // If we don't do this, inject_network_boundaries splits them up to spread out the load. + .with_one_task_per_leaf() + .execute() + .await?; + assert_snapshot!(display, @r" + ┌───── DistributedExec + │ ProjectionExec: expr=[count(Int64(1))@0 as count(*)] + │ AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] + │ CoalescePartitionsExec + │ [Stage 2] => NetworkCoalesceExec: output_partitions=14, input_tasks=2 + └────────────────────────────────────────────────── + ┌───── Stage 2 ── tasks=2, partitions=14 + │ AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] + │ HashJoinExec: mode=CollectLeft, join_type=RightSemi, on=[(key@0, MinTemp@0)], projection=[] + │ CoalescePartitionsExec + │ [Stage 1] => NetworkBroadcastExec: partitions_per_consumer=3, stage_partitions=6, input_tasks=1 + │ DistributedUnionExec: t0:[c0, c2, c4] t1:[c1, c3] + │ DistributedLeafExec: + │ t0: DataSourceExec: file_groups={3 groups: [[/testdata/weather/result-000000.parquet:..], [/testdata/weather/result-000000.parquet:.., /testdata/weather/result-000001.parquet:.., /testdata/weather/result-000002.parquet:..], [/testdata/weather/result-000002.parquet:..]]}, projection=[MinTemp], file_type=parquet, predicate=DynamicFilter [ MinTemp@0 >= -5.3 AND MinTemp@0 <= 20.9 AND true ], dynamic_rg_pruning=eligible, pruning_predicate=MinTemp_null_count@1 != row_count@2 AND MinTemp_max@0 >= -5.3 AND MinTemp_null_count@1 != row_count@2 AND MinTemp_min@3 <= 20.9, required_guarantees=[] + │ ProjectionExec: expr=[CAST(-1000 AS Float64) as MinTemp] + │ PlaceholderRowExec + │ DistributedLeafExec: + │ t0: DataSourceExec: file_groups={3 groups: [[/testdata/weather/result-000000.parquet:..], [/testdata/weather/result-000000.parquet:.., /testdata/weather/result-000001.parquet:.., /testdata/weather/result-000002.parquet:..], [/testdata/weather/result-000002.parquet:..]]}, projection=[MinTemp], file_type=parquet, predicate=DynamicFilter [ MinTemp@0 >= -5.3 AND MinTemp@0 <= 20.9 AND true ], dynamic_rg_pruning=eligible, pruning_predicate=MinTemp_null_count@1 != row_count@2 AND MinTemp_max@0 >= -5.3 AND MinTemp_null_count@1 != row_count@2 AND MinTemp_min@3 <= 20.9, required_guarantees=[] + │ ProjectionExec: expr=[CAST(-1001 AS Float64) as MinTemp] + │ PlaceholderRowExec + │ ProjectionExec: expr=[CAST(-1002 AS Float64) as MinTemp] + │ PlaceholderRowExec + └────────────────────────────────────────────────── + ┌───── Stage 1 ── tasks=1, partitions=6 + │ BroadcastExec: input_partitions=3, consumer_tasks=2, output_partitions=6 + │ AggregateExec: mode=FinalPartitioned, gby=[key@0 as key], aggr=[] + │ RepartitionExec: partitioning=Hash([key@0], 3), input_partitions=3 + │ AggregateExec: mode=Partial, gby=[key@0 as key], aggr=[] + │ DistributedLeafExec: + │ t0: DataSourceExec: file_groups={3 groups: [[/testdata/weather/result-000000.parquet:..], [/testdata/weather/result-000000.parquet:.., /testdata/weather/result-000001.parquet:.., /testdata/weather/result-000002.parquet:..], [/testdata/weather/result-000002.parquet:..]]}, projection=[MinTemp@0 as key], file_type=parquet + └────────────────────────────────────────────────── + "); + Ok(()) + } +} diff --git a/tests/dynamic_filtering/common.rs b/tests/dynamic_filtering/common.rs new file mode 100644 index 000000000..1ecb67c7b --- /dev/null +++ b/tests/dynamic_filtering/common.rs @@ -0,0 +1,165 @@ +use datafusion::arrow::datatypes::DataType; +use datafusion::common::{Result, ScalarValue, SplitPoint}; +use datafusion::datasource::file_format::parquet::ParquetFormat; +use datafusion::datasource::listing::{ + ListingOptions, ListingTable, ListingTableConfig, ListingTableUrl, +}; +use datafusion::logical_expr::{Partitioning, RangePartitioning}; +use datafusion::physical_plan::collect; +use datafusion::prelude::{SessionContext, col}; +use datafusion_distributed::test_utils::localhost::start_localhost_context; +use datafusion_distributed::test_utils::parquet::register_parquet_tables; +use datafusion_distributed::test_utils::routing::url_emitter_route_tasks; +use datafusion_distributed::{ + DefaultSessionBuilder, DistributedExt, display_plan_ascii, + rewrite_distributed_plan_with_dynamic_filters, +}; +use std::sync::Arc; + +pub(crate) struct TestQuery<'a> { + sql: &'a str, + expected_rows: usize, + broadcast_joins: bool, + one_task_per_leaf: bool, + collect_dynamic_filters: bool, +} + +impl<'a> TestQuery<'a> { + pub(crate) fn new(sql: &'a str) -> Self { + Self { + sql, + expected_rows: 1, + broadcast_joins: false, + one_task_per_leaf: false, + collect_dynamic_filters: true, + } + } + + /// Assert the number of rows after the query runs. + pub(crate) fn with_expected_rows(mut self, expected_rows: usize) -> Self { + self.expected_rows = expected_rows; + self + } + + /// Forces collect left joins and enables distributed broadcast joins. + pub(crate) fn with_broadcast_joins(mut self) -> Self { + self.broadcast_joins = true; + self + } + + /// Sets the desired task count to 1. + pub(crate) fn with_one_task_per_leaf(mut self) -> Self { + self.one_task_per_leaf = true; + self + } + + /// Disables dynamic filter collection. + pub(crate) fn without_dynamic_filter_collection(mut self) -> Self { + self.collect_dynamic_filters = false; + self + } + + pub(crate) async fn execute(self) -> Result { + let (ctx, _guard, _) = start_localhost_context(2, DefaultSessionBuilder).await; + let mut ctx = ctx + .with_distributed_broadcast_joins(self.broadcast_joins)? + .with_distributed_dynamic_filter_collection(self.collect_dynamic_filters)?; + if self.one_task_per_leaf { + ctx = ctx.with_distributed_desired_task_count_handler(1usize); + } + if !self.broadcast_joins { + // Force partitioned hash joins. + let state = ctx.state_ref(); + let mut state = state.write(); + let optimizer = &mut state.config_mut().options_mut().optimizer; + optimizer.hash_join_single_partition_threshold = 0; + optimizer.hash_join_single_partition_threshold_rows = 0; + } + register_parquet_tables(&ctx).await?; + execute_query_and_display( + &ctx, + self.sql, + self.expected_rows, + self.collect_dynamic_filters, + ) + .await + } +} + +pub(crate) async fn execute_range_partitioned_query( + sql: &str, + expected_rows: usize, +) -> Result { + let (ctx, _guard, _) = start_localhost_context(3, DefaultSessionBuilder).await; + let ctx = ctx + .with_distributed_broadcast_joins(false)? + .with_distributed_desired_task_count_handler(2usize) + .with_distributed_route_tasks_handler(url_emitter_route_tasks); + { + let state = ctx.state_ref(); + let mut state = state.write(); + let options = state.config_mut().options_mut(); + options.execution.target_partitions = 2; + options.optimizer.hash_join_single_partition_threshold = 0; + options.optimizer.hash_join_single_partition_threshold_rows = 0; + } + + register_range_partitioned_table(&ctx, "dim", "testdata/join/parquet/dim", "d_dkey").await?; + register_range_partitioned_table(&ctx, "fact", "testdata/join/parquet/fact", "f_dkey").await?; + + execute_query_and_display(&ctx, sql, expected_rows, true).await +} + +async fn register_range_partitioned_table( + ctx: &SessionContext, + name: &str, + path: &str, + partition_column: &str, +) -> Result<()> { + let table_url = ListingTableUrl::parse(path)?; + let output_partitioning = Partitioning::Range(RangePartitioning::try_new( + vec![col(partition_column).sort(true, false)], + vec![SplitPoint::new(vec![ScalarValue::Utf8(Some( + "C".to_string(), + ))])], + )?); + let options = ListingOptions::new(Arc::new(ParquetFormat::default())) + .with_table_partition_cols(vec![(partition_column.to_string(), DataType::Utf8)]) + .with_output_partitioning(Some(output_partitioning)); + let config = ListingTableConfig::new(table_url) + .with_listing_options(options) + .infer_schema(&ctx.state()) + .await?; + ctx.register_table(name, Arc::new(ListingTable::try_new(config)?))?; + Ok(()) +} + +async fn execute_query_and_display( + ctx: &SessionContext, + sql: &str, + expected_rows: usize, + collect_dynamic_filters: bool, +) -> Result { + let plan = ctx.sql(sql).await?.create_physical_plan().await?; + let task_ctx = ctx.task_ctx(); + + let results = collect(Arc::clone(&plan), Arc::clone(&task_ctx)).await?; + assert_eq!( + results.iter().map(|batch| batch.num_rows()).sum::(), + expected_rows + ); + + let original_display = display_plan_ascii(plan.as_ref(), false); + let plan_with_dynamic_filters = + rewrite_distributed_plan_with_dynamic_filters(Arc::clone(&plan), &task_ctx).await?; + assert_eq!( + Arc::ptr_eq(&plan, &plan_with_dynamic_filters), + !collect_dynamic_filters + ); + assert_eq!(display_plan_ascii(plan.as_ref(), false), original_display); + + Ok(display_plan_ascii( + plan_with_dynamic_filters.as_ref(), + false, + )) +} diff --git a/tests/dynamic_filtering/config.rs b/tests/dynamic_filtering/config.rs new file mode 100644 index 000000000..0a34337d0 --- /dev/null +++ b/tests/dynamic_filtering/config.rs @@ -0,0 +1,25 @@ +#[cfg(test)] +mod tests { + use crate::common::TestQuery; + use datafusion::common::Result; + + #[tokio::test] + async fn completed_filter_collection_can_be_disabled() -> Result<()> { + TestQuery::new( + r#" + SELECT COUNT(*) + FROM ( + SELECT DISTINCT "RainToday" AS key + FROM weather + ) build + JOIN weather probe ON build.key = probe."RainToday" + "#, + ) + .without_dynamic_filter_collection() + // execute() asserts that dynamic filters were not collected for display by asserting + // that the plan is not rewritten. + .execute() + .await?; + Ok(()) + } +} diff --git a/tests/dynamic_filtering/main.rs b/tests/dynamic_filtering/main.rs new file mode 100644 index 000000000..f7bdf8839 --- /dev/null +++ b/tests/dynamic_filtering/main.rs @@ -0,0 +1,6 @@ +mod aggregates; +mod collect_left_join; +mod common; +mod config; +mod partitioned_join; +mod sorts; diff --git a/tests/dynamic_filtering/partitioned_join.rs b/tests/dynamic_filtering/partitioned_join.rs new file mode 100644 index 000000000..b0c530c4c --- /dev/null +++ b/tests/dynamic_filtering/partitioned_join.rs @@ -0,0 +1,92 @@ +#[cfg(test)] +mod tests { + use crate::common::{TestQuery, execute_range_partitioned_query}; + use datafusion::common::Result; + use datafusion_distributed::assert_snapshot; + + /// A Partitioned HashJoinExec propagates dynamic filters to local consumers. + #[tokio::test] + async fn local_dynamic_filters() -> Result<()> { + let display = execute_range_partitioned_query( + r#" + SELECT d.env, COUNT(*) AS n + FROM dim d + JOIN fact f ON d.d_dkey = f.f_dkey + WHERE d.service = 'log' + GROUP BY d.env + "#, + 2, + ) + .await?; + assert_snapshot!(display, @r" + ┌───── DistributedExec + │ CoalescePartitionsExec + │ [Stage 2] => NetworkCoalesceExec: output_partitions=4, input_tasks=2 + └────────────────────────────────────────────────── + ┌───── Stage 2 ── tasks=2, partitions=2 + │ ProjectionExec: expr=[env@0 as env, count(Int64(1))@1 as n] + │ AggregateExec: mode=FinalPartitioned, gby=[env@0 as env], aggr=[count(Int64(1))] + │ [Stage 1] => NetworkShuffleExec: output_partitions=2, input_tasks=2 + └────────────────────────────────────────────────── + ┌───── Stage 1 ── tasks=2, partitions=4 + │ RepartitionExec: partitioning=Hash([env@0], 4), input_partitions=2 + │ AggregateExec: mode=Partial, gby=[env@0 as env], aggr=[count(Int64(1))] + │ HashJoinExec: mode=Partitioned, join_type=Inner, on=[(d_dkey@1, f_dkey@0)], projection=[env@0] + │ FilterExec: service@1 = log, projection=[env@0, d_dkey@2] + │ DistributedLeafExec: + │ t0: DataSourceExec: file_groups={2 groups: [[/testdata/join/parquet/dim/d_dkey=A/data0.parquet], [/testdata/join/parquet/dim/d_dkey=C/data0.parquet]]}, projection=[env, service, d_dkey], output_partitioning=Range([d_dkey@2 ASC NULLS LAST], [(C)], 2), file_type=parquet, predicate=service@1 = log, pruning_predicate=service_null_count@2 != row_count@3 AND service_min@0 <= log AND log <= service_max@1, required_guarantees=[service in (log)] + │ t1: DataSourceExec: file_groups={2 groups: [[/testdata/join/parquet/dim/d_dkey=B/data0.parquet], [/testdata/join/parquet/dim/d_dkey=D/data0.parquet]]}, projection=[env, service, d_dkey], output_partitioning=Range([d_dkey@2 ASC NULLS LAST], [(C)], 2), file_type=parquet, predicate=service@1 = log, pruning_predicate=service_null_count@2 != row_count@3 AND service_min@0 <= log AND log <= service_max@1, required_guarantees=[service in (log)] + │ DistributedLeafExec: + │ t0: DataSourceExec: file_groups={2 groups: [[/testdata/join/parquet/fact/f_dkey=A/data0.parquet], [/testdata/join/parquet/fact/f_dkey=C/data0.parquet]]}, projection=[f_dkey], output_partitioning=Range([f_dkey@0 ASC NULLS LAST], [(C)], 2), file_type=parquet, predicate=DynamicFilter [ f_dkey@2 >= A AND f_dkey@2 <= A AND f_dkey@2 IN (SET) ([]) ], dynamic_rg_pruning=eligible, pruning_predicate=f_dkey_null_count@1 != row_count@2 AND f_dkey_max@0 >= A AND f_dkey_null_count@1 != row_count@2 AND f_dkey_min@3 <= A AND f_dkey_null_count@1 != row_count@2 AND f_dkey_min@3 <= A AND A <= f_dkey_max@0, required_guarantees=[f_dkey in (A)] + │ t1: DataSourceExec: file_groups={2 groups: [[/testdata/join/parquet/fact/f_dkey=B/data0.parquet], [/testdata/join/parquet/fact/f_dkey=D/data0.parquet]]}, projection=[f_dkey], output_partitioning=Range([f_dkey@0 ASC NULLS LAST], [(C)], 2), file_type=parquet, predicate=DynamicFilter [ f_dkey@2 >= B AND f_dkey@2 <= B AND f_dkey@2 IN (SET) ([]) ], dynamic_rg_pruning=eligible, pruning_predicate=f_dkey_null_count@1 != row_count@2 AND f_dkey_max@0 >= B AND f_dkey_null_count@1 != row_count@2 AND f_dkey_min@3 <= B AND f_dkey_null_count@1 != row_count@2 AND f_dkey_min@3 <= B AND B <= f_dkey_max@0, required_guarantees=[f_dkey in (B)] + └────────────────────────────────────────────────── + "); + Ok(()) + } + + /// A Partitioned HashJoinExec does not propagate dynamic filters to a remote consumer. + #[tokio::test] + async fn remote_dynamic_filters() -> Result<()> { + let display = TestQuery::new( + r#" + SELECT COUNT(*) + FROM ( + SELECT DISTINCT "RainToday" AS key + FROM weather + ) build + JOIN weather probe ON build.key = probe."RainToday" + "#, + ) + .execute() + .await?; + assert_snapshot!(display, @" + ┌───── DistributedExec + │ ProjectionExec: expr=[count(Int64(1))@0 as count(*)] + │ AggregateExec: mode=Final, gby=[], aggr=[count(Int64(1))] + │ CoalescePartitionsExec + │ [Stage 3] => NetworkCoalesceExec: output_partitions=6, input_tasks=2 + └────────────────────────────────────────────────── + ┌───── Stage 3 ── tasks=2, partitions=3 + │ AggregateExec: mode=Partial, gby=[], aggr=[count(Int64(1))] + │ HashJoinExec: mode=Partitioned, join_type=RightSemi, on=[(key@0, RainToday@0)], projection=[] + │ AggregateExec: mode=FinalPartitioned, gby=[key@0 as key], aggr=[] + │ [Stage 1] => NetworkShuffleExec: output_partitions=3, input_tasks=2 + │ [Stage 2] => NetworkShuffleExec: output_partitions=3, input_tasks=2 + └────────────────────────────────────────────────── + ┌───── Stage 1 ── tasks=2, partitions=6 + │ RepartitionExec: partitioning=Hash([key@0], 6), input_partitions=3 + │ AggregateExec: mode=Partial, gby=[key@0 as key], aggr=[] + │ DistributedLeafExec: + │ t0: DataSourceExec: file_groups={3 groups: [[/testdata/weather/result-000000.parquet:..], [/testdata/weather/result-000000.parquet:.., /testdata/weather/result-000001.parquet:..], [/testdata/weather/result-000002.parquet:..]]}, projection=[RainToday@19 as key], file_type=parquet + │ t1: DataSourceExec: file_groups={3 groups: [[/testdata/weather/result-000000.parquet:..], [/testdata/weather/result-000001.parquet:.., /testdata/weather/result-000002.parquet:..], [/testdata/weather/result-000002.parquet:..]]}, projection=[RainToday@19 as key], file_type=parquet + └────────────────────────────────────────────────── + ┌───── Stage 2 ── tasks=2, partitions=6 + │ RepartitionExec: partitioning=Hash([RainToday@0], 6), input_partitions=3 + │ DistributedLeafExec: + │ t0: DataSourceExec: file_groups={3 groups: [[/testdata/weather/result-000000.parquet:..], [/testdata/weather/result-000000.parquet:.., /testdata/weather/result-000001.parquet:..], [/testdata/weather/result-000002.parquet:..]]}, projection=[RainToday], file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible + │ t1: DataSourceExec: file_groups={3 groups: [[/testdata/weather/result-000000.parquet:..], [/testdata/weather/result-000001.parquet:.., /testdata/weather/result-000002.parquet:..], [/testdata/weather/result-000002.parquet:..]]}, projection=[RainToday], file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible + └────────────────────────────────────────────────── + "); + Ok(()) + } +} diff --git a/tests/dynamic_filtering/sorts.rs b/tests/dynamic_filtering/sorts.rs new file mode 100644 index 000000000..48ad97772 --- /dev/null +++ b/tests/dynamic_filtering/sorts.rs @@ -0,0 +1,79 @@ +#[cfg(test)] +mod tests { + use crate::common::TestQuery; + use datafusion::common::Result; + use datafusion_distributed::{assert_snapshot, test_utils::insta::settings}; + + /// A TopK SortExec applies dynamic filters to local data sources. + #[tokio::test] + async fn local_dynamic_filters() -> Result<()> { + let display = TestQuery::new( + r#" + SELECT "MinTemp" + FROM weather + ORDER BY "MinTemp" DESC + LIMIT 10 + "#, + ) + .with_expected_rows(10) + .execute() + .await?; + let mut settings = settings(); + settings.add_filter( + r"(DynamicFilter \[[^\]\n]*? > )-?\d+(?:\.\d+)?( \])", + "${1}${2}", + ); + settings.add_filter(r"(_max@\d+ > )-?\d+(?:\.\d+)?", "${1}"); + settings.bind(|| assert_snapshot!(display, @" + ┌───── DistributedExec + │ SortPreservingMergeExec: [MinTemp@0 DESC], fetch=10 + │ [Stage 1] => NetworkCoalesceExec: output_partitions=6, input_tasks=2 + └────────────────────────────────────────────────── + ┌───── Stage 1 ── tasks=2, partitions=6 + │ SortExec: TopK(fetch=10), expr=[MinTemp@0 DESC], preserve_partitioning=[true] + │ DistributedLeafExec: + │ t0: DataSourceExec: file_groups={3 groups: [[/testdata/weather/result-000000.parquet:..], [/testdata/weather/result-000000.parquet:.., /testdata/weather/result-000001.parquet:..], [/testdata/weather/result-000002.parquet:..]]}, projection=[MinTemp], file_type=parquet, predicate=DynamicFilter [ MinTemp@0 IS NULL OR MinTemp@0 > ], dynamic_rg_pruning=eligible, pruning_predicate=MinTemp_null_count@0 > 0 OR MinTemp_null_count@0 != row_count@2 AND MinTemp_max@1 > , required_guarantees=[] + │ t1: DataSourceExec: file_groups={3 groups: [[/testdata/weather/result-000000.parquet:..], [/testdata/weather/result-000001.parquet:.., /testdata/weather/result-000002.parquet:..], [/testdata/weather/result-000002.parquet:..]]}, projection=[MinTemp], file_type=parquet, predicate=DynamicFilter [ MinTemp@0 IS NULL OR MinTemp@0 > ], dynamic_rg_pruning=eligible, pruning_predicate=MinTemp_null_count@0 > 0 OR MinTemp_null_count@0 != row_count@2 AND MinTemp_max@1 > , required_guarantees=[] + └────────────────────────────────────────────────── + ")); + Ok(()) + } + + /// A TopK sort does not yet update dynamic filters to remote consumers. + #[tokio::test] + async fn remote_dynamic_filters() -> Result<()> { + let display = TestQuery::new( + r#" + SELECT key + FROM ( + SELECT DISTINCT "MinTemp" AS key + FROM weather + ) + ORDER BY key DESC + LIMIT 10 + "#, + ) + .with_expected_rows(10) + .execute() + .await?; + assert_snapshot!(display, @" + ┌───── DistributedExec + │ SortPreservingMergeExec: [key@0 DESC], fetch=10 + │ [Stage 2] => NetworkCoalesceExec: output_partitions=6, input_tasks=2 + └────────────────────────────────────────────────── + ┌───── Stage 2 ── tasks=2, partitions=3 + │ SortExec: TopK(fetch=10), expr=[key@0 DESC], preserve_partitioning=[true] + │ AggregateExec: mode=FinalPartitioned, gby=[key@0 as key], aggr=[], lim=[10] + │ [Stage 1] => NetworkShuffleExec: output_partitions=3, input_tasks=2 + └────────────────────────────────────────────────── + ┌───── Stage 1 ── tasks=2, partitions=6 + │ RepartitionExec: partitioning=Hash([key@0], 6), input_partitions=3 + │ AggregateExec: mode=Partial, gby=[key@0 as key], aggr=[], lim=[10] + │ DistributedLeafExec: + │ t0: DataSourceExec: file_groups={3 groups: [[/testdata/weather/result-000000.parquet:..], [/testdata/weather/result-000000.parquet:.., /testdata/weather/result-000001.parquet:..], [/testdata/weather/result-000002.parquet:..]]}, projection=[MinTemp@0 as key], file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible + │ t1: DataSourceExec: file_groups={3 groups: [[/testdata/weather/result-000000.parquet:..], [/testdata/weather/result-000001.parquet:.., /testdata/weather/result-000002.parquet:..], [/testdata/weather/result-000002.parquet:..]]}, projection=[MinTemp@0 as key], file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible + └────────────────────────────────────────────────── + "); + Ok(()) + } +} diff --git a/tests/multi_task_collect_join_repros.rs b/tests/multi_task_collect_join_repros.rs index fb15fbb93..7518c4d30 100644 --- a/tests/multi_task_collect_join_repros.rs +++ b/tests/multi_task_collect_join_repros.rs @@ -295,18 +295,18 @@ mod tests { ┌───── Stage 1 ── tasks=4, partitions=8 │ SortExec: TopK(fetch=5), expr=[id@0 ASC NULLS LAST], preserve_partitioning=[true] │ DistributedLeafExec: - │ t0: DataSourceExec: file_groups={2 groups: [[/target/multi_task_collect_join_repros/build_side/part-0.parquet:..], [/target/multi_task_collect_join_repros/build_side/part-2.parquet:..]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ empty ], sort_order_for_reorder=[id@0 ASC NULLS LAST], dynamic_rg_pruning=eligible - │ t1: DataSourceExec: file_groups={2 groups: [[/target/multi_task_collect_join_repros/build_side/part-0.parquet:.., /target/multi_task_collect_join_repros/build_side/part-1.parquet:..], [/target/multi_task_collect_join_repros/build_side/part-2.parquet:.., /target/multi_task_collect_join_repros/build_side/part-3.parquet:..]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ empty ], sort_order_for_reorder=[id@0 ASC NULLS LAST], dynamic_rg_pruning=eligible - │ t2: DataSourceExec: file_groups={2 groups: [[/target/multi_task_collect_join_repros/build_side/part-1.parquet:..], [/target/multi_task_collect_join_repros/build_side/part-3.parquet:..]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ empty ], sort_order_for_reorder=[id@0 ASC NULLS LAST], dynamic_rg_pruning=eligible - │ t3: DataSourceExec: file_groups={2 groups: [[/target/multi_task_collect_join_repros/build_side/part-1.parquet:.., /target/multi_task_collect_join_repros/build_side/part-2.parquet:..], [/target/multi_task_collect_join_repros/build_side/part-3.parquet:..]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ empty ], sort_order_for_reorder=[id@0 ASC NULLS LAST], dynamic_rg_pruning=eligible + │ t0: DataSourceExec: file_groups={2 groups: [[/target/multi_task_collect_join_repros/build_side/part-0.parquet:..], [/target/multi_task_collect_join_repros/build_side/part-2.parquet:..]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible + │ t1: DataSourceExec: file_groups={2 groups: [[/target/multi_task_collect_join_repros/build_side/part-0.parquet:.., /target/multi_task_collect_join_repros/build_side/part-1.parquet:..], [/target/multi_task_collect_join_repros/build_side/part-2.parquet:.., /target/multi_task_collect_join_repros/build_side/part-3.parquet:..]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible + │ t2: DataSourceExec: file_groups={2 groups: [[/target/multi_task_collect_join_repros/build_side/part-1.parquet:..], [/target/multi_task_collect_join_repros/build_side/part-3.parquet:..]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible + │ t3: DataSourceExec: file_groups={2 groups: [[/target/multi_task_collect_join_repros/build_side/part-1.parquet:.., /target/multi_task_collect_join_repros/build_side/part-2.parquet:..], [/target/multi_task_collect_join_repros/build_side/part-3.parquet:..]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible └────────────────────────────────────────────────── ┌───── Stage 2 ── tasks=4, partitions=8 │ SortExec: TopK(fetch=1000000), expr=[id@0 ASC NULLS LAST], preserve_partitioning=[true] │ DistributedLeafExec: - │ t0: DataSourceExec: file_groups={2 groups: [[/target/multi_task_collect_join_repros/probe_side/part-0.parquet:..], [/target/multi_task_collect_join_repros/probe_side/part-3.parquet:..]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ empty ], sort_order_for_reorder=[id@0 ASC NULLS LAST], dynamic_rg_pruning=eligible - │ t1: DataSourceExec: file_groups={2 groups: [[/target/multi_task_collect_join_repros/probe_side/part-0.parquet:..], [/target/multi_task_collect_join_repros/probe_side/part-3.parquet:..]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ empty ], sort_order_for_reorder=[id@0 ASC NULLS LAST], dynamic_rg_pruning=eligible - │ t2: DataSourceExec: file_groups={2 groups: [[/target/multi_task_collect_join_repros/probe_side/part-1.parquet:..], [/target/multi_task_collect_join_repros/probe_side/part-2.parquet:..]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ empty ], sort_order_for_reorder=[id@0 ASC NULLS LAST], dynamic_rg_pruning=eligible - │ t3: DataSourceExec: file_groups={2 groups: [[/target/multi_task_collect_join_repros/probe_side/part-1.parquet:..], [/target/multi_task_collect_join_repros/probe_side/part-2.parquet:..]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ empty ], sort_order_for_reorder=[id@0 ASC NULLS LAST], dynamic_rg_pruning=eligible + │ t0: DataSourceExec: file_groups={2 groups: [[/target/multi_task_collect_join_repros/probe_side/part-0.parquet:..], [/target/multi_task_collect_join_repros/probe_side/part-3.parquet:..]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible + │ t1: DataSourceExec: file_groups={2 groups: [[/target/multi_task_collect_join_repros/probe_side/part-0.parquet:..], [/target/multi_task_collect_join_repros/probe_side/part-3.parquet:..]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible + │ t2: DataSourceExec: file_groups={2 groups: [[/target/multi_task_collect_join_repros/probe_side/part-1.parquet:..], [/target/multi_task_collect_join_repros/probe_side/part-2.parquet:..]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible + │ t3: DataSourceExec: file_groups={2 groups: [[/target/multi_task_collect_join_repros/probe_side/part-1.parquet:..], [/target/multi_task_collect_join_repros/probe_side/part-2.parquet:..]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible └────────────────────────────────────────────────── ") } @@ -349,10 +349,10 @@ mod tests { ┌───── Stage 2 ── tasks=4, partitions=8 │ SortExec: TopK(fetch=1000000), expr=[id@0 ASC NULLS LAST], preserve_partitioning=[true] │ DistributedLeafExec: - │ t0: DataSourceExec: file_groups={2 groups: [[/target/multi_task_collect_join_repros/probe_side/part-0.parquet:..], [/target/multi_task_collect_join_repros/probe_side/part-3.parquet:..]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ empty ], sort_order_for_reorder=[id@0 ASC NULLS LAST], dynamic_rg_pruning=eligible - │ t1: DataSourceExec: file_groups={2 groups: [[/target/multi_task_collect_join_repros/probe_side/part-0.parquet:..], [/target/multi_task_collect_join_repros/probe_side/part-3.parquet:..]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ empty ], sort_order_for_reorder=[id@0 ASC NULLS LAST], dynamic_rg_pruning=eligible - │ t2: DataSourceExec: file_groups={2 groups: [[/target/multi_task_collect_join_repros/probe_side/part-1.parquet:..], [/target/multi_task_collect_join_repros/probe_side/part-2.parquet:..]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ empty ], sort_order_for_reorder=[id@0 ASC NULLS LAST], dynamic_rg_pruning=eligible - │ t3: DataSourceExec: file_groups={2 groups: [[/target/multi_task_collect_join_repros/probe_side/part-1.parquet:..], [/target/multi_task_collect_join_repros/probe_side/part-2.parquet:..]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ empty ], sort_order_for_reorder=[id@0 ASC NULLS LAST], dynamic_rg_pruning=eligible + │ t0: DataSourceExec: file_groups={2 groups: [[/target/multi_task_collect_join_repros/probe_side/part-0.parquet:..], [/target/multi_task_collect_join_repros/probe_side/part-3.parquet:..]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible + │ t1: DataSourceExec: file_groups={2 groups: [[/target/multi_task_collect_join_repros/probe_side/part-0.parquet:..], [/target/multi_task_collect_join_repros/probe_side/part-3.parquet:..]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible + │ t2: DataSourceExec: file_groups={2 groups: [[/target/multi_task_collect_join_repros/probe_side/part-1.parquet:..], [/target/multi_task_collect_join_repros/probe_side/part-2.parquet:..]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible + │ t3: DataSourceExec: file_groups={2 groups: [[/target/multi_task_collect_join_repros/probe_side/part-1.parquet:..], [/target/multi_task_collect_join_repros/probe_side/part-2.parquet:..]]}, projection=[id], file_type=parquet, predicate=DynamicFilter [ empty ], dynamic_rg_pruning=eligible └────────────────────────────────────────────────── ") }