From 26ab9ef9082892530be9272d64c998a17b757fdb Mon Sep 17 00:00:00 2001 From: Tam Nguyen Duc <1218621+tamnd@users.noreply.github.com> Date: Tue, 25 Aug 2026 16:48:59 +0700 Subject: [PATCH] Cut the scan finer now the pool keeps its workers --- bench/budgets.toml | 16 ++++++++++++++++ crates/zu-exec/src/run.rs | 28 +++++++++++++++++++++------- 2 files changed, 37 insertions(+), 7 deletions(-) diff --git a/bench/budgets.toml b/bench/budgets.toml index 27df5381..8e32e9dc 100644 --- a/bench/budgets.toml +++ b/bench/budgets.toml @@ -705,6 +705,22 @@ vec_unpack_gvals_s = 0.9 # worker times moved 2 percent the wrong way on both, which is a # single worker run not going near the pool at all. The same box # measured 3.6 to 3.7x when the note above was written. +# +# With the pool no longer losing a worker it was worth asking again how +# finely the scan should be cut, since the note in make_morsels saying +# sixteen morsels a worker gave nothing over eight was written while a +# query asking for eight hands was running on seven. Thirty two a +# worker, over eight paired alternating runs on the same box: the +# expand query 1.03 ms to 0.93 at eight workers in seven of the eight +# pairs, 5.4x to 5.9x, and the scan query 0.62 ms to 0.60. The single +# worker times moved with them, the scan by five percent in eight of +# eight, and one worker has no tail and nothing to contend over, so a +# shorter morsel is buying cache as well as balance: it is a pass over +# rows that are still warm when the next operator above the scan asks +# for them. The groupby bench agreed at eight workers over three runs +# apiece, 28.3 ms to 26.6 on the hundred thousand group shape, 8.4 to +# 7.2 on the twelve group one and 32.5 to 27.9 on the string key, all +# of them unmoved at one worker. exec_scale_8x = 2.0 # The grouping sink at one worker, M rows/s (perf/05 section 4): ten diff --git a/crates/zu-exec/src/run.rs b/crates/zu-exec/src/run.rs index 5519544a..b7e9fa16 100644 --- a/crates/zu-exec/src/run.rs +++ b/crates/zu-exec/src/run.rs @@ -949,19 +949,33 @@ fn intersects(plan: &ExecPlan) -> bool { plan.ops.iter().any(|op| matches!(op, Op::Intersect { .. })) } -/// Splits the scan into morsels: chunk-multiple sizes targeting eight -/// morsels per worker, never crossing a storage group boundary so a +/// Splits the scan into morsels: chunk-multiple sizes targeting thirty +/// two morsels per worker, never crossing a storage group boundary so a /// morsel's CSR pins and zone reads stay within one group. /// -/// Eight rather than four because the tail decides the query: on a +/// Many rather than few because the tail decides the query: on a /// machine with slow and fast cores the slow worker picks up a last /// morsel nobody else can finish for it, so the whole query waits on /// one morsel's worth of rows. Halving the morsel halves that tail. -/// Sixteen per worker gave nothing back over eight, so the claim -/// traffic starts costing what the shorter tail saves somewhere in -/// between. +/// This used to be eight, on a measurement saying sixteen gave nothing +/// back over it, but that was taken while the pool was still losing a +/// worker to a lost wakeup, and a query running on seven hands when it +/// asked for eight is not measuring how well eight of them balance. +/// With the pool fixed, thirty two beat eight over eight paired runs on +/// a quiet 32 core box: the expand query went 1.03 ms to 0.93 ms at +/// eight workers in seven of the eight pairs, and 5.4x scaling to 5.9x. +/// +/// The single worker figures moved with them, the scan by five percent +/// over eight pairs of eight, and one worker has no tail and no claim to +/// contend over. So this is not only balance. A morsel is a pass over +/// its rows by every operator above the scan in turn, and a shorter one +/// is a pass whose rows are still in cache when the next operator asks +/// for them. Thirty two of them at ten million rows and eight workers is +/// about forty thousand rows a morsel, which is where that starts to +/// tell. The claim traffic that pays for it is one `fetch_add` a morsel +/// against work that is tens of microseconds. fn make_morsels(rows: u64, workers: usize) -> Vec<(u64, u64)> { - split_morsels(rows, rows / (workers as u64 * 8)) + split_morsels(rows, rows / (workers as u64 * 32)) } /// The same tiling with the morsel size asked for rather than derived