diff --git a/.config/nextest.toml b/.config/nextest.toml index e0c3ed7b..c122f222 100644 --- a/.config/nextest.toml +++ b/.config/nextest.toml @@ -19,7 +19,7 @@ threads-required = 'num-cpus' # Serialize wall-clock perf assertions β€” CPU contention causes false failures, not the # assertions. Don't #[ignore] them, isolate instead. [[profile.default.overrides]] -filter = "test(storage_buffered_raft_log::performance_test)" +filter = "test(raft_log_core::performance_test)" threads-required = 'num-cpus' # Per-directory, not per-file β€” a few dirs also hold light tests, throttled as fallout. @@ -47,7 +47,7 @@ filter = "test(tmp_db) | test(config_path_env) | test(data_dir_lock) | test(sing threads-required = 'num-cpus' [[profile.ci.overrides]] -filter = "test(storage_buffered_raft_log::performance_test)" +filter = "test(raft_log_core::performance_test)" threads-required = 'num-cpus' [[profile.ci.overrides]] diff --git a/.dockerignore b/.dockerignore index 2a8dfc95..f20b5623 100644 --- a/.dockerignore +++ b/.dockerignore @@ -1,3 +1,5 @@ **/target/ .git/ -examples/ +examples/* +!examples/three-nodes-standalone +!examples/client-usage-standalone diff --git a/.github/PULL_REQUEST_TEMPLATE.md b/.github/PULL_REQUEST_TEMPLATE.md index 4e9f8766..aaa0692c 100644 --- a/.github/PULL_REQUEST_TEMPLATE.md +++ b/.github/PULL_REQUEST_TEMPLATE.md @@ -61,6 +61,12 @@ --- +## AI Assistance + +- [ ] This PR was written in part with the assistance of generative AI. All ideas and architecture decisions are mine; I have fully reviewed all changes. + +--- + ## Reviewer Notes (Optional: anything reviewers should focus on) diff --git a/CHANGELOG.md b/CHANGELOG.md index d9012493..3299062d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -32,6 +32,18 @@ All notable changes to this project will be documented in this file. returns immediately, and entries arriving during an in-flight fsync are coalesced into the same physical disk flush. Storage-level group commit is restored without artificial batching windows. +- **πŸ›‘ Client-acknowledged writes could be lost on correlated power loss (#446)**: Raft commit quorum + counted the leader's own log contribution using its in-memory tail (`last_entry_id()`), not its + fsync-confirmed position (`durable_index()`) β€” a write could reach a majority-looking commit index, + and be acknowledged to the client, before enough replicas had actually synced it to disk. If those + nodes then lost power before their next fsync, the acknowledged write was gone. Fixed: leader quorum + calculation, follower `AppendEntries` ACK timing (a follower now withholds its response until its own + `durable_index` reaches the acknowledged entry), and single-voter clusters (previously exempted from + this class of fix, see #329) all gate on `durable_index`. RPO=0 for acknowledged writes is now a + mandatory invariant. Net effect: write acknowledgment latency now includes fsync time on a quorum of + replicas β€” see [Throughput Optimization Guide](./d-engine/src/docs/performance/throughput-optimization-guide.md) + for tuning `idle_flush_interval_ms`. + ### Changed - **MSRV raised to Rust 1.89**: The `data_dir` startup lock (prevents two node processes from @@ -65,6 +77,17 @@ All notable changes to this project will be documented in this file. - **`NodeBuilder` is no longer public** β€” use `EmbeddedEngine::start_custom`/`StandaloneEngine::run_custom` to plug in a custom storage engine or state machine. See [Migration Guide](./MIGRATION_GUIDE.md) for details. +- **⚠️ `[raft] ordered_channel_capacity` renamed to `max_pending_append_responses`** (#446): Follows the + gRPC `AppendEntries` forwarder rewrite (`FuturesUnordered`-based, no longer strict-FIFO) that shipped + alongside the durability fix above. Old field name is silently ignored, not an error β€” update existing + configs to the new name to keep the setting in effect. + +- **⚠️ `[raft.persistence] strategy` removed** (#446): `PersistenceStrategy` was a single-variant enum + (`MemFirst`) left over from #268; its only meaning now lives in whether an entry has reached + `durable_index`, which is no longer a configurable choice. Existing configs setting `strategy = + "MemFirst"` or `"DiskFirst"` are silently ignored, not an error β€” remove the field, `flush_policy` + is the only persistence knob now. + --- ## [v0.2.4] - 2026-05-23 diff --git a/Cargo.lock b/Cargo.lock index a411d838..5c74fa52 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1830,9 +1830,9 @@ dependencies = [ [[package]] name = "rustls" -version = "0.23.40" +version = "0.23.45" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ef86cd5876211988985292b91c96a8f2d298df24e75989a43a3c73f2d4d8168b" +checksum = "0d41d731c7d2f962d1ccc364cec258de3c0e93b38c2fb3ba97ac74513048d634" dependencies = [ "log", "once_cell", @@ -1854,9 +1854,9 @@ dependencies = [ [[package]] name = "rustls-webpki" -version = "0.103.13" +version = "0.103.15" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "61c429a8649f110dddef65e2a5ad240f747e85f7758a6bccc7e5777bd33f756e" +checksum = "f3c3cf1d8b1e7d4927e2d154c3fcb02979afb9939629c62cd9048d4f07b60ac2" dependencies = [ "ring", "rustls-pki-types", diff --git a/Cargo.toml b/Cargo.toml index 2f80f4ab..86914470 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -69,8 +69,7 @@ opt-level = 3 # All dependency optimization levels [profile.release] incremental = true debug = false -# opt-level = 3 -opt-level = 'z' # Optimized for minimum size +opt-level = 3 overflow-checks = true lto = true # Eliminate redundant code and reduce size codegen-units = 1 # Reduce parallel code generation units to improve optimization effect diff --git a/benches/embedded-bench/Cargo.toml b/benches/embedded-bench/Cargo.toml index 956de599..45c2b9e1 100644 --- a/benches/embedded-bench/Cargo.toml +++ b/benches/embedded-bench/Cargo.toml @@ -18,3 +18,4 @@ futures = "0.3" tracing-subscriber = { version = "0.3", features = ["env-filter"] } tracing = { version = "0.1" } metrics-exporter-prometheus = "0.17.2" +console-subscriber = "0.5" diff --git a/benches/embedded-bench/Makefile b/benches/embedded-bench/Makefile index 966619b3..281c685e 100644 --- a/benches/embedded-bench/Makefile +++ b/benches/embedded-bench/Makefile @@ -1,27 +1,25 @@ # Makefile for embedded-bench # Provides benchmark commands matching embedded-bench/reports/v0.2.0/report_v0.2.0_final.md -.PHONY: help build clean test-single-write test-high-conc-write test-linearizable-read test-lease-read test-eventual-read test-hot-key all-tests +.PHONY: help build clean clean-log-db test-single-write test-high-conc-write test-linearizable-read test-lease-read test-eventual-read test-hot-key all-tests \ + all-ram-tests ramdisk-create clean-ram-log-db ramdisk-release +# =============================== +# Global Variables +# =============================== BENCH_BIN := ./target/release/embedded-bench CONFIG_DIR := ./config +LOG_LEVEL ?= warn # Node selection (default: n1) NODE ?= n1 CONFIG_PATH := $(CONFIG_DIR)/$(NODE).toml SINGLE_NODE_CONFIG_PATH := $(CONFIG_DIR)/single_node.toml -DATA_DIR := ./data/$(NODE) -SINGLE_NODE_DATA_DIR := ./data/single-node # Metrics port per node β€” same scheme as examples/three-nodes-standalone # (8081/8082/8083), reused because the two setups never run concurrently. METRICS_PORT := $(if $(filter n2,$(NODE)),8082,$(if $(filter n3,$(NODE)),8083,8081)) -# =============================== -# Global Variables -# =============================== -LOG_LEVEL ?= warn - # Common parameters (matching Standalone tests) KEY_SIZE := 8 VALUE_SIZE := 256 @@ -35,6 +33,25 @@ CLIENTS ?= 1000 VERIFY ?= false VERIFY_FLAG := $(if $(filter true,$(VERIFY)),--verify-write,) +# RAM disk (macOS only, optional) β€” isolates this node's storage on its own +# independent RAM disk volume, to strip physical-disk latency out of a +# benchmark. Not a general fix for unrelated flakiness β€” see +# tickets/milestones/v0.2.5/446-perf-batching-measurement-2026-09-13.md. +# One-shot: `make all-ram-tests`. Manual/single-node: add RAMDISK=true to +# any test-* target (needs the other two nodes already running for quorum). +RAMDISK ?= false +RAMDISK_SIZE_MB ?= 1024 +RAMDISK_SECTORS := $(shell echo $$(( $(RAMDISK_SIZE_MB) * 2048 ))) +RAMDISK_VOLUMES := RAMDisk1 RAMDisk2 RAMDisk3 +RAMDISK_INDEX := $(if $(filter n2,$(NODE)),2,$(if $(filter n3,$(NODE)),3,1)) + +ifeq ($(RAMDISK),true) +DATA_DIR := /Volumes/RAMDisk$(RAMDISK_INDEX)/$(NODE) +SINGLE_NODE_DATA_DIR := /Volumes/RAMDisk1/single-node +else +DATA_DIR := ./data/$(NODE) +SINGLE_NODE_DATA_DIR := ./data/single-node +endif # On macOS with Homebrew: auto-detect compression lib paths to skip bundled C++ # compilation of RocksDB dependencies, which fails under macOS 26 + Xcode 26 @@ -58,7 +75,6 @@ ifneq ($(ZSTD_PREFIX),) BREW_ROCKSDB_ENV += ZSTD_LIB_DIR=$(ZSTD_PREFIX)/lib endif - help: @echo "Embedded-bench Makefile - Performance Testing" @echo "" @@ -83,6 +99,14 @@ help: @echo "Run All:" @echo " make all-tests Run all benchmark tests" @echo "" + @echo "RAM Disk Cluster (macOS only, optional β€” isolates disk latency):" + @echo " make all-ram-tests NODE=nX Create RAM disks, run full --batch suite on this node (like all-tests)" + @echo " Run in 3 terminals with NODE=n1/n2/n3 to form the cluster" + @echo " make ramdisk-create Create RAMDisk1/2/3 if not already mounted" + @echo " make clean-ram-log-db Wipe node data off the RAM disks (keeps them mounted)" + @echo " make ramdisk-release Unmount RAMDisk1/2/3, freeing the memory" + @echo " Add RAMDISK=true to any single test-* target (needs the other 2 nodes already running)" + @echo "" @echo "Examples:" @echo " make test-linearizable-read # Run on n1 (default)" @echo " make test-linearizable-read NODE=n2 # Run on n2" @@ -109,6 +133,7 @@ clean-log-db: rm -rf ./logs/* rm -rf ./data/* rm -rf ./snapshots/* + # ============================================ # Write Performance Tests # ============================================ @@ -262,4 +287,34 @@ all-tests: build put @echo "" @echo "Compare results with Standalone mode:" - @echo " Standalone report: ../../benches/standalone-bench/reports/v0.2.2/report_v0.2.2.md" + +# ============================================ +# RAM Disk Cluster (macOS only, optional) +# ============================================ +# Isolates each node's storage on its own independent RAM disk volume β€” use +# this to strip physical-disk latency out of a benchmark (e.g. to study +# fsync/scheduling behavior in isolation), not as a general fix for +# unrelated flakiness. See tickets/milestones/v0.2.5/446-perf-batching-measurement-2026-09-13.md. + +# Ensures the RAM disks exist, then runs the full --batch suite for this +# node on its own RAM disk volume β€” same scope/shape as `make all-tests +# NODE=nX`, just on RAM disk. This is single-node, like all-tests: run it in +# 3 separate terminals with NODE=n1/n2/n3 to form the cluster, e.g. +# make all-ram-tests NODE=n2 CLIENTS=100 +all-ram-tests: ramdisk-create + $(MAKE) all-tests RAMDISK=true NODE=$(NODE) CLIENTS=$(CLIENTS) + +# Idempotent β€” mounts any of RAMDisk1/2/3 that aren't already present. +ramdisk-create: + @[ "$$(uname)" = "Darwin" ] || { echo "RAM disk targets need macOS."; exit 1; } + @for v in $(RAMDISK_VOLUMES); do \ + [ -d "/Volumes/$$v" ] || diskutil erasevolume HFS+ $$v `hdiutil attach -nomount ram://$(RAMDISK_SECTORS)` >/dev/null; \ + done + +# Wipes node data off the RAM disks; keeps the volumes mounted. +clean-ram-log-db: + @for v in $(RAMDISK_VOLUMES); do rm -rf /Volumes/$$v/*; done + +# Unmounts the RAM disks entirely, releasing the memory back to the OS. +ramdisk-release: + @for v in $(RAMDISK_VOLUMES); do diskutil eject /Volumes/$$v 2>/dev/null || true; done diff --git a/benches/embedded-bench/config/n1.toml b/benches/embedded-bench/config/n1.toml index 52055b93..4877ce21 100644 --- a/benches/embedded-bench/config/n1.toml +++ b/benches/embedded-bench/config/n1.toml @@ -15,13 +15,6 @@ max_drain = 1024 # Maximum number of commands to accumulate in a single batch during drain operations max_batch_size = 200 -[raft.persistence] -strategy = "MemFirst" -flush_policy = { Batch = { idle_flush_interval_ms = 1000 } } -# Maximum number of log entries to buffer in memory -# when using async persistence strategies (MemFirst/Batched) -max_buffered_entries = 10000 - [raft.metrics] enable_backpressure = false enable_batch = false @@ -31,3 +24,8 @@ enable = false [storage] unified_db = false + +[raft.replication] +max_inflight_append_requests = 256 +append_entries_max_entries_per_replication = 256 +replication_send_queue_capacity = 1024 diff --git a/benches/embedded-bench/config/n2.toml b/benches/embedded-bench/config/n2.toml index 56c9cf13..6a7e0540 100644 --- a/benches/embedded-bench/config/n2.toml +++ b/benches/embedded-bench/config/n2.toml @@ -15,13 +15,6 @@ max_drain = 1024 # Maximum number of commands to accumulate in a single batch during drain operations max_batch_size = 200 -[raft.persistence] -strategy = "MemFirst" -flush_policy = { Batch = { idle_flush_interval_ms = 1000 } } -# Maximum number of log entries to buffer in memory -# when using async persistence strategies (MemFirst/Batched) -max_buffered_entries = 10000 - [raft.metrics] enable_backpressure = false enable_batch = false @@ -31,3 +24,8 @@ enable = false [storage] unified_db = false + +[raft.replication] +max_inflight_append_requests = 256 +append_entries_max_entries_per_replication = 256 +replication_send_queue_capacity = 1024 diff --git a/benches/embedded-bench/config/n3.toml b/benches/embedded-bench/config/n3.toml index 701a1551..5daeb435 100644 --- a/benches/embedded-bench/config/n3.toml +++ b/benches/embedded-bench/config/n3.toml @@ -15,13 +15,6 @@ max_drain = 1024 # Maximum number of commands to accumulate in a single batch during drain operations max_batch_size = 200 -[raft.persistence] -strategy = "MemFirst" -flush_policy = { Batch = { idle_flush_interval_ms = 1000 } } -# Maximum number of log entries to buffer in memory -# when using async persistence strategies (MemFirst/Batched) -max_buffered_entries = 10000 - [raft.metrics] enable_backpressure = false enable_batch = false @@ -31,3 +24,8 @@ enable = false [storage] unified_db = false + +[raft.replication] +max_inflight_append_requests = 256 +append_entries_max_entries_per_replication = 256 +replication_send_queue_capacity = 1024 diff --git a/benches/embedded-bench/src/main.rs b/benches/embedded-bench/src/main.rs index c0d699d3..d96ef876 100644 --- a/benches/embedded-bench/src/main.rs +++ b/benches/embedded-bench/src/main.rs @@ -171,15 +171,25 @@ fn generate_value(size: usize) -> Vec { (0..size).map(|_| rng.random()).collect() } -#[tokio::main] +#[tokio::main(flavor = "multi_thread", worker_threads = 2)] async fn main() { - // Initialize logging - tracing_subscriber::fmt() - .with_env_filter( - tracing_subscriber::EnvFilter::from_default_env() - .add_directive(tracing::Level::INFO.into()), - ) - .init(); + if std::env::var("TOKIO_CONSOLE").is_ok() { + let tokio_console_port: u16 = std::env::var("TOKIO_CONSOLE_PORT") + .map(|v| v.parse::().expect("TOKIO_CONSOLE_PORT must be a valid port")) + .unwrap_or(6669); + println!("Tokio Console port: {tokio_console_port}"); + console_subscriber::Builder::default() + .server_addr(([127, 0, 0, 1], tokio_console_port)) + .init(); + } else { + // Initialize logging + tracing_subscriber::fmt() + .with_env_filter( + tracing_subscriber::EnvFilter::from_default_env() + .add_directive(tracing::Level::INFO.into()), + ) + .init(); + } let config_path = std::env::var("CONFIG_PATH").ok(); let data_dir = std::env::var("DATA_DIR").ok(); @@ -232,6 +242,20 @@ async fn start_metrics_server(port: u16) { MS_BUCKETS, ) .expect("failed to configure _ms buckets") + .set_buckets_for_metric( + metrics_exporter_prometheus::Matcher::Full( + "core.raft.peer.in_flight_at_dispatch".into(), + ), + &[0.0, 1.0, 2.0, 4.0, 8.0, 16.0, 32.0, 64.0, 128.0, 256.0], + ) + .expect("buckets") + .set_buckets_for_metric( + metrics_exporter_prometheus::Matcher::Full( + "core.raft.replication.entries_per_request".into(), + ), + &[1.0, 2.0, 4.0, 8.0, 16.0, 32.0, 64.0, 128.0, 256.0], + ) + .expect("buckets") .install() .expect("failed to start Prometheus metrics exporter"); } @@ -256,6 +280,7 @@ async fn run_benchmark_task( let stats = Arc::new(BenchmarkStats::new()); let key_counter = Arc::new(AtomicU64::new(0)); + let failed_count = Arc::new(AtomicU64::new(0)); let start_time = Instant::now(); let mut handles = Vec::with_capacity(clients); @@ -264,6 +289,7 @@ async fn run_benchmark_task( let engine = engine.clone(); let stats = stats.clone(); let key_counter = key_counter.clone(); + let failed_count = failed_count.clone(); let command = command.clone(); let handle = tokio::spawn(async move { @@ -289,7 +315,15 @@ async fn run_benchmark_task( } } } - Err(_) => continue, + Err(e) => { + let n = failed_count.fetch_add(1, Ordering::Relaxed); + if n < 5 { + eprintln!("Put failed: {e:?}"); + } else if n == 5 { + eprintln!("Put failed: (further failures suppressed)"); + } + continue; + } } } Commands::Get { consistency } => { @@ -327,6 +361,10 @@ async fn run_benchmark_task( futures::future::join_all(handles).await; stats.summary(start_time.elapsed()); + let failed = failed_count.load(Ordering::Relaxed); + if failed > 0 { + println!("Failed requests: {failed}"); + } } /// Run all benchmark tests in batch mode @@ -500,6 +538,9 @@ async fn run_local_benchmark(cli: Cli) { println!("Leader elected: {}", leader_info.leader_id); println!("Node ID: {}", engine.node_id()); + // Let every peer's replication stream settle before load starts; see #450 startup race. + tokio::time::sleep(Duration::from_secs(5)).await; + // Signal file used to coordinate follower auto-shutdown in batch mode let done_signal_path = "/tmp/embedded-bench-done"; @@ -563,6 +604,7 @@ async fn run_local_benchmark(cli: Cli) { let stats = Arc::new(BenchmarkStats::new()); let key_counter = Arc::new(AtomicU64::new(0)); + let failed_count = Arc::new(AtomicU64::new(0)); let start_time = Instant::now(); let mut handles = Vec::with_capacity(cli.clients); @@ -571,6 +613,7 @@ async fn run_local_benchmark(cli: Cli) { let engine = engine.clone(); let stats = stats.clone(); let key_counter = key_counter.clone(); + let failed_count = failed_count.clone(); let cli = cli.clone(); let handle = tokio::spawn(async move { @@ -625,8 +668,13 @@ async fn run_local_benchmark(cli: Cli) { } } } - Err(_) => { - // Write failed - skip recording + Err(e) => { + let n = failed_count.fetch_add(1, Ordering::Relaxed); + if n < 5 { + eprintln!("Put failed: {e:?}"); + } else if n == 5 { + eprintln!("Put failed: (further failures suppressed)"); + } continue; } } @@ -676,6 +724,10 @@ async fn run_local_benchmark(cli: Cli) { futures::future::join_all(handles).await; stats.summary(start_time.elapsed()); + let failed = failed_count.load(Ordering::Relaxed); + if failed > 0 { + println!("Failed requests: {failed}"); + } println!("\nBenchmark completed. Press Ctrl+C to shutdown."); let _ = shutdown_rx.changed().await; diff --git a/benches/reports/v0.2.5/bench_report_v0.2.5.md b/benches/reports/v0.2.5/bench_report_v0.2.5.md index 8e2bcb69..3a7f0f7d 100644 --- a/benches/reports/v0.2.5/bench_report_v0.2.5.md +++ b/benches/reports/v0.2.5/bench_report_v0.2.5.md @@ -2,136 +2,59 @@ **Test Environments**: -- **Local**: Apple M2 Mac mini (8-core, 16GB RAM, 3-node cluster on localhost) - **AWS**: EC2 c5.2xlarge (8 vCPUs, 16GB RAM, 50GB SSD) Γ— 3 nodes - -**Test Dates**: - -- **Local v0.2.5 vs v0.2.4**: June 2, 2026 (6-round average, embedded) -- **Local 2026-07-12**: July 12, 2026 (embedded: 6-round average at CLIENTS=100; standalone: 4-round average at conns=200/clients=200; both after stopping the background Docker monitoring stack) -- **AWS v0.2.5**: June 2, 2026 (2 Γ— 5-round average, 10 rounds total; embedded and standalone) - -**Key/Value**: 8 bytes / 256 bytes +- **Key/Value**: 8 bytes / 256 bytes --- -## Local 3-Node Cluster: d-engine v0.2.5 vs d-engine v0.2.4 - -### Embedded Mode: v0.2.5 vs v0.2.4 - -_(v0.2.5: 6-round average; v0.2.4: 4-round average; v0.2.3: 4-round average (Lease/Eventual re-measured on same machine). 2026-07-12: 6-round average (CLIENTS=100, Docker monitoring stack stopped). See Benchmark Configuration for settings.)_ - -| **Scenario** | **Metric** | **v0.2.3** | **v0.2.4** | **v0.2.5** | **Ξ” (v0.2.4β†’v0.2.5)** | **0712** | **Ξ” (v0.2.5β†’0712)** | -| ------------------- | ----------- | ------------- | ------------- | ------------- | --------------------- | ------------- | ------------------- | -| Single Client Write | Throughput | 10,075 ops/s | 9,740 ops/s | 9,658 ops/s | -0.8% β†’ | 15,424 ops/s | **+59.7%** βœ… | -| | Avg Latency | 0.099 ms | 0.102 ms | 0.103 ms | stable | 0.065 ms | **-37.2%** βœ… | -| | p99 Latency | 0.139 ms | 0.177 ms | 0.203 ms | +14.7% β†’ | 0.163 ms | **-20.0%** βœ… | -| High Conc. Write | Throughput | 176,314 ops/s | 233,821 ops/s | 224,176 ops/s | -4.1% β†’ | 200,621 ops/s | **-10.5%** ⚠️ | -| | Avg Latency | 0.566 ms | 0.426 ms | 0.447 ms | +4.9% β†’ | 0.500 ms | **+11.9%** ⚠️ | -| | p99 Latency | 1.566 ms | 1.164 ms | 0.912 ms | **-21.6%** βœ… | 1.128 ms | **+23.7%** ⚠️ | -| Linearizable Read | Throughput | 508,264 ops/s | 630,789 ops/s | 586,767 ops/s | -7.0% β†’ | 561,344 ops/s | -4.3% β†’ | -| | Avg Latency | 0.197 ms | 0.157 ms | 0.170 ms | +8.3% β†’ | 0.176 ms | +3.6% β†’ | -| | p99 Latency | 0.710 ms | 0.367 ms | 0.483 ms | +31.6% ⚠️ | 0.373 ms | **-22.7%** βœ… | -| Lease Read | Throughput | 705,530 ops/s | 730,836 ops/s | 893,254 ops/s | **+22.2%** βœ… | 724,861 ops/s | **-18.9%** ⚠️ | -| | Avg Latency | 0.116 ms | 0.136 ms | 0.007 ms | **-94.9%** βœ… | 0.009 ms | **+28.3%** ⚠️ | -| | p99 Latency | 0.342 ms | 0.337 ms | 0.066 ms | **-80.4%** βœ… | 0.059 ms | **-10.4%** βœ… | -| Eventual Read | Throughput | 742,602 ops/s | 752,198 ops/s | 884,781 ops/s | **+17.6%** βœ… | 768,119 ops/s | **-13.2%** ⚠️ | -| | Avg Latency | 0.115 ms | 0.132 ms | 0.007 ms | **-94.7%** βœ… | 0.008 ms | **+20.1%** ⚠️ | -| | p99 Latency | 0.382 ms | 0.343 ms | 0.067 ms | **-80.5%** βœ… | 0.058 ms | **-13.9%** βœ… | -| Hot-Key (10 keys) | Throughput | 499,527 ops/s | 659,429 ops/s | 622,459 ops/s | -5.6% β†’ | 567,529 ops/s | -8.8% β†’ | -| | Avg Latency | 0.205 ms | 0.153 ms | 0.160 ms | stable | 0.174 ms | +8.9% β†’ | -| | p99 Latency | 0.638 ms | 0.343 ms | 0.425 ms | +23.9% ⚠️ | 0.352 ms | **-17.1%** βœ… | - -**Notes**: - -- **Lease/Eventual Read: headline win vs v0.2.4** β€” ReadActor fast path (#392) reduces avg latency from ~0.130 ms to ~0.007 ms (**-94%+**), throughput +22.2%/+17.6%. Direct SM reads eliminate all Raft/channel hops for Eventual/LeaseRead under 100 concurrent clients. -- v0.2.3 Lease/Eventual throughput (705K/742K) is the re-measured baseline from the same machine; original v0.2.3 values (852K/859K) were not reproducible due to different system state at the time of that test. -- Lease Read run-to-run variance: 819K–1,001K across 6 rounds (6-round avg 893K). Round 5 reached 1,001K; round 4 dipped to 819K (OS scheduler noise on a loaded Mac mini). -- HC Write p99 **-21.6%** vs v0.2.4 β€” tail latency improvement despite slight avg regression; 6-round variance includes runs with max p99.9 up to 2,943 Β΅s skewing avg higher. -- Linearizable Read p99 +31.6% and HC Write avg +4.9% are within local benchmark noise on a shared Mac mini; neither read-index nor write-path changed from v0.2.4. -- SC Write and Hot-Key throughput within Β±6% of v0.2.4; consistent with normal run-to-run variance across 6 rounds. -- Default config (`read_actor_channel_capacity = 512`) yields Lease/Eventual ~790K/~806K (+8%/+7% vs v0.2.4). Results above use tuned settings (see configuration). - ---- +![d-engine v0.2.5 vs v0.2.4 Embedded Mode](d-engine_v0.2.5_vs_v0.2.4_embedded_mode.png) -### Standalone Mode: v0.2.5 vs v0.2.4 vs v0.2.3 - -_(v0.2.5: 5-round average; v0.2.4: 5-round average; v0.2.3: 5-round average. All manually collected. 2026-07-12: 4-round average (conns=200, clients=200, Docker monitoring stack stopped).)_ - -| **Scenario** | **Metric** | **v0.2.3** | **v0.2.4** | **v0.2.5** | **Ξ” (v0.2.4β†’v0.2.5)** | **0712** | **Ξ” (v0.2.5β†’0712)** | -| ------------------- | ----------- | ------------ | ------------ | ------------ | --------------------- | ------------ | ------------------- | -| Single Client Write | Throughput | 6,421 ops/s | 5,245 ops/s | 5,234 ops/s | stable | 9,450 ops/s | **+80.5%** βœ… | -| | Avg Latency | 0.155 ms | 0.190 ms | 0.190 ms | stable | 0.105 ms | **-44.6%** βœ… | -| | p99 Latency | 0.200 ms | 0.235 ms | 0.237 ms | stable | 0.223 ms | -6.0% β†’ | -| High Conc. Write | Throughput | 55,285 ops/s | 59,733 ops/s | 60,608 ops/s | +1.5% β†’ | 61,025 ops/s | +0.7% β†’ | -| | Avg Latency | 3.610 ms | 3.346 ms | 3.297 ms | -1.5% β†’ | 3.275 ms | -0.7% β†’ | -| | p99 Latency | 6.720 ms | 6.325 ms | 6.372 ms | stable | 6.063 ms | -4.8% β†’ | -| Linearizable Read | Throughput | 63,210 ops/s | 71,702 ops/s | 71,918 ops/s | stable | 83,771 ops/s | **+16.5%** βœ… | -| | Avg Latency | 3.160 ms | 2.791 ms | 2.781 ms | stable | 2.387 ms | **-14.2%** βœ… | -| | p99 Latency | 5.810 ms | 5.933 ms | 6.078 ms | stable | 5.074 ms | **-16.5%** βœ… | -| Lease Read | Throughput | 67,878 ops/s | 72,593 ops/s | 80,449 ops/s | **+10.8%** βœ… | 94,744 ops/s | **+17.8%** βœ… | -| | Avg Latency | 2.950 ms | 2.756 ms | 2.493 ms | **-9.5%** βœ… | 2.114 ms | **-15.2%** βœ… | -| | p99 Latency | 6.200 ms | 5.762 ms | 5.848 ms | stable | 4.992 ms | **-14.6%** βœ… | -| Eventual Read | Throughput | 91,174 ops/s | 94,956 ops/s | 97,188 ops/s | +2.4% β†’ | 92,253 ops/s | -5.1% β†’ | -| | Avg Latency | 2.190 ms | 2.103 ms | 2.054 ms | -2.3% β†’ | 2.163 ms | +5.3% β†’ | -| | p99 Latency | 13.970 ms | 9.762 ms | 10.084 ms | stable | 8.350 ms | **-17.2%** βœ… | -| Hot-Key (10 keys) | Throughput | 74,017 ops/s | 84,863 ops/s | 84,911 ops/s | stable | 103,131 ops/s | **+21.5%** βœ… | -| | Avg Latency | 2.700 ms | 2.360 ms | 2.356 ms | stable | 1.938 ms | **-17.7%** βœ… | -| | p99 Latency | 5.490 ms | 5.602 ms | 5.492 ms | -2.0% β†’ | 4.374 ms | **-20.4%** βœ… | - -**Notes**: - -- **v0.2.5 Standalone is stable vs v0.2.4** β€” all write and non-Lease read scenarios within Β±2%, confirming ReadActor fast path (#392) is additive and introduces no regressions in Standalone mode. -- **Lease Read: +10.8% throughput / -9.5% avg latency vs v0.2.4** β€” ReadActor shared infrastructure benefits Standalone mode; gRPC RTT masks the direct SM path, so improvement is smaller than Embedded. -- SC Write shows -18.3% vs v0.2.3 but is stable vs v0.2.4; the latency floor shift (155 Β΅s β†’ 190 Β΅s) was introduced before v0.2.4, not a #392 regression. -- Eventual Read p99 has high run-to-run variance (~Β±1 ms) due to Mac mini OS scheduler noise under 1000 concurrent clients; the improvement trend from v0.2.3 (13.97 ms) to v0.2.4 (9.76 ms) is the meaningful signal. -- Lease Read round-to-run variance: ~75K–91K across 5 rounds (round 4 reached 91K; other rounds avg ~78K). 5-round avg 80K. -- HC Write within Β±5% of v0.2.5; macOS fdatasync ~9ms is the ceiling for Level 3 (MemFirst) at this concurrency level. +![d-engine v0.2.5 vs v0.2.4 Standalone Mode](d-engine_v0.2.5_vs_v0.2.4_standalone_mode.png) ---- +![d-engine v0.2.5 vs etcd 3.2.0](d-engine_comparison_v0.2.5.png) ## AWS 3-Node Cluster: d-engine v0.2.5 vs v0.2.4 / etcd 3.2.0 -**Hardware**: AWS EC2 c5.2xlarge (8 vCPUs, 16GB RAM, 50GB SSD) Γ— 3 nodes -**Date**: June 2, 2026 | 2 Γ— 5-round average (10 rounds total) | Key/Value: 8 bytes / 256 bytes +**Hardware**: AWS EC2 c5.2xlarge (8 vCPUs, 16GB RAM, 50GB SSD) Γ— 3 nodes +**Date**: September 26, 2026 | 3-round average | Key/Value: 8 bytes / 256 bytes +**Embedded mode concurrency**: `--clients 1000` (previously omitted from the AWS harness β€” see Benchmark Configuration note below) **etcd reference**: Official etcd benchmark (GCE, 8 vCPUs + 16GB + SSD Γ— 3 nodes, etcd 3.2.0)Β² -![d-engine v0.2.5 vs v0.2.4 Embedded Mode](d-engine_v0.2.5_vs_v0.2.4_embedded_mode.png) - -![d-engine v0.2.5 vs v0.2.4 Standalone Mode](d-engine_v0.2.5_vs_v0.2.4_standalone_mode.png) - -![d-engine v0.2.5 vs etcd 3.2.0](d-engine_comparison_v0.2.5.png) +> Updated 2026-09-26: this section now reflects v0.2.5 **with #446 merged** +> (quorum commit gated on physical fsync β€” durable_index only advances after +> fdatasync, per ADR-038 Level 3 / RPO=0). The June 2, 2026 numbers previously +> here predate #446 and measured #392 (ReadActor) alone; #446 changes write-path +> latency substantially and is reflected below. ### Embedded Mode: v0.2.5 vs v0.2.4 / etcd 3.2.0 -| **Scenario** | **Metric** | **v0.2.4 (AWS)** | **v0.2.5 (AWS)** | **Ξ” vs v0.2.4** | **etcd 3.2.0Β²** | **Ξ” vs etcd** | -| ------------------- | ----------- | ---------------- | ---------------- | --------------- | --------------- | ------------- | -| Single Client Write | Throughput | 4,398 ops/s | 4,428 ops/s | +0.7% β†’ | 583 ops/s | **+7.6x** βœ… | -| | Avg Latency | 0.227 ms | 0.225 ms | stable | 1.6 ms | **-86%** βœ… | -| | p99 Latency | 0.316 ms | 0.303 ms | -4.1% β†’ | β€” | β€” | -| High Conc. Write | Throughput | 110,798 ops/s | 111,618 ops/s | +0.7% β†’ | 44,341 ops/s | **+152%** βœ… | -| | Avg Latency | 0.902 ms | 0.897 ms | stable | 22.0 ms | **-95.9%** βœ… | -| | p99 Latency | 1.063 ms | 1.060 ms | stable | β€” | β€” | -| Linearizable Read | Throughput | 327,355 ops/s | 345,872 ops/s | +5.7% β†’ | 141,578 ops/s | **+144%** βœ… | -| | Avg Latency | 0.305 ms | 0.289 ms | -5.2% β†’ | 5.5 ms | **-94.7%** βœ… | -| | p99 Latency | 0.374 ms | 0.348 ms | -7.0% β†’ | β€” | β€” | -| Lease Read | Throughput | 341,254 ops/s | 1,231,541 ops/s | **+260.9%** βœ… | β€”Β³ | β€” | -| | Avg Latency | 0.293 ms | 0.005 ms | **-98.3%** βœ… | β€” | β€” | -| | p99 Latency | 0.367 ms | 0.018 ms | **-95.1%** βœ… | β€” | β€” | -| Eventual Read | Throughput | 363,407 ops/s | 1,235,832 ops/s | **+240.1%** βœ… | 185,758 ops/s | **+6.7x** βœ… | -| | Avg Latency | 0.274 ms | 0.005 ms | **-98.2%** βœ… | 2.2 ms | **-99.8%** βœ… | -| | p99 Latency | 0.343 ms | 0.019 ms | **-94.5%** βœ… | β€” | β€” | -| Hot-Key (10 keys) | Throughput | 319,957 ops/s | 346,830 ops/s | **+8.4%** βœ… | β€”Β³ | β€” | -| | Avg Latency | 0.313 ms | 0.288 ms | -8.0% β†’ | β€” | β€” | -| | p99 Latency | 0.360 ms | 0.331 ms | -8.1% β†’ | β€” | β€” | +| **Scenario** | **Metric** | **v0.2.4 (AWS)** | **v0.2.5 (AWS)** | **Ξ” vs v0.2.4** | **etcd 3.2.0Β²** | **Ξ” vs etcd** | +| ------------------- | ----------- | ---------------- | ---------------- | --------------- | --------------- | -------------- | +| Single Client Write | Throughput | 4,398 ops/s | 334 ops/s | **-92.4%** ⚠️ | 583 ops/s | -42.8% ⚠️ | +| | Avg Latency | 0.227 ms | 2.994 ms | **+1219%** ⚠️ | 1.6 ms | +87.1% ⚠️ | +| | p99 Latency | 0.316 ms | 3.154 ms | **+898%** ⚠️ | β€” | β€” | +| High Conc. Write | Throughput | 110,798 ops/s | 115,648 ops/s | +4.4% β†’ | 44,341 ops/s | **+2.6x** βœ… | +| | Avg Latency | 0.902 ms | 8.586 ms | **+852%** ⚠️ | 22.0 ms | -61.0% βœ… | +| | p99 Latency | 1.063 ms | 12.012 ms | **+1030%** ⚠️ | β€” | β€” | +| Linearizable Read | Throughput | 327,355 ops/s | 404,684 ops/s | +23.6% βœ… | 141,578 ops/s | **+2.9x** βœ… | +| | Avg Latency | 0.305 ms | 2.454 ms | **+705%** ⚠️ | 5.5 ms | -55.4% βœ… | +| | p99 Latency | 0.374 ms | 4.378 ms | **+1071%** ⚠️ | β€” | β€” | +| Lease Read | Throughput | 341,254 ops/s | 1,186,318 ops/s | **+247.7%** βœ… | β€”Β³ | β€” | +| | Avg Latency | 0.293 ms | 0.001 ms | **-99.7%** βœ… | β€” | β€” | +| | p99 Latency | 0.367 ms | 0.002 ms | **-99.5%** βœ… | β€” | β€” | +| Eventual Read | Throughput | 363,407 ops/s | 1,220,515 ops/s | **+235.9%** βœ… | 185,758 ops/s | **+6.6x** βœ… | +| | Avg Latency | 0.274 ms | 0.001 ms | **-99.6%** βœ… | 2.2 ms | **-99.95%** βœ… | +| | p99 Latency | 0.343 ms | 0.002 ms | **-99.4%** βœ… | β€” | β€” | +| Hot-Key (10 keys) | Throughput | 319,957 ops/s | 410,173 ops/s | +28.2% βœ… | β€”Β³ | β€” | +| | Avg Latency | 0.313 ms | 2.421 ms | **+674%** ⚠️ | β€” | β€” | +| | p99 Latency | 0.360 ms | 4.858 ms | **+1249%** ⚠️ | β€” | β€” | **Key Findings**: -- **Lease/Eventual Read: ~3.6x throughput improvement vs v0.2.4** β€” ReadActor fast path (#392) delivers 261%/240% throughput gain and 95%+ latency reduction. On AWS, this is the headline win: zero-channel direct SM reads eliminate all Raft/channel contention. -- **Write performance stable vs v0.2.4** β€” HC Write +0.7%, SC Write +0.7%, within AWS noise tolerance. -- **Linearizable Read +5.7% vs v0.2.4** β€” observed benchmark delta; Linearizable Read path remains on the Raft loop and is unchanged by #392. -- **vs etcd 3.2.0** β€” Embedded mode outperforms etcd across all comparable metrics: SC Write **+7.6x**, HC Write **+152%**, Lin Read **+144%**, Eventual Read **+6.7x** throughput; avg latency **-86~99%** lower. -- **v0.2.5 vs v0.2.3 baseline** β€” v0.2.4 had Lease/Eventual regressions vs v0.2.3 (378K/395K). v0.2.5 with #392 ReadActor achieves **+3.3x** vs v0.2.3. +- **Single Client Write is the clearest #446 signal, and it's a regression**: throughput -92.4%, avg latency 0.227ms β†’ 2.994ms (13x). No concurrency to amortize the cost β€” every request pays the full quorum-fsync round trip directly. d-engine's SC Write is now _slower_ than etcd's reference number (-42.8% throughput), reversing the previous +7.6x lead. +- **Concurrent write/read throughput holds up (HC Write +4.4%, Lin Read +23.6%, Hot-Key +28.2%)** β€” with enough in-flight requests, individual fsync/replication latency overlaps instead of stacking, so aggregate throughput survives. But **every scenario's avg/p99 latency is 7-13x worse than v0.2.4**, including reads that don't touch fsync (Hot-Key avg +674%) β€” consistent with the write path now serializing behind real fsync+quorum-ack round trips that the old `MemFirst` config never paid. +- **Lease/Eventual Read remain #392's win, untouched by #446** (+248%/+236% vs v0.2.4, both are pure in-memory reads with no Raft RPO=0 gating). +- **This is a much larger cost than #446's own loopback A/B measurements** (`tickets/milestones/v0.2.5/RESULTS-main-vs-446.md`: +9.9% `propose_to_apply`, -2.1% throughput). AWS's EBS gp3 is network-attached storage β€” every fsync is a real network round trip, unlike loopback disk I/O. Confirms the loopback numbers understate #446's real-world cost by roughly an order of magnitude. --- @@ -139,74 +62,37 @@ _(v0.2.5: 5-round average; v0.2.4: 5-round average; v0.2.3: 5-round average. All | **Scenario** | **Metric** | **v0.2.4 (AWS)** | **v0.2.5 (AWS)** | **Ξ” vs v0.2.4** | **etcd 3.2.0Β²** | **Ξ” vs etcd** | | ------------------- | ----------- | ---------------- | ---------------- | --------------- | --------------- | ------------- | -| Single Client Write | Throughput | 2,305 ops/s | 2,095 ops/s | -9.1% ⚠️ | 583 ops/s | **+3.6x** βœ… | -| | Avg Latency | 0.433 ms | 0.480 ms | +10.9% ⚠️ | 1.6 ms | **-70%** βœ… | -| | p99 Latency | 0.316 ms | 0.600 ms | +89.9% ⚠️ | β€” | β€” | -| High Conc. Write | Throughput | 59,687 ops/s | 53,338 ops/s | -10.6% ⚠️ | 44,341 ops/s | +20% β†’ | -| | Avg Latency | 0.40 ms | 4.353 ms | +988% ⚠️ | 22.0 ms | **-80.2%** βœ… | -| | p99 Latency | 1.20 ms | 12.923 ms | +977% ⚠️ | β€” | β€” | -| Linearizable Read | Throughput | 77,907 ops/s | 80,145 ops/s | +2.9% β†’ | 141,578 ops/s | -43% ⚠️ | -| | Avg Latency | 2.79 ms | 2.495 ms | -10.6% β†’ | 5.5 ms | **-54.6%** βœ… | -| | p99 Latency | 5.93 ms | 5.913 ms | stable | β€” | β€” | -| Lease Read | Throughput | 76,144 ops/s | 86,466 ops/s | **+13.6%** βœ… | β€”Β³ | β€” | -| | Avg Latency | 2.76 ms | 2.311 ms | -16.3% βœ… | β€” | β€” | -| | p99 Latency | 5.76 ms | 5.957 ms | stable | β€” | β€” | -| Eventual Read | Throughput | 170,000 ops/s | 157,358 ops/s | -7.4% ⚠️ | 185,758 ops/s | -15% ⚠️ | -| | Avg Latency | 2.10 ms | 1.539 ms | -26.7% βœ… | 2.2 ms | -30% β†’ | -| | p99 Latency | 9.76 ms | 6.674 ms | **-31.6%** βœ… | β€” | β€” | -| Hot-Key (10 keys) | Throughput | 90,609 ops/s | 92,711 ops/s | +2.3% β†’ | β€”Β³ | β€” | -| | Avg Latency | 2.36 ms | 2.154 ms | -8.7% β†’ | β€” | β€” | -| | p99 Latency | 5.60 ms | 5.922 ms | +5.8% β†’ | β€” | β€” | +| Single Client Write | Throughput | 2,305 ops/s | 302 ops/s | **-86.9%** ⚠️ | 583 ops/s | -48.2% ⚠️ | +| | Avg Latency | 0.433 ms | 3.316 ms | **+666%** ⚠️ | 1.6 ms | +107.3% ⚠️ | +| | p99 Latency | 0.567 ms | 3.513 ms | **+520%** ⚠️ | β€” | β€” | +| High Conc. Write | Throughput | 59,687 ops/s | 70,395 ops/s | +17.9% βœ… | 44,341 ops/s | +58.8% βœ… | +| | Avg Latency | 3.346 ms | 14.137 ms | **+323%** ⚠️ | 22.0 ms | -35.7% βœ… | +| | p99 Latency | 7.106 ms | 20.943 ms | **+195%** ⚠️ | β€” | β€” | +| Linearizable Read | Throughput | 77,907 ops/s | 89,218 ops/s | +14.5% βœ… | 141,578 ops/s | -37.0% ⚠️ | +| | Avg Latency | 2.563 ms | 11.145 ms | **+335%** ⚠️ | 5.5 ms | +102.6% ⚠️ | +| | p99 Latency | 6.308 ms | 22.922 ms | **+263%** ⚠️ | β€” | β€” | +| Lease Read | Throughput | 76,144 ops/s | 99,699 ops/s | **+30.9%** βœ… | β€”Β³ | β€” | +| | Avg Latency | 2.624 ms | 9.972 ms | **+280%** ⚠️ | β€” | β€” | +| | p99 Latency | 6.130 ms | 22.058 ms | **+260%** ⚠️ | β€” | β€” | +| Eventual Read | Throughput | 170,000 ops/s | 243,775 ops/s | **+43.4%** βœ… | 185,758 ops/s | +31.2% βœ… | +| | Avg Latency | 1.171 ms | 4.044 ms | **+245%** ⚠️ | 2.2 ms | +83.8% ⚠️ | +| | p99 Latency | 2.245 ms | 9.087 ms | **+305%** ⚠️ | β€” | β€” | +| Hot-Key (10 keys) | Throughput | 90,609 ops/s | 104,701 ops/s | +15.6% βœ… | β€”Β³ | β€” | +| | Avg Latency | 2.203 ms | 9.493 ms | **+331%** ⚠️ | β€” | β€” | +| | p99 Latency | 6.127 ms | 19.060 ms | **+211%** ⚠️ | β€” | β€” | **Key Findings**: -- **Lease Read +13.6% vs v0.2.4** β€” ReadActor fast path benefit carries to Standalone mode (ReadActor is shared infrastructure). -- **Eventual Read: Mixed results vs v0.2.4** β€” Throughput -7.4%, but latency **-26.7%** avg / **-31.6%** p99, indicating improved tail behavior. -- **HC Write regression severe** β€” Latency spike 0.4ms β†’ 4.4ms; ConnectionTimeout errors observed in raw rounds, suggesting network saturation or TCP pressure on c5.2xlarge under high concurrency. -- **SC Write regression** β€” -9.1% throughput vs v0.2.4, correlated with HC Write contention overhead. -- **vs etcd 3.2.0** β€” SC Write **+3.6x** and all latency metrics significantly better; Lin Read throughput below etcd (-43%) but latency -54.6% lower; HC Write and Eventual Read throughput regressions (vs v0.2.4) pull both metrics near or below etcd level. +- **Same pattern as Embedded, more pronounced**: SC Write throughput -86.9%, avg latency 0.433ms β†’ 3.316ms. Standalone adds real gRPC network RTT on top of the same fsync cost, so it's the worst-case number. +- **Concurrent scenarios (HC Write, Lin Read, Lease/Eventual Read, Hot-Key) all gain throughput (+15-43%)** β€” likely #392 ReadActor + other carried-over improvements outweighing #446's per-op cost once enough concurrency hides it β€” **but every one of them has 2-4x worse avg/p99 latency**. Throughput alone is not the full picture for this release; latency is where #446's cost is visible everywhere. +- **vs etcd**: d-engine has flipped from beating etcd on Single Client Write (+3.6x in v0.2.4) to losing to it (-48.2%). Linearizable Read is now worse than etcd on both throughput _and_ latency (previously only throughput was behind). This scenario deserves explicit sign-off before shipping v0.2.5 β€” RPO=0 is a deliberate trade-off (see ADR-038), but the "vs etcd" story materially changes. -Β² etcd data sourced from [etcd official benchmark documentation](https://etcd.io/docs/v3.6/op-guide/performance/), tested on GCE infrastructure. Different cloud platform; results are for reference only. +Β² etcd data sourced from [etcd official benchmark documentation](https://etcd.io/docs/v3.6/op-guide/performance/), tested on GCE infrastructure. Different cloud platform; results are for reference only. Β³ etcd does not have an equivalent mode. --- -## Key Changes Driving Results - -| Change | Impact | -| --------------------------------------- | ----------------------------------------------------------------------------------------------------------------- | -| ReadActor fast path (#392) | Lease/Eventual Read avg latency -94%+ vs v0.2.4; throughput +22.2%/+17.6%; bypasses Raft loop entirely | -| ReadLease.revoke() (#392) | Atomic lease invalidation on leader demotion; replaces `invalidate()` | -| Configurable ReadActor (#392) | `read_actor_channel_capacity` and `read_actor_max_drain` in `[raft]` | -| Fix: RocksDB LOCK on stop() (#392) | ReadActor is sole `Arc` holder; LOCK released before Raft shutdown | -| FsyncCoordinator (#422 follow-up) | Single-flight fsync; eliminates `wal_write_mutex_` lock convoy; SC Write +74% (standalone), +68% (embedded) | -| disableWAL=true for SM (#422 follow-up) | Removes redundant SM WAL write; eliminates `WriteGroupToWAL` stack from apply path; avg latency -40%+ on SC Write | - ---- - -## ReadActor Parameter Tuning (Local, Embedded Mode) - -Four configurations tested to characterize `read_actor_channel_capacity` and `read_actor_max_drain` sensitivity: - -| Config | cap / drain / batch | Lease Read | Eventual Read | HC Write | vs v0.2.3 (705K/742K) | -| ------ | ---------------------- | ---------- | ------------- | --------- | --------------------- | -| C1 | 1024 / 1000 / 200 | ~790K | ~806K | ~232K | +12% / +9% | -| C2 | 1024 / 1000 / 300 | ~785K | ~803K | ~231K | +11% / +8% | -| C3 | 1024 / 2000 / 200 | ~756K | ~789K | ~227K | +7% / +6% | -| **C4** | **10240 / 2000 / 200** | **~810K** | **~851K** | **~232K** | **+15% / +15%** | - -**Key findings**: - -- `channel_capacity` is the dominant knob: 512β†’1024 yields +30%, 1024β†’10240 yields +3–6%. -- `max_drain` > `channel_capacity` has no effect: drain loop exits early once channel is empty. -- `max_batch_size = 300` does not improve reads and introduces write p99 instability. -- Rule of thumb: `read_actor_channel_capacity = 2Γ— peak_concurrent_readers`. - ---- - -## Benchmark Configuration - -All Local Embedded results above were collected with the following configuration: + ```toml [raft] @@ -214,9 +100,10 @@ read_actor_channel_capacity = 10240 read_actor_max_drain = 2000 [raft.persistence] -strategy = "MemFirst" flush_policy = { Batch = { idle_flush_interval_ms = 1000 } } [raft.batching] max_batch_size = 200 ``` + +Embedded-bench invocation: `--clients 1000` (AWS harness previously omitted this flag; Single Client Write always used `clients=1` regardless of this setting). diff --git a/benches/reports/v0.2.5/d-engine_comparison_v0.2.5.png b/benches/reports/v0.2.5/d-engine_comparison_v0.2.5.png index 48d6aed8..f6524404 100644 Binary files a/benches/reports/v0.2.5/d-engine_comparison_v0.2.5.png and b/benches/reports/v0.2.5/d-engine_comparison_v0.2.5.png differ diff --git a/benches/reports/v0.2.5/d-engine_v0.2.5_vs_v0.2.4_embedded_mode.png b/benches/reports/v0.2.5/d-engine_v0.2.5_vs_v0.2.4_embedded_mode.png index a1851eda..bee61812 100644 Binary files a/benches/reports/v0.2.5/d-engine_v0.2.5_vs_v0.2.4_embedded_mode.png and b/benches/reports/v0.2.5/d-engine_v0.2.5_vs_v0.2.4_embedded_mode.png differ diff --git a/benches/reports/v0.2.5/d-engine_v0.2.5_vs_v0.2.4_standalone_mode.png b/benches/reports/v0.2.5/d-engine_v0.2.5_vs_v0.2.4_standalone_mode.png index bd60e8a4..032e71b9 100644 Binary files a/benches/reports/v0.2.5/d-engine_v0.2.5_vs_v0.2.4_standalone_mode.png and b/benches/reports/v0.2.5/d-engine_v0.2.5_vs_v0.2.4_standalone_mode.png differ diff --git a/benches/standalone-bench/Cargo.toml b/benches/standalone-bench/Cargo.toml index 165eb05a..134a713a 100644 --- a/benches/standalone-bench/Cargo.toml +++ b/benches/standalone-bench/Cargo.toml @@ -12,7 +12,6 @@ tokio = { version = "1.44", features = ["full"] } clap = { version = "4.5", features = ["derive"] } rand = { version = "0.8", features = ["small_rng"] } hdrhistogram = "7.5" -atomic-shim = "0.2.0" bytes = "1.10" futures = "0.3" thiserror = "1.0" diff --git a/d-engine-client/src/pool_test.rs b/d-engine-client/src/pool_test.rs index 4ff692bf..0b3d59b6 100644 --- a/d-engine-client/src/pool_test.rs +++ b/d-engine-client/src/pool_test.rs @@ -118,9 +118,16 @@ async fn test_create_channel_success() { cluster_ready_timeout: Duration::from_secs(5), }; - // Test with an invalid address to verify timeout behavior + // Bind a port and immediately drop the listener β€” the port is now guaranteed + // closed, so the connection below fails deterministically, independent of the + // host's DNS resolver (some resolvers redirect nonexistent hostnames to a + // sinkhole IP instead of failing lookup, which broke this test on such hosts). + let closed_port = { + let listener = std::net::TcpListener::bind("127.0.0.1:0").unwrap(); + listener.local_addr().unwrap().port() + }; let result = - ConnectionPool::create_channel("http://invalid.address:50051".to_string(), &config).await; + ConnectionPool::create_channel(format!("http://127.0.0.1:{closed_port}"), &config).await; assert!(result.is_err()); } diff --git a/d-engine-core/Cargo.toml b/d-engine-core/Cargo.toml index d1533629..699209ec 100644 --- a/d-engine-core/Cargo.toml +++ b/d-engine-core/Cargo.toml @@ -61,7 +61,7 @@ crc32fast = "1.4.2" # used in stream http-body = "1.0" http-body-util = "0.1.3" -# Lock-free skip list used as the in-memory entry index in BufferedRaftLog +# Lock-free skip list used as the in-memory entry index in RaftLogCore crossbeam-skiplist = "0.1" # Compress/decompress data stream async-compression = { version = "0.4", features = ["tokio", "gzip"] } diff --git a/d-engine-core/src/config/raft.rs b/d-engine-core/src/config/raft.rs index 78e3a333..ef661d8f 100644 --- a/d-engine-core/src/config/raft.rs +++ b/d-engine-core/src/config/raft.rs @@ -79,11 +79,15 @@ pub struct RaftConfig { #[serde(default = "default_cmd_channel_capacity")] pub cmd_channel_capacity: usize, - /// Ordered channel capacity for stream_append_entries ordering - /// Controls buffering of response receivers in FIFO order - /// Default value is set via default_ordered_channel_capacity() function - #[serde(default = "default_ordered_channel_capacity")] - pub ordered_channel_capacity: usize, + /// Max in-flight AppendEntries requests on `stream_append_entries` that can be + /// dispatched to the Raft loop and awaiting their response at once. Once this many + /// are pending, the stream stops reading new requests until one completes β€” this + /// bounds memory/task growth if this node's own durable_index stalls (RPO=0, #446). + /// Also used directly as the output channel's buffer size, since completed + /// responses can never outnumber in-flight requests. + /// Default value is set via default_max_pending_append_responses() function + #[serde(default = "default_max_pending_append_responses")] + pub max_pending_append_responses: usize, /// ReadActor configuration β€” tuning for the dedicated Eventual/LeaseRead fast path. #[serde(default)] @@ -141,7 +145,7 @@ impl Default for RaftConfig { auto_join: AutoJoinConfig::default(), snapshot_rpc_timeout_ms: default_snapshot_rpc_timeout_ms(), cmd_channel_capacity: default_cmd_channel_capacity(), - ordered_channel_capacity: default_ordered_channel_capacity(), + max_pending_append_responses: default_max_pending_append_responses(), read_actor: ReadActorConfig::default(), read_consistency: ReadConsistencyConfig::default(), backpressure: BackpressureConfig::default(), @@ -165,6 +169,14 @@ impl RaftConfig { ))); } + if self.max_pending_append_responses == 0 { + return Err(Error::Config(ConfigError::Message( + "max_pending_append_responses must be at least 1 \ + (0 causes mpsc::channel to panic)" + .into(), + ))); + } + self.replication.validate()?; self.batching.validate()?; self.election.validate()?; @@ -201,7 +213,7 @@ fn default_cmd_channel_capacity() -> usize { 1024 } -fn default_ordered_channel_capacity() -> usize { +fn default_max_pending_append_responses() -> usize { 1024 } @@ -275,6 +287,15 @@ pub struct ReplicationConfig { /// Maximum log entries per single AppendEntries RPC to a follower. #[serde(default = "default_entries_per_replication")] pub append_entries_max_entries_per_replication: u64, + + /// Max un-acknowledged AppendEntries requests allowed in flight + #[serde(default = "default_max_inflight_append_requests")] + pub max_inflight_append_requests: usize, + + /// Per-peer send queue capacity (requests). Must be >= max_inflight_append_requests, + /// otherwise the queue reports Full before the in-flight window does. + #[serde(default = "default_replication_send_queue_capacity")] + pub replication_send_queue_capacity: usize, } impl Default for ReplicationConfig { @@ -282,6 +303,8 @@ impl Default for ReplicationConfig { Self { rpc_append_entries_clock_in_ms: default_append_interval(), append_entries_max_entries_per_replication: default_entries_per_replication(), + max_inflight_append_requests: default_max_inflight_append_requests(), + replication_send_queue_capacity: default_replication_send_queue_capacity(), } } } @@ -299,6 +322,18 @@ impl ReplicationConfig { ))); } + if self.max_inflight_append_requests == 0 { + return Err(Error::Config(ConfigError::Message( + "max_inflight_append_requests must be > 0".into(), + ))); + } + + if self.replication_send_queue_capacity < self.max_inflight_append_requests { + return Err(Error::Config(ConfigError::Message( + "replication_send_queue_capacity must be >= max_inflight_append_requests".into(), + ))); + } + Ok(()) } } @@ -372,6 +407,13 @@ fn default_max_merge_entries() -> usize { fn default_entries_per_replication() -> u64 { 100 } + +fn default_max_inflight_append_requests() -> usize { + 256 +} +fn default_replication_send_queue_capacity() -> usize { + 1024 +} #[derive(Debug, Serialize, Deserialize, Clone)] pub struct ElectionConfig { #[serde(default = "default_election_timeout_min")] @@ -817,68 +859,10 @@ impl Default for PromotionConfig { fn default_stale_learner_threshold() -> Duration { Duration::from_secs(300) } -/// Defines how Raft log entries are persisted and accessed. -/// -/// All strategies use a configurable [`FlushPolicy`] to control when memory contents -/// are flushed to disk, affecting write latency and durability guarantees. -/// -/// **Note:** Both strategies now fully load all log entries from disk into memory at startup. -/// The in-memory `SkipMap` serves as the primary data structure for reads in all modes. -#[derive(Clone, Debug, Serialize, Deserialize, PartialEq)] -pub enum PersistenceStrategy { - /// Memory-first persistence strategy. - /// - /// - **Write path**: On append, the log entry is first written to the in-memory `SkipMap` and - /// acknowledged immediately. Disk persistence happens asynchronously in the background, - /// governed by [`FlushPolicy`]. - /// - /// - **Read path**: Reads are always served from the in-memory `SkipMap`. - /// - /// - **Startup behavior**: All log entries are loaded from disk into memory at startup. - /// - MemFirst, -} - -/// Controls when in-memory logs should be flushed to disk. -/// -/// Flush is triggered by whichever comes first: -/// - An explicit `flush()` call (immediate, no wait). -/// - `append_entries` calls `write_notify.notify_one()` for an immediate persist+fsync. -/// - The idle safety-net timer fires after `idle_flush_interval_ms` of inactivity. -/// -/// `idle_flush_interval_ms` must be greater than zero. It only fires when no -/// writes have arrived for that duration; normal-path latency is determined by -/// the fsync execution time (drain-then-fsync architecture). -#[derive(Clone, Debug, Serialize, Deserialize, PartialEq)] -pub enum FlushPolicy { - Batch { idle_flush_interval_ms: u64 }, -} /// Configuration parameters for log persistence behavior #[derive(Serialize, Deserialize, Clone, Debug)] pub struct PersistenceConfig { - /// Strategy for persisting Raft logs - /// - /// This controls the trade-off between durability guarantees and performance - /// characteristics. The choice impacts both write throughput and recovery - /// behavior after node failures. - #[serde(default = "default_persistence_strategy")] - pub strategy: PersistenceStrategy, - - /// Flush policy for asynchronous strategies - /// - /// This controls when log entries are flushed to disk. The choice impacts - /// write performance and durability guarantees. - #[serde(default = "default_flush_policy")] - pub flush_policy: FlushPolicy, - - /// Maximum number of in-memory log entries to buffer when using async strategies - /// - /// This acts as a safety valve to prevent memory exhaustion during periods of - /// high write throughput or when disk persistence is slow. - #[serde(default = "default_max_buffered_entries")] - pub max_buffered_entries: usize, - /// Maximum time to wait, on shutdown, for an in-flight fsync task to finish /// before giving up. Bounds close() against a stuck/slow disk β€” the task /// itself is not cancelled, it keeps running in the background regardless. @@ -886,40 +870,12 @@ pub struct PersistenceConfig { pub shutdown_timeout_ms: u64, } -/// Default persistence strategy (optimized for balanced workloads) -fn default_persistence_strategy() -> PersistenceStrategy { - PersistenceStrategy::MemFirst -} - -/// Default flush policy for asynchronous strategies -/// -/// This controls when log entries are flushed to disk. The choice impacts -/// write performance and durability guarantees. -fn default_flush_policy() -> FlushPolicy { - FlushPolicy::Batch { - idle_flush_interval_ms: 1000, - } -} - -/// Default maximum buffered log entries -fn default_max_buffered_entries() -> usize { - 10_000 -} - fn default_shutdown_timeout_ms() -> u64 { 5_000 } impl PersistenceConfig { pub fn validate(&self) -> Result<()> { - let FlushPolicy::Batch { - idle_flush_interval_ms, - } = self.flush_policy; - if idle_flush_interval_ms == 0 { - return Err(Error::Config(ConfigError::Message( - "flush_policy.idle_flush_interval_ms must be greater than 0".into(), - ))); - } if self.shutdown_timeout_ms == 0 { return Err(Error::Config(ConfigError::Message( "shutdown_timeout_ms must be greater than 0".into(), @@ -933,9 +889,6 @@ impl PersistenceConfig { impl Default for PersistenceConfig { fn default() -> Self { Self { - strategy: default_persistence_strategy(), - flush_policy: default_flush_policy(), - max_buffered_entries: default_max_buffered_entries(), shutdown_timeout_ms: default_shutdown_timeout_ms(), } } diff --git a/d-engine-core/src/config/raft_test.rs b/d-engine-core/src/config/raft_test.rs index 7f6e44df..de77d3f7 100644 --- a/d-engine-core/src/config/raft_test.rs +++ b/d-engine-core/src/config/raft_test.rs @@ -330,6 +330,31 @@ fn test_raft_config_propagates_read_actor_validation() { ); } +#[test] +fn test_raft_config_max_pending_append_responses_zero_is_invalid() { + let config = RaftConfig { + max_pending_append_responses: 0, + ..Default::default() + }; + assert!( + config.validate().is_err(), + "max_pending_append_responses = 0 must be rejected (mpsc::channel(0) panics, \ + and the stream can never read a request)" + ); +} + +#[test] +fn test_raft_config_max_pending_append_responses_one_is_valid() { + let config = RaftConfig { + max_pending_append_responses: 1, + ..Default::default() + }; + assert!( + config.validate().is_ok(), + "max_pending_append_responses = 1 is the smallest legal value" + ); +} + /// lease_duration_ms + network_rtt_p99_ms/2 must account for RTT/2 in the safety bound. /// /// Invariant (Raft Β§6.4): lease_duration_ms + rtt_p99_ms/2 < election_timeout_min @@ -414,3 +439,92 @@ fn test_startup_quorum_timeout_is_developer_configurable() { Duration::from_secs(90) ); } + +#[test] +fn test_max_inflight_append_requests_default_is_256() { + let config = RaftConfig::default(); + assert_eq!(config.replication.max_inflight_append_requests, 256); + assert!(config.validate().is_ok()); +} + +#[test] +fn test_max_inflight_append_requests_zero_is_invalid() { + let mut config = RaftConfig::default(); + config.replication.max_inflight_append_requests = 0; + + let err = config.validate().expect_err("a zero window must be rejected"); + assert!( + err.to_string().contains("max_inflight_append_requests"), + "the error must name the offending field, got: {err}" + ); +} + +#[test] +fn test_max_inflight_append_requests_missing_key_uses_default() { + // A config file written before this key existed must still load with the default window. + let config: RaftConfig = load_toml("[replication]\nrpc_append_entries_clock_in_ms = 100\n"); + assert_eq!(config.replication.max_inflight_append_requests, 256); +} + +#[test] +fn test_max_inflight_append_requests_explicit_value_is_honored() { + let config: RaftConfig = load_toml("[replication]\nmax_inflight_append_requests = 3\n"); + assert_eq!(config.replication.max_inflight_append_requests, 3); +} + +#[test] +fn test_replication_send_queue_capacity_default_is_1024() { + let config = RaftConfig::default(); + assert_eq!(config.replication.replication_send_queue_capacity, 1024); +} + +#[test] +fn test_replication_send_queue_capacity_default_covers_default_window() { + let config = RaftConfig::default(); + assert!( + config.replication.replication_send_queue_capacity + >= config.replication.max_inflight_append_requests, + "defaults must not report Full before the in-flight window does" + ); +} + +#[test] +fn test_replication_send_queue_capacity_below_window_is_invalid() { + let mut config = RaftConfig::default(); + config.replication.max_inflight_append_requests = 256; + config.replication.replication_send_queue_capacity = 255; + let err = config.validate().expect_err("capacity below the window must be rejected"); + assert!( + err.to_string().contains("replication_send_queue_capacity"), + "error must name the offending key, got: {err}" + ); +} + +#[test] +fn test_replication_send_queue_capacity_equal_to_window_is_valid() { + let mut config = RaftConfig::default(); + config.replication.max_inflight_append_requests = 256; + config.replication.replication_send_queue_capacity = 256; + config.validate().expect("capacity == window is the tightest valid setting"); +} + +#[test] +fn test_replication_send_queue_capacity_missing_key_uses_default() { + let config: RaftConfig = load_toml("[replication]\nmax_inflight_append_requests = 3\n"); + assert_eq!(config.replication.replication_send_queue_capacity, 1024); +} + +#[test] +fn test_replication_send_queue_capacity_explicit_value_is_honored() { + let config: RaftConfig = load_toml("[replication]\nreplication_send_queue_capacity = 2048\n"); + assert_eq!(config.replication.replication_send_queue_capacity, 2048); +} + +fn load_toml(text: &str) -> RaftConfig { + config::Config::builder() + .add_source(config::File::from_str(text, config::FileFormat::Toml)) + .build() + .expect("toml must parse") + .try_deserialize() + .expect("config must deserialize") +} diff --git a/d-engine-core/src/election/election_handler.rs b/d-engine-core/src/election/election_handler.rs index 79075fbc..020fe81a 100644 --- a/d-engine-core/src/election/election_handler.rs +++ b/d-engine-core/src/election/election_handler.rs @@ -1,14 +1,12 @@ -use std::collections::HashSet; -use std::fmt::Debug; -use std::marker::PhantomData; -use std::sync::Arc; - use async_trait::async_trait; use d_engine_proto::common::LogId; use d_engine_proto::server::election::VoteRequest; use d_engine_proto::server::election::VotedFor; -use futures::StreamExt; -use futures::stream::FuturesUnordered; +use std::collections::HashSet; +use std::fmt::Debug; +use std::marker::PhantomData; +use std::sync::Arc; +use tokio::task::JoinSet; use tracing::debug; use tracing::error; use tracing::info; @@ -84,8 +82,9 @@ where last_log_term, }; - // Dispatch one independent task per peer from the core layer - let mut tasks = FuturesUnordered::new(); + // One task per peer. Dropping the JoinSet aborts RPCs still in flight, so a decided + // election never leaves retry loops behind. + let mut tasks = JoinSet::new(); let mut peer_ids = HashSet::new(); for peer in members { let peer_id = peer.id; @@ -97,29 +96,33 @@ where let transport = transport.clone(); let membership = membership.clone(); let retry = settings.retry.clone(); - tasks.push(tokio::spawn(async move { + tasks.spawn(async move { transport.send_vote_request(peer_id, request, &retry, membership).await - })); + }); } - let mut responses = Vec::new(); - while let Some(result) = tasks.next().await { - match result { - Ok(r) => responses.push(r), + let required = peer_ids.len() + 1; + let mut succeed = 1; // own vote + // Tally as responses arrive and decide as soon as the outcome is known. Waiting + // for every peer lets one dead peer's RPC retries (seconds) hold the Raft loop + // past our own election timeout, discarding a majority we already won. + while let Some(joined) = tasks.join_next().await { + let response = match joined { + Ok(r) => r, Err(e) => { error!("Task failed with error: {:?}", &e); - responses.push(Err(Error::from(NetworkError::TaskFailed(e)))); + Err(Error::from(NetworkError::TaskFailed(e))) } - } - } - - let mut succeed = 1; - for response in responses { + }; match response { Ok(vote_response) => { if vote_response.vote_granted { debug!("send_vote_requests_to_peers success!"); succeed += 1; + if is_majority(succeed, required) { + debug!("send_vote_requests receives majority."); + return Ok(()); + } } else { debug!( "if_higher_term_found({}, {}, false)", @@ -157,19 +160,12 @@ where } } } + debug!( - "send_vote_requests to: {:?} with succeed number = {}", + "failed to receive majority votes: {:?} succeed number = {}", &peer_ids, succeed ); - - let required = peer_ids.len() + 1; - if !peer_ids.is_empty() && is_majority(succeed, required) { - debug!("send_vote_requests receives majority."); - Ok(()) - } else { - debug!("failed to receive majority votes."); - Err(ElectionError::QuorumFailure { required, succeed }.into()) - } + Err(ElectionError::QuorumFailure { required, succeed }.into()) } async fn handle_vote_request( diff --git a/d-engine-core/src/election/election_handler_test.rs b/d-engine-core/src/election/election_handler_test.rs index df5964ec..2f63ef4f 100644 --- a/d-engine-core/src/election/election_handler_test.rs +++ b/d-engine-core/src/election/election_handler_test.rs @@ -972,6 +972,175 @@ mod single_node_election_tests { ); } + /// Transport for the slow-peer election test: peer 2 grants the vote at once, any other + /// peer stays silent for a long time and then fails, like a dead node whose RPC retries + /// are still running. + /// + /// Hand-written instead of `MockTransport`: mockall holds an internal lock while a mock + /// closure runs, so a "slow" closure would also stall the fast peer's call and make the + /// timing depend on which task happened to run first. + struct SlowPeerTransport; + + const SLOW_PEER_DELAY: std::time::Duration = std::time::Duration::from_secs(30); + + #[async_trait::async_trait] + impl crate::Transport for SlowPeerTransport { + async fn send_vote_request( + &self, + peer_id: u32, + _request: VoteRequest, + _retry: &crate::RetryPolicies, + _membership: Arc>, + ) -> crate::Result { + if peer_id == 2 { + return Ok(VoteResponse { + term: 1, + vote_granted: true, + last_log_index: 1, + last_log_term: 1, + }); + } + tokio::time::sleep(SLOW_PEER_DELAY).await; + Err(Error::from(crate::NetworkError::ServiceUnavailable( + "peer is down".to_string(), + ))) + } + + async fn send_cluster_update( + &self, + _req: d_engine_proto::server::cluster::ClusterConfChangeRequest, + _retry: &crate::RetryPolicies, + _membership: Arc>, + ) -> crate::Result { + unimplemented!("not used by the election test") + } + + async fn join_cluster( + &self, + _leader_id: u32, + _request: d_engine_proto::server::cluster::JoinRequest, + _retry: crate::BackoffPolicy, + _membership: Arc>, + ) -> crate::Result { + unimplemented!("not used by the election test") + } + + async fn discover_leader( + &self, + _request: d_engine_proto::server::cluster::LeaderDiscoveryRequest, + _rpc_enable_compression: bool, + _membership: Arc>, + ) -> crate::Result> { + unimplemented!("not used by the election test") + } + + async fn send_snapshot( + &self, + _peer_id: u32, + _metadata: d_engine_proto::server::storage::SnapshotMetadata, + _leader_term: u64, + _state_machine_handler: Arc>, + _membership: Arc>, + _config: crate::SnapshotConfig, + ) -> crate::Result<()> { + unimplemented!("not used by the election test") + } + + async fn open_replication_stream( + &self, + _peer_id: u32, + _membership: Arc>, + _compress: bool, + _send_queue_capacity: usize, + ) -> crate::Result { + unimplemented!("not used by the election test") + } + } + + /// `MockTypeConfig` with `SlowPeerTransport` as the transport. + #[derive(Debug)] + struct SlowVoteTypeConfig; + + impl crate::TypeConfig for SlowVoteTypeConfig { + type R = MockRaftLog; + type SE = crate::MockStorageEngine; + type E = crate::MockElectionCore; + type TR = SlowPeerTransport; + type SM = crate::MockStateMachine; + type M = MockMembership; + type REP = crate::MockReplicationCore; + type C = crate::MockCommitHandler; + type SMH = crate::MockStateMachineHandler; + type SMW = crate::MockStateMachineWriterOps; + type SNP = crate::MockSnapshotPolicy; + type PE = crate::MockPurgeExecutor; + } + + /// Test: broadcast_vote_requests must not wait for slow peers once a majority has voted + /// + /// Scenario: + /// - Three-node cluster: candidate (node 1) + peers 2 and 3 + /// - Peer 2 grants the vote immediately (self + peer 2 = 2/3, a majority) + /// - Peer 3 is dead: its RPC only fails after a long retry sequence + /// + /// Expected: + /// - Returns Ok(()) as soon as the majority is reached, without waiting for peer 3 + /// + /// The caller awaits this inside the Raft loop. If it waits for the dead peer, the + /// candidate's own election timer expires meanwhile and the won election is discarded + /// by the next term, so a cluster with one node down can fail to elect a leader. + /// + /// The clock is paused: time only moves when every task is idle. A broadcast that + /// waits for peer 3 lets the runtime jump to the 1 s timeout first, deterministically. + #[tokio::test(start_paused = true)] + async fn test_broadcast_vote_requests_returns_on_majority_without_waiting_for_slow_peer() { + let election_handler = ElectionHandler::::new(1); + + let mut raft_log_mock = MockRaftLog::new(); + raft_log_mock + .expect_last_log_id() + .times(1) + .returning(|| Some(LogId { index: 1, term: 1 })); + + let mut membership = MockMembership::::new(); + membership.expect_is_single_node_cluster().returning(|| false); + membership.expect_voters().returning(|| { + [2, 3] + .into_iter() + .map(|id| NodeMeta { + id, + address: format!("http://127.0.0.1:5500{id}"), + role: 0, + status: 2, + }) + .collect() + }); + + let node_config = RaftNodeConfig::new().expect("Should create default config"); + let node_config = node_config.validate().expect("Should validate config"); + + let result = tokio::time::timeout( + std::time::Duration::from_secs(1), + election_handler.broadcast_vote_requests( + 1, + Arc::new(membership), + &Arc::new(raft_log_mock), + &Arc::new(SlowPeerTransport), + &Arc::new(node_config), + ), + ) + .await + .expect( + "broadcast_vote_requests waited for the slow peer instead of returning once a \ + majority had voted", + ); + + assert!( + result.is_ok(), + "majority (self + peer 2) must win the election, got: {result:?}" + ); + } + /// Test: broadcast_vote_requests steps down when a peer responds with a higher term /// /// Scenario: diff --git a/d-engine-core/src/event.rs b/d-engine-core/src/event.rs index 3f35dbf3..be59beff 100644 --- a/d-engine-core/src/event.rs +++ b/d-engine-core/src/event.rs @@ -71,12 +71,23 @@ pub enum InternalEvent { durable_index: u64, }, + /// Raw fsync-completion mark β€” NOT yet validated. Consumer must call + /// `raft_log().try_advance_durable_index(mark)`, which re-checks the entry's + /// term before advancing `durable_index`. + FsyncCompleted { + mark: LogId, + sent_at: tokio::time::Instant, + }, + /// AppendEntries result from a per-follower ReplicationWorker back to the Raft loop. /// Leader processes this in handle_append_result: updates match_index, re-calculates commit, /// and drains pending_client_writes when quorum is achieved. + /// + /// `sent_at`: captured when this event was pushed onto `internal_event_tx` AppendResult { follower_id: u32, result: Result, + sent_at: tokio::time::Instant, }, /// Snapshot push result from a per-follower ReplicationWorker back to the Raft loop. diff --git a/d-engine-core/src/lib.rs b/d-engine-core/src/lib.rs index 93333741..387324b2 100644 --- a/d-engine-core/src/lib.rs +++ b/d-engine-core/src/lib.rs @@ -173,6 +173,11 @@ pub(crate) fn if_higher_term_found( /// entries in the logs. If the logs have last entries with different terms, then the log with the /// later term is more up-to-date. If the logs end with the same term, then whichever log is longer /// is more up-to-date. +/// +/// #446: callers must pass the in-memory last-log-id (last_entry_id), never durable_index. +/// A node with an un-fsynced tail must still be able to reject a candidate whose log is +/// genuinely less up to date β€” voting eligibility and commit-durability are separate +/// concerns and must not share the same index source. pub(crate) fn is_target_log_more_recent( my_last_log_index: u64, my_last_log_term: u64, diff --git a/d-engine-core/src/network/mod.rs b/d-engine-core/src/network/mod.rs index 8c80cac6..05006e56 100644 --- a/d-engine-core/src/network/mod.rs +++ b/d-engine-core/src/network/mod.rs @@ -10,6 +10,8 @@ mod background_snapshot_transfer; #[cfg(test)] mod background_snapshot_transfer_test; +#[cfg(test)] +mod peer_update_test; #[cfg(any(test, feature = "__test_support"))] mod snapshot_transfer_gate; @@ -46,7 +48,7 @@ use crate::TypeConfig; /// pushes `AppendEntriesRequest` batches; the receiver yields /// `AppendEntriesResponse` ACKs. Dropping the sender closes the stream. pub struct ReplicationStream { - /// Push AppendEntries batches directly into the open h2 stream (capacity = 128). + /// Push AppendEntries batches directly into the open h2 stream. pub sender: mpsc::Sender, /// Receive ACKs from the follower; items are `Result<_, tonic::Status>`. pub receiver: BoxStream<'static, std::result::Result>, @@ -93,6 +95,17 @@ impl PeerUpdate { success: false, } } + + /// Caps the update at the leader's own log: a peer cannot have matched past + /// `leader_last_index`, and `next_index` is never useful beyond the entry after it. + pub fn bounded_by_leader_log( + mut self, + leader_last_index: u64, + ) -> Self { + self.match_index = self.match_index.map(|index| index.min(leader_last_index)); + self.next_index = self.next_index.min(leader_last_index + 1); + self + } } #[derive(Debug)] @@ -244,6 +257,7 @@ where peer_id: u32, membership: std::sync::Arc>, compress: bool, + send_queue_capacity: usize, ) -> Result; } diff --git a/d-engine-core/src/network/peer_update_test.rs b/d-engine-core/src/network/peer_update_test.rs new file mode 100644 index 00000000..79dd392e --- /dev/null +++ b/d-engine-core/src/network/peer_update_test.rs @@ -0,0 +1,77 @@ +use crate::PeerUpdate; + +fn update( + match_index: Option, + next_index: u64, +) -> PeerUpdate { + PeerUpdate { + match_index, + next_index, + success: true, + } +} + +#[test] +fn test_match_index_past_the_leader_log_is_capped() { + let bounded = update(Some(50), 51).bounded_by_leader_log(20); + + assert_eq!(bounded.match_index, Some(20)); +} + +#[test] +fn test_next_index_past_the_entry_after_the_leader_log_is_capped() { + let bounded = update(Some(50), 51).bounded_by_leader_log(20); + + assert_eq!(bounded.next_index, 21); +} + +#[test] +fn test_update_inside_the_leader_log_is_unchanged() { + let original = update(Some(15), 16); + + assert_eq!(original.clone().bounded_by_leader_log(20), original); +} + +#[test] +fn test_update_exactly_at_the_leader_log_end_is_unchanged() { + let original = update(Some(20), 21); + + assert_eq!(original.clone().bounded_by_leader_log(20), original); +} + +#[test] +fn test_missing_match_index_stays_missing() { + let bounded = update(None, 1).bounded_by_leader_log(20); + + assert_eq!(bounded.match_index, None); +} + +#[test] +fn test_conflict_hint_below_the_leader_log_is_not_raised() { + let conflict = PeerUpdate { + match_index: None, + next_index: 4, + success: false, + }; + + assert_eq!(conflict.clone().bounded_by_leader_log(20), conflict); +} + +#[test] +fn test_empty_leader_log_caps_everything_to_the_start() { + let bounded = update(Some(5), 6).bounded_by_leader_log(0); + + assert_eq!(bounded.match_index, Some(0)); + assert_eq!(bounded.next_index, 1); +} + +#[test] +fn test_success_flag_is_preserved() { + let rejected = PeerUpdate { + match_index: Some(50), + next_index: 51, + success: false, + }; + + assert!(!rejected.bounded_by_leader_log(20).success); +} diff --git a/d-engine-core/src/raft.rs b/d-engine-core/src/raft.rs index c8e2eac6..98e268a9 100644 --- a/d-engine-core/src/raft.rs +++ b/d-engine-core/src/raft.rs @@ -43,6 +43,11 @@ where event_tx: mpsc::Sender, event_rx: mpsc::Receiver, buffered_inbound_event: VecDeque, + /// Set to `Instant::now()` when `buffered_inbound_event` goes emptyβ†’non-empty, + /// cleared on the next `process_inbound_events` drain. Batch-level, not + /// per-event β€” same simplification as `ProposeBatchBuffer::oldest_pending_since`. + /// Measures inbound-event queue wait before this node's event loop drains it (#446). + inbound_event_oldest_pending_since: Option, // Client commands (drain-driven) cmd_tx: mpsc::Sender, @@ -139,6 +144,7 @@ where event_tx: signal_params.event_tx, event_rx: signal_params.event_rx, buffered_inbound_event: VecDeque::new(), + inbound_event_oldest_pending_since: None, cmd_tx: signal_params.cmd_tx, cmd_rx: signal_params.cmd_rx, @@ -278,6 +284,9 @@ where // P4: Other events β€” handle first, drain rest after select Some(inbound_event) = self.event_rx.recv() => { trace!(%self.node_id, ?inbound_event, "receive inbound event"); + if self.buffered_inbound_event.is_empty() { + self.inbound_event_oldest_pending_since = Some(std::time::Instant::now()); + } self.buffered_inbound_event.push_back(inbound_event); self.drain_inbound_events().await?; } @@ -297,7 +306,7 @@ where async fn handle_shutdown(&mut self) -> Result<()> { info!("[Raft:{}] shutdown signal received.", self.node_id); - // Close IO thread BEFORE returning (before runtime shutdown) + // Close the raft log (before runtime shutdown) // This ensures RocksDB file lock is released before tokio runtime shuts down self.ctx.storage.raft_log.close().await; // Unblock any tasks stuck in event_tx.send().await (e.g. gRPC stream handlers). @@ -314,6 +323,9 @@ where while count < max { match self.event_rx.try_recv() { Ok(inbound_event) => { + if self.buffered_inbound_event.is_empty() { + self.inbound_event_oldest_pending_since = Some(std::time::Instant::now()); + } self.buffered_inbound_event.push_back(inbound_event); count += 1; } @@ -354,6 +366,7 @@ where } if count > 0 { trace!("Drained {} client commands", count); + metrics::histogram!("core.raft.client_cmd.batch_size").record(count as f64); } Ok(()) } @@ -372,6 +385,10 @@ where } async fn process_inbound_events(&mut self) -> Result<()> { + if let Some(since) = self.inbound_event_oldest_pending_since.take() { + metrics::histogram!("core.raft.inbound_event_queue_wait_ms") + .record(since.elapsed().as_secs_f64() * 1_000.0); + } while !self.buffered_inbound_event.is_empty() { // Avoid none AE event pop and push into queue if matches!( @@ -475,7 +492,10 @@ where let _ = self.role.drain_read_buffer(); debug!("BecomeFollower"); - self.role = self.role.become_follower()?; + let mut new_role = self.role.become_follower()?; + let withheld_acks = self.role.take_pending_acks(); + new_role.restore_pending_acks(withheld_acks); + self.role = new_role; // Reset vote when stepping down (new term, no vote yet) self.role.state_mut().commit_vote_reset(&self.ctx)?; @@ -494,7 +514,10 @@ where let _ = self.role.drain_read_buffer(); debug!("BecomeCandidate"); - self.role = self.role.become_candidate()?; + let mut new_role = self.role.become_candidate()?; + let withheld_acks = self.role.take_pending_acks(); + new_role.restore_pending_acks(withheld_acks); + self.role = new_role; // No leader during candidate state let current_term = self.role.current_term(); @@ -505,7 +528,10 @@ where } InternalEvent::BecomeLeader => { debug!("BecomeLeader"); - self.role = self.role.become_leader()?; + let mut new_role = self.role.become_leader()?; + let withheld_acks = self.role.take_pending_acks(); + new_role.restore_pending_acks(withheld_acks); + self.role = new_role; // Mark vote as committed (candidate β†’ leader transition) let current_term = self.role.current_term(); @@ -551,7 +577,10 @@ where let _ = self.role.drain_read_buffer(); debug!("BecomeLearner"); - self.role = self.role.become_learner()?; + let mut new_role = self.role.become_learner()?; + let withheld_acks = self.role.take_pending_acks(); + new_role.restore_pending_acks(withheld_acks); + self.role = new_role; // Learner has no leader initially let current_term = self.role.current_term(); @@ -609,10 +638,23 @@ where .handle_log_flushed(durable_index, &self.ctx, &self.internal_event_tx) .await; } + InternalEvent::FsyncCompleted { mark, sent_at } => { + metrics::histogram!("core.raft.fsync.completion_wait_ms") + .record(sent_at.elapsed().as_secs_f64() * 1_000.0); + + if let Some(new_durable) = self.ctx.raft_log().try_advance_durable_index(mark) { + self.role + .handle_log_flushed(new_durable, &self.ctx, &self.internal_event_tx) + .await; + } + } InternalEvent::AppendResult { follower_id, result, + sent_at, } => { + metrics::histogram!("core.raft.central_loop.append_result_queue_wait_ms") + .record(sent_at.elapsed().as_secs_f64() * 1_000.0); debug!("AppendResult: follower_id={}", follower_id); if let Err(e) = self .role @@ -651,7 +693,9 @@ where } InternalEvent::PeerStreamError { peer_id } => { debug!(%peer_id, "PeerStreamError: bidi stream disconnected, resetting next_index"); - self.role.handle_peer_stream_error(peer_id); + if let RaftRole::Leader(leader) = &mut self.role { + leader.handle_peer_stream_error(peer_id); + } } InternalEvent::ZombieDetected(node_id) => { debug!(%node_id, "ZombieDetected: forwarding to leader for BatchRemove"); @@ -687,9 +731,11 @@ where // Peer-state guard: seeding is only valid for the peer's CURRENT // in-flight snapshot attempt. A straggler completion that arrives after // the peer has moved on must not touch next_index. - if self.role.state().peer_replication_state(peer_id) - != PeerReplicationState::Snapshot - { + let in_snapshot_state = matches!( + &self.role, + RaftRole::Leader(leader) if leader.peer_replication_state(peer_id) == PeerReplicationState::Snapshot + ); + if !in_snapshot_state { debug!(%peer_id, "dropping SnapshotPushCompleted: peer not in Snapshot state"); return Ok(()); } diff --git a/d-engine-core/src/raft_role/buffers/propose_batch_buffer.rs b/d-engine-core/src/raft_role/buffers/propose_batch_buffer.rs index ec1af7d1..389773c5 100644 --- a/d-engine-core/src/raft_role/buffers/propose_batch_buffer.rs +++ b/d-engine-core/src/raft_role/buffers/propose_batch_buffer.rs @@ -48,6 +48,8 @@ pub struct ProposeBatchBuffer { pub last_flush: Instant, /// Pre-allocated metric labels for zero-allocation hot path. metrics_labels: Option>, + /// Set when the buffer receives the first payload of a new batch, cleared on flush β€” measures how long the first request waited before the batch got drained (#446). + oldest_pending_since: Option, } impl ProposeBatchBuffer { @@ -59,6 +61,7 @@ impl ProposeBatchBuffer { senders: Vec::with_capacity(initial_capacity), last_flush: Instant::now(), metrics_labels: None, + oldest_pending_since: None, } } @@ -87,6 +90,9 @@ impl ProposeBatchBuffer { payload: EntryPayload, sender: ProposeSender, ) { + if self.payloads.is_empty() { + self.oldest_pending_since = Some(Instant::now()); + } self.payloads.push(payload); self.senders.push(sender); @@ -109,6 +115,11 @@ impl ProposeBatchBuffer { } self.last_flush = Instant::now(); + if let Some(since) = self.oldest_pending_since.take() { + metrics::histogram!("core.raft.propose_buffer.oldest_pending_wait_ms") + .record(since.elapsed().as_secs_f64() * 1_000.0); + } + // Allocate replacement Vecs sized to actual batch length, not a fixed max. // After swap: self holds the exact-sized empty Vecs; caller gets the filled ones. let n = self.payloads.len(); diff --git a/d-engine-core/src/raft_role/candidate_state_test.rs b/d-engine-core/src/raft_role/candidate_state_test.rs index 0be00c92..accb688c 100644 --- a/d-engine-core/src/raft_role/candidate_state_test.rs +++ b/d-engine-core/src/raft_role/candidate_state_test.rs @@ -1632,7 +1632,7 @@ async fn test_new_leader_initializes_empty_buffers() { let mut replication_handler = crate::MockReplicationCore::new(); replication_handler .expect_prepare_batch_requests() - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); raft.ctx.storage.raft_log = Arc::new(raft_log); raft.ctx.handlers.replication_handler = replication_handler; diff --git a/d-engine-core/src/raft_role/follower_state.rs b/d-engine-core/src/raft_role/follower_state.rs index c1e30ebf..b9b380fb 100644 --- a/d-engine-core/src/raft_role/follower_state.rs +++ b/d-engine-core/src/raft_role/follower_state.rs @@ -7,6 +7,7 @@ use d_engine_proto::server::cluster::ClusterConfUpdateResponse; use d_engine_proto::server::cluster::LeaderDiscoveryResponse; use d_engine_proto::server::election::VoteResponse; use d_engine_proto::server::storage::SnapshotMetadata; +use std::collections::BTreeMap; use std::fmt::Debug; use std::marker::PhantomData; use std::sync::Arc; @@ -43,6 +44,7 @@ use crate::RaftNodeConfig; use crate::Result; use crate::StateTransitionError; use crate::TypeConfig; +use crate::role_state::PendingAcks; use crate::role_state::schedule_and_execute_purge; use crate::utils::cluster::error; use crate::utils::cluster_printer::print_role_transition_line; @@ -73,6 +75,10 @@ pub struct FollowerState { /// Last physically purged log index (inclusive) pub last_purged_index: Option, + /// AppendEntries responses withheld pending this node's own durable_index. + /// See `role_state::PendingAcks`. + pending_append_acks: PendingAcks, + // -- Snapshot Management -- /// Prevents concurrent snapshot creation /// @@ -463,6 +469,10 @@ impl RaftRoleState for FollowerState { fn pending_purge_upto_mut(&mut self) -> Option<&mut Option> { Some(&mut self.pending_purge_upto) } + + fn pending_append_acks_mut(&mut self) -> Option<&mut PendingAcks> { + Some(&mut self.pending_append_acks) + } } impl FollowerState { @@ -484,6 +494,7 @@ impl FollowerState { node_config.raft.election.election_timeout_max, )), node_config, + pending_append_acks: BTreeMap::new(), snapshot_in_progress: AtomicBool::new(false), _marker: PhantomData, last_purged_index: None, @@ -511,6 +522,7 @@ impl From<&CandidateState> for FollowerState { )), node_config: candidate_state.node_config.clone(), snapshot_in_progress: AtomicBool::new(false), + pending_append_acks: BTreeMap::new(), last_purged_index: candidate_state.last_purged_index, // scheduled_purge_upto: None, _marker: PhantomData, @@ -527,6 +539,7 @@ impl From<&LeaderState> for FollowerState { leader_state.node_config.raft.election.election_timeout_max, )), node_config: leader_state.node_config.clone(), + pending_append_acks: BTreeMap::new(), snapshot_in_progress: AtomicBool::new( leader_state.snapshot_in_progress.load(Ordering::SeqCst), ), @@ -548,6 +561,7 @@ impl From<&LearnerState> for FollowerState { )), node_config: learner_state.node_config.clone(), snapshot_in_progress: AtomicBool::new(false), + pending_append_acks: BTreeMap::new(), last_purged_index: learner_state.last_purged_index, pending_purge_upto: learner_state.pending_purge_upto, _marker: PhantomData, diff --git a/d-engine-core/src/raft_role/follower_state_test.rs b/d-engine-core/src/raft_role/follower_state_test.rs index d6ab39a8..7d67c1ed 100644 --- a/d-engine-core/src/raft_role/follower_state_test.rs +++ b/d-engine-core/src/raft_role/follower_state_test.rs @@ -994,11 +994,96 @@ async fn test_handle_append_entries_success_from_new_leader() { "Should update commit_index" ); + // RPO=0 (#446): the success ACK is withheld until durable_index reaches the claimed index. + assert!( + resp_rx.try_recv().is_err(), + "ACK must be withheld until durable_index catches up" + ); + + // Simulate LogFlushed: durable_index advances to 1, releasing the withheld ACK. + let (flush_tx, _flush_rx) = mpsc::unbounded_channel(); + state.handle_log_flushed(1, &context, &flush_tx).await; + // Verify: Response with success=true let response = resp_rx.recv().await.expect("should receive response").unwrap(); assert!(response.is_success(), "Response should indicate success"); } +/// Test: the `AppendEntriesResponse` sent back to the leader must report the follower's +/// just-updated term, not whatever term was captured in the `StateSnapshot` before +/// `commit_hard_state` ran. +/// +/// # Why this needs its own test +/// `test_handle_append_entries_success_from_new_leader` above hard-codes `new_leader_term` +/// directly into the mocked response, so it never actually reads what +/// `handle_append_entries_request_workflow` (role_state.rs) passes as the `state_snapshot` +/// argument to `handle_append_entries` β€” it can't catch a caller that passes a stale snapshot. +/// This test's mock instead echoes back `state_snapshot.current_term`, exactly mirroring what +/// the real `ReplicationHandler::handle_append_entries` does +/// (`replication_handler.rs`: `let current_term = state_snapshot.current_term;`), so it's +/// sensitive to whether the caller passes the pre-update or post-update snapshot. +/// +/// # Scenario +/// Follower (term=1) receives AppendEntries from a leader at term=2 β€” its first-ever contact. +/// +/// # Expected (RED until fixed) +/// `response.term == 2` β€” the follower's real term was correctly updated by `commit_hard_state` +/// before responding; the response must reflect that, not the term=1 snapshot taken before it. +#[tokio::test] +async fn test_handle_append_entries_response_reports_updated_term_not_stale_snapshot() { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let (mut context, _temp_dir) = mock_raft_context_with_temp(graceful_rx, None); + + let follower_term = 1; + let new_leader_term = follower_term + 1; + + let mut replication_handler = MockReplicationCore::new(); + replication_handler + .expect_handle_append_entries() + .returning(move |_, state_snapshot, _| { + Ok(AppendResponseWithUpdates { + response: AppendEntriesResponse::success(1, state_snapshot.current_term, None), + commit_index_update: None, + }) + }); + + context.membership = Arc::new(MockMembership::new()); + context.handlers.replication_handler = replication_handler; + + let mut state = + FollowerState::::new(1, context.node_config.clone(), None, None); + state.shared_state_mut().update_current_term(follower_term); + + let append_entries_request = AppendEntriesRequest { + term: new_leader_term, + leader_id: 5, + prev_log_index: 0, + prev_log_term: 0, + entries: vec![], + leader_commit_index: 0, + }; + let (resp_tx, mut resp_rx) = MaybeCloneOneshot::new(); + let inbound_event = InboundEvent::AppendEntries(append_entries_request, vec![resp_tx]); + let (internal_event_tx, _internal_event_rx) = mpsc::unbounded_channel(); + + assert!( + state + .handle_inbound_event(inbound_event, &context, internal_event_tx) + .await + .is_ok(), + "handle_inbound_event should succeed" + ); + + // No claimed log entry in this batch, so RPO=0 withhold doesn't apply β€” response is + // available immediately. + let response = resp_rx.recv().await.expect("should receive response").unwrap(); + assert_eq!( + response.term, new_leader_term, + "AppendEntriesResponse must report the follower's just-updated term ({new_leader_term}), \ + not a StateSnapshot captured before commit_hard_state updated it ({follower_term})" + ); +} + /// Test: FollowerState rejects AppendEntries with stale term /// /// Scenario: @@ -2882,15 +2967,17 @@ async fn test_follower_rejects_strong_consistency_reads() { } // ============================================================================ -// MemFirst ACK Tests +// Withheld-ACK tests (RPO=0, #446) +// +// End-to-end coverage of the AppendEntries workflow deciding to withhold or send. +// The queue mechanics (release / reject / carry across a role transition) are +// unit-tested in `pending_ack_test.rs`. // ============================================================================ -/// Follower ACKs leader immediately after memory write (MemFirst). -/// -/// The IO thread continues to fsync asynchronously. Safety: before commit, -/// the leader's durable_index >= N (quorum uses durable_index). +/// The workflow withholds a success ACK while durable_index is behind the claimed +/// index, and releases it once `handle_log_flushed` reports that index durable. #[tokio::test] -async fn test_follower_acks_immediately_after_memory_write() { +async fn test_follower_withholds_ack_until_durable() { let (_graceful_tx, graceful_rx) = watch::channel(()); let (mut context, _temp_dir) = mock_raft_context_with_temp(graceful_rx, None); @@ -2937,9 +3024,202 @@ async fn test_follower_acks_immediately_after_memory_write() { .is_ok() ); - // MemFirst: ACK sent immediately, no waiting for fsync - let response = resp_rx.try_recv().expect("ACK must be sent immediately after memory write"); - assert!(response.unwrap().is_success()); + // RPO=0: the ACK is withheld while durable_index < claimed index (5). + assert!( + resp_rx.try_recv().is_err(), + "ACK must be withheld until durable_index catches up" + ); + + // Simulate LogFlushed: durable_index advances to 5, releasing the withheld ACK. + let (flush_tx, _flush_rx) = mpsc::unbounded_channel(); + state.handle_log_flushed(appended_index, &context, &flush_tx).await; + + let response = resp_rx.try_recv().expect("ACK must be released once durable").unwrap(); + assert!(response.is_success()); +} + +/// #446: two independent AppendEntries requests (e.g. a leader retry) that both claim +/// the same threshold index must both eventually receive a response β€” the second one +/// landing on `pending_append_acks` must not silently overwrite the first. +#[tokio::test] +async fn test_multiple_requests_at_same_threshold_all_receive_response() { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let (mut context, _temp_dir) = mock_raft_context_with_temp(graceful_rx, None); + + let leader_term = 2u64; + let claimed_index = 5u64; + + let mut replication_handler = MockReplicationCore::new(); + replication_handler + .expect_handle_append_entries() + .times(2) + .returning(move |_, _, _| { + Ok(AppendResponseWithUpdates { + response: AppendEntriesResponse::success( + 1, + leader_term, + Some(LogId { + term: leader_term, + index: claimed_index, + }), + ), + commit_index_update: None, + }) + }); + context.handlers.replication_handler = replication_handler; + context.membership = Arc::new(MockMembership::new()); + + let mut state = + FollowerState::::new(1, context.node_config.clone(), None, None); + state.shared_state_mut().update_current_term(leader_term); + + let append_request = AppendEntriesRequest { + term: leader_term, + leader_id: 2, + prev_log_index: 0, + prev_log_term: 0, + entries: vec![], + leader_commit_index: 0, + }; + let (internal_event_tx, _internal_event_rx) = mpsc::unbounded_channel(); + + // First request lands on threshold=5 and gets withheld. + let (resp_tx1, mut resp_rx1) = MaybeCloneOneshot::new(); + state + .handle_inbound_event( + InboundEvent::AppendEntries(append_request.clone(), vec![resp_tx1]), + &context, + internal_event_tx.clone(), + ) + .await + .unwrap(); + + // A second, independent request (e.g. leader retry) also claims index 5. + let (resp_tx2, mut resp_rx2) = MaybeCloneOneshot::new(); + state + .handle_inbound_event( + InboundEvent::AppendEntries(append_request, vec![resp_tx2]), + &context, + internal_event_tx, + ) + .await + .unwrap(); + + assert!( + resp_rx1.try_recv().is_err(), + "first request must still be withheld" + ); + assert!( + resp_rx2.try_recv().is_err(), + "second request must still be withheld" + ); + + let (flush_tx, _flush_rx) = mpsc::unbounded_channel(); + state.handle_log_flushed(claimed_index, &context, &flush_tx).await; + + // Both senders β€” not just one β€” must receive a response. A single-value + // `BTreeMap` without `.entry().or_insert_with(...)` merging would + // let the second insert silently overwrite the first, dropping this ACK forever. + let response1 = resp_rx1.try_recv().expect("first sender must receive a response").unwrap(); + let response2 = resp_rx2 + .try_recv() + .expect("second sender must also receive a response, not be silently overwritten") + .unwrap(); + assert!(response1.is_success()); + assert!(response2.is_success()); +} + +/// #446: a request from a newer term that claims an index already withheld under an +/// older term must be answered on its own merits β€” not inherit the stale entry's +/// conflict. +/// +/// term 2: request A claims index 5, withheld. +/// term 3: a new leader's request B claims the same index 5, withheld. +/// index 5 becomes durable: A (term 2, now stale) gets a conflict; B (term 3) gets success. +#[tokio::test] +async fn test_newer_term_request_on_same_index_is_not_failed_by_stale_pending_ack() { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let (mut context, _temp_dir) = mock_raft_context_with_temp(graceful_rx, None); + + let claimed_index = 5u64; + let calls = Arc::new(std::sync::atomic::AtomicU64::new(0)); + let calls_in_mock = calls.clone(); + + let mut replication_handler = MockReplicationCore::new(); + replication_handler + .expect_handle_append_entries() + .times(2) + .returning(move |_, _, _| { + let term = if calls_in_mock.fetch_add(1, std::sync::atomic::Ordering::SeqCst) == 0 { + 2 + } else { + 3 + }; + Ok(AppendResponseWithUpdates { + response: AppendEntriesResponse::success( + 1, + term, + Some(LogId { + term, + index: claimed_index, + }), + ), + commit_index_update: None, + }) + }); + context.handlers.replication_handler = replication_handler; + context.membership = Arc::new(MockMembership::new()); + + let mut state = + FollowerState::::new(1, context.node_config.clone(), None, None); + state.shared_state_mut().update_current_term(2); + + let request = |term: u64| AppendEntriesRequest { + term, + leader_id: 2, + prev_log_index: 0, + prev_log_term: 0, + entries: vec![], + leader_commit_index: 0, + }; + let (internal_event_tx, _internal_event_rx) = mpsc::unbounded_channel(); + + let (tx_old, mut rx_old) = MaybeCloneOneshot::new(); + state + .handle_inbound_event( + InboundEvent::AppendEntries(request(2), vec![tx_old]), + &context, + internal_event_tx.clone(), + ) + .await + .unwrap(); + + // A new leader takes over at term 3 before index 5 is durable. The term bump + // must come from the request itself (via `commit_hard_state`), as in production. + let (tx_new, mut rx_new) = MaybeCloneOneshot::new(); + state + .handle_inbound_event( + InboundEvent::AppendEntries(request(3), vec![tx_new]), + &context, + internal_event_tx, + ) + .await + .unwrap(); + + let (flush_tx, _flush_rx) = mpsc::unbounded_channel(); + state.handle_log_flushed(claimed_index, &context, &flush_tx).await; + + let old = rx_old.try_recv().expect("term-2 request must be answered").unwrap(); + assert!( + !old.is_success(), + "term-2 ACK was withheld under a term this node has left β€” must be a conflict" + ); + let new = rx_new.try_recv().expect("term-3 request must be answered").unwrap(); + assert!( + new.is_success(), + "term-3 request claims a durable index under the current term β€” it must not \ + be failed by the stale term-2 entry on the same index" + ); } /// Follower sends ACK immediately for heartbeat (no entries). @@ -3042,8 +3322,18 @@ async fn test_follower_commit_index_and_ack_both_sent_immediately() { new_commit, "commit_index must advance immediately" ); - let response = resp_rx.try_recv().expect("ACK must be sent immediately"); - assert!(response.unwrap().is_success()); + // RPO=0: commit_index advances immediately, but the ACK is withheld until durable. + assert!( + resp_rx.try_recv().is_err(), + "ACK must be withheld until durable_index catches up" + ); + + // Simulate LogFlushed: durable_index advances to 5, releasing the withheld ACK. + let (flush_tx, _flush_rx) = mpsc::unbounded_channel(); + state.handle_log_flushed(appended_index, &context, &flush_tx).await; + + let response = resp_rx.try_recv().expect("ACK must be released once durable").unwrap(); + assert!(response.is_success()); } /// Spawns a fake Worker that answers exactly one `InstallSnapshot` command with `result`, diff --git a/d-engine-core/src/raft_role/leader_state.rs b/d-engine-core/src/raft_role/leader_state.rs index 06b21cb8..79c8ffc6 100644 --- a/d-engine-core/src/raft_role/leader_state.rs +++ b/d-engine-core/src/raft_role/leader_state.rs @@ -68,6 +68,7 @@ use d_engine_proto::server::replication::AppendEntriesResponse; use d_engine_proto::server::replication::append_entries_response; use d_engine_proto::server::storage::SnapshotMetadata; use futures::StreamExt; +use parking_lot::Mutex; use rand::distr::SampleString; use std::collections::BTreeMap; use std::collections::HashMap; @@ -104,6 +105,14 @@ pub(super) struct WriteMetadata { // NOTE: expiry is checked on each tick; actual timeout may exceed deadline by up to // one tick interval (~150-300ms). Acceptable for Raft workloads. pub(super) deadline: Instant, + pub(super) proposed_at: std::time::Instant, +} + +/// A committed write waiting for its state machine result. +pub(super) struct PendingApply { + pub(super) sender: MaybeCloneOneshotSender>, + pub(super) proposed_at: std::time::Instant, + pub(super) committed_at: std::time::Instant, } /// A lease read request waiting for quorum ACK to refresh the lease. @@ -201,15 +210,16 @@ pub struct ClusterMetadata { /// Task routed to a per-follower replication worker. enum ReplicationTask { - /// Normal AppendEntries replication. - Append(AppendEntriesRequest), + /// Normal AppendEntries replication. Second field is the enqueue time, used to + /// measure how long the task waited in `task_tx` before the worker sent it. + Append(AppendEntriesRequest, Instant), /// Peer's next_index fell below the purge boundary; transfer the latest snapshot. Snapshot(SnapshotMetadata, u64), } /// Immutable configuration passed to every per-follower replication worker. /// -/// Bundles the seven items that are always passed together to `spawn_worker`, +/// Bundles the eight items that are always passed together to `spawn_worker`, /// `run_replication_worker`, and `send_to_worker_or_spawn`, keeping those /// function signatures within the clippy `too_many_arguments` limit. struct ReplicationWorkerConfig { @@ -217,11 +227,62 @@ struct ReplicationWorkerConfig { membership: Arc>, retry_policies: RetryPolicies, response_compress_enabled: bool, + send_queue_capacity: usize, internal_event_tx: mpsc::UnboundedSender, state_machine_handler: Arc>, snapshot_config: SnapshotConfig, } +impl ReplicationWorkerConfig { + fn new( + ctx: &RaftContext, + internal_event_tx: &mpsc::UnboundedSender, + ) -> Self { + Self { + transport: ctx.transport.clone(), + membership: ctx.membership.clone(), + retry_policies: ctx.node_config.retry.clone(), + response_compress_enabled: ctx.node_config.raft.rpc_compression.replication_response, + send_queue_capacity: ctx.node_config.raft.replication.replication_send_queue_capacity, + internal_event_tx: internal_event_tx.clone(), + state_machine_handler: ctx.state_machine_handler().clone(), + snapshot_config: ctx.node_config.raft.snapshot.clone(), + } + } +} + +/// Per-peer metric handles, built once so the hot path skips the label +/// allocation and registry lookup that `metrics::gauge!(..)` does per call. +struct PeerMetrics { + in_flight: metrics::Gauge, + match_index: metrics::Gauge, + last_ack_timestamp_seconds: metrics::Gauge, + dispatch_gated_total: metrics::Counter, + in_flight_at_dispatch: metrics::Histogram, +} + +impl PeerMetrics { + fn new(peer_id: u32) -> Self { + let peer = peer_id.to_string(); + Self { + in_flight: metrics::gauge!("core.raft.peer.in_flight", "peer_id" => peer.clone()), + match_index: metrics::gauge!("core.raft.peer.match_index", "peer_id" => peer.clone()), + last_ack_timestamp_seconds: metrics::gauge!( + "core.raft.peer.last_ack_timestamp_seconds", + "peer_id" => peer.clone() + ), + dispatch_gated_total: metrics::counter!( + "core.raft.peer.dispatch_gated_total", + "peer_id" => peer.clone() + ), + in_flight_at_dispatch: metrics::histogram!( + "core.raft.peer.in_flight_at_dispatch", + "peer_id" => peer + ), + } + } +} + /// Handle for a per-follower replication worker task. /// /// Dropping this handle (via `LeaderState::replication_workers`) signals the worker to exit: @@ -265,6 +326,13 @@ pub struct LeaderState { /// keep this change minimal and consistent with existing style. pub(super) peer_replication_state: HashMap, + /// Last log index of each un-ACKed data request per peer, oldest first. + /// Its length is the window occupancy. Heartbeats are never recorded. + in_flight: HashMap>, + + /// Metric handles per peer, created on first use. + peer_metrics: HashMap, + /// === Volatile State === /// Temporary storage for no-op entry log ID during leader initialization #[doc(hidden)] @@ -378,11 +446,10 @@ pub struct LeaderState { // -- Client Request Tracking -- /// Writes committed but awaiting state machine apply before responding. - /// Key: log index. Value: response sender. + /// Key: log index. Value: response sender plus its propose/commit timestamps. /// Populated when `wait_for_apply=true` (e.g. CAS, or any write needing SM result). /// Drained by `handle_apply_completed`; cleared with error on step-down / fatal error. - pub(super) pending_write_apply: - HashMap>>, + pub(super) pending_write_apply: HashMap, /// Pending linearizable reads waiting for SM to apply up to read_index. /// Key: read_index (SM must reach this index before reads can be served) @@ -421,16 +488,6 @@ pub struct LeaderState { /// - `pending_client_writes`: User data, may require SM apply (CAS, etc.) pub(super) pending_commit_actions: BTreeMap, - /// DIAG: wall-clock timestamp when each write entry was registered (propose time). - /// Key: log entry index. Cleared on apply or error drain. - write_propose_times: HashMap, - - /// DIAG: wall-clock timestamp when each write entry's commit was observed. - /// Key: log entry index. Populated in `drain_pending_client_writes`, consumed in - /// `handle_apply_completed` to measure the commitβ†’apply segment in isolation. - /// Cleared on apply or error drain. - write_commit_times: HashMap, - /// Per-follower replication worker handles. Key = follower node_id. /// Workers send AppendEntries via transport and relay results back as InternalEvent::AppendResult. /// Dropped on role change (LeaderState drop) β†’ workers exit via channel close. @@ -517,6 +574,7 @@ impl RaftRoleState for LeaderState { let current = self.match_index.get(&node_id).copied().unwrap_or(0); if new_match_id > current { self.match_index.insert(node_id, new_match_id); + self.peer_metrics(node_id).match_index.set(new_match_id as f64); } Ok(()) } @@ -1188,15 +1246,14 @@ impl RaftRoleState for LeaderState { error!("[Leader] Fatal error from {}: {}", source, error); let fatal_status = || tonic::Status::internal(format!("Node fatal error: {error}")); // Notify all pending write requests - self.write_commit_times.clear(); let pending: Vec<_> = self.pending_write_apply.drain().collect(); if !pending.is_empty() { warn!( "[Leader] FatalError: notifying {} pending write requests", pending.len() ); - for (_index, sender) in pending { - let _ = sender.send(Err(fatal_status())); + for (_index, pending) in pending { + let _ = pending.sender.send(Err(fatal_status())); } } // Notify buffered linearizable reads (pre-flush) @@ -1264,23 +1321,14 @@ impl RaftRoleState for LeaderState { // Match apply results to pending client requests and send responses. let responses: Vec<_> = results .into_iter() - .filter_map(|r| self.pending_write_apply.remove(&r.index).map(|sender| (r, sender))) + .filter_map(|r| self.pending_write_apply.remove(&r.index).map(|p| (r, p))) .collect(); - for (result, sender) in responses { - let e2e_ms = self - .write_propose_times - .remove(&result.index) - .map(|t| t.elapsed().as_secs_f64() * 1000.0) - .unwrap_or(0.0); - if e2e_ms > 0.0 { - metrics::histogram!("core.raft.write.propose_to_apply_ms").record(e2e_ms); - } - if let Some(t) = self.write_commit_times.remove(&result.index) { - let commit_to_apply_ms = t.elapsed().as_secs_f64() * 1000.0; - metrics::histogram!("core.raft.write.commit_to_apply_ms") - .record(commit_to_apply_ms); - } + for (result, pending) in responses { + metrics::histogram!("core.raft.write.propose_to_apply_ms") + .record(pending.proposed_at.elapsed().as_secs_f64() * 1000.0); + metrics::histogram!("core.raft.write.commit_to_apply_ms") + .record(pending.committed_at.elapsed().as_secs_f64() * 1000.0); let response = if result.succeeded { ClientResponse::write_success() @@ -1295,7 +1343,7 @@ impl RaftRoleState for LeaderState { result.succeeded ); - if sender.send(Ok(response)).is_err() { + if pending.sender.send(Ok(response)).is_err() { trace!( "[Leader-{}] Client receiver dropped for index {}", self.node_id(), @@ -1343,18 +1391,11 @@ impl RaftRoleState for LeaderState { ctx: &RaftContext, internal_event_tx: &mpsc::UnboundedSender, ) { - let new_commit_index = if self.cluster_metadata.single_voter { - // MemFirst single-voter: LogFlushed(durable) is the IO checkpoint. - // Commit to last_entry_id() β€” not just durable β€” to allow pipelining - // across IO batch boundaries. Matches multi-voter MemFirst where leader - // contributes last_entry_id() to quorum (not durable_index). - let last_log_index = ctx.raft_log().last_entry_id(); - debug_assert!( - last_log_index >= durable, - "last_entry_id ({last_log_index}) must be >= durable ({durable})" - ); - if last_log_index > self.commit_index() { - Some(last_log_index) + let next_commit_index = if self.cluster_metadata.single_voter { + // RPO=0 (#446): single-voter has no majority to fall back on β€” commit + // must not advance past what this node has itself fsynced. + if durable > self.commit_index() { + Some(durable) } else { None } @@ -1362,10 +1403,11 @@ impl RaftRoleState for LeaderState { // Multi-voter: quorum of match_index determines commit. // Only voter peers (non-Learner role) may contribute to majority. // Learners replicate entries but must never count toward commit quorum. - self.calculate_new_commit_index(ctx.raft_log()) + let (commit_index, majority) = self.majority_matched_index(ctx.raft_log()); + Self::new_commit_index(commit_index, majority) }; - if let Some(new_commit) = new_commit_index { + if let Some(new_commit) = next_commit_index { if let Err(e) = self.update_commit_index_with_signal( Leader as i32, self.current_term(), @@ -1406,24 +1448,24 @@ impl RaftRoleState for LeaderState { let leader_term = self.current_term(); let response = match result { - Err(e) => { + Err(Error::System(SystemError::Network(NetworkError::ResponseChannelClosed))) => { + // If the follower's pending_response_tx was replaced, reset in-flight count. + self.reset_in_flight(follower_id); + // ResponseChannelClosed means the follower's pending_response_tx was replaced // by a newer AppendEntries (pipeline replication). The old sender was dropped, // but the follower still has the entry in memory and will flush it. // Safety: by Raft Log Matching Property, if follower ACKs index=N, // it implicitly confirms all entries up to N including this one. // Do NOT retry β€” the higher-index ACK will update match_index correctly. - if matches!( - e, - Error::System(SystemError::Network(NetworkError::ResponseChannelClosed)) - ) { - debug!( - "follower {} sender superseded (ResponseChannelClosed) β€” \ + debug!( + "follower {} sender superseded (ResponseChannelClosed) β€” \ waiting for higher-index ACK", - follower_id - ); - return Ok(()); - } + follower_id + ); + return Ok(()); + } + Err(e) => { // Real network error β€” log and continue; worker handles retries per BackoffPolicy. warn!( "AppendEntries to peer {} network error: {:?}", @@ -1443,6 +1485,15 @@ impl RaftRoleState for LeaderState { return Ok(()); } + // Any response in our term proves the peer is alive. Window slots are + // released by index in `update_peer_index`, not once per response. + self.peer_metrics(follower_id).last_ack_timestamp_seconds.set( + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap_or_default() + .as_secs_f64(), + ); + // Higher term: step down immediately. if response.term > leader_term { warn!( @@ -1470,12 +1521,19 @@ impl RaftRoleState for LeaderState { .handle_success_response(follower_id, response.term, success, leader_term)?, Some(append_entries_response::Result::Conflict(conflict)) => { let current_next_index = self.next_index.get(&follower_id).copied().unwrap_or(1); - ctx.replication_handler().handle_conflict_response( + let update = ctx.replication_handler().handle_conflict_response( follower_id, conflict, ctx.raft_log(), current_next_index, - )? + ); + if update.is_err() { + // A failed conflict resolution leaves the speculative pipeline suspect; + // clear the window and drop to Probe, or this peer is gated forever. + self.reset_in_flight(follower_id); + self.set_peer_replication_state(follower_id, PeerReplicationState::Probe); + } + update? } Some(append_entries_response::Result::HigherTerm(term)) => { if term > leader_term { @@ -1487,6 +1545,7 @@ impl RaftRoleState for LeaderState { self.send_become_follower_event(None, internal_event_tx)?; return Err(ReplicationError::HigherTerm(term).into()); } + // Not actually higher: a late rejection of an earlier term's request. return Ok(()); } None => { @@ -1494,10 +1553,14 @@ impl RaftRoleState for LeaderState { "AppendResult from peer {} has no result variant", follower_id ); + // Carries no index to release by, so drop the window instead. + self.reset_in_flight(follower_id); return Ok(()); } }; + let peer_update = peer_update.bounded_by_leader_log(ctx.raft_log().last_entry_id()); + // Update next_index and match_index for this peer. self.update_peer_index(follower_id, &peer_update); @@ -1536,7 +1599,10 @@ impl RaftRoleState for LeaderState { // Re-calculate commit index after updating this voter's match_index. if peer_update.success && is_voter { - if let Some(new_commit) = self.calculate_new_commit_index(ctx.raft_log()) { + let (commit_index, majority) = self.majority_matched_index(ctx.raft_log()); + let quorum_confirmed = majority.is_some(); + + if let Some(new_commit) = Self::new_commit_index(commit_index, majority) { self.update_commit_index_with_signal( Leader as i32, self.current_term(), @@ -1550,25 +1616,9 @@ impl RaftRoleState for LeaderState { // Lease refresh and pending_lease_reads drain are triggered by quorum ACK, // independent of whether commit_index advanced. When an expired lease fires an // empty AppendEntries heartbeat, commit_index does not change (nothing new to - // commit), so calculate_new_commit_index returns None and the block above is - // skipped. We must check quorum confirmation separately here. - let quorum_confirmed = ctx - .raft_log() - .calculate_majority_matched_index( - self.current_term(), - self.commit_index(), - self.match_index - .iter() - .filter(|(id, _)| { - self.cluster_metadata.replication_targets.iter().any(|n| { - n.id == **id - && n.role != d_engine_proto::common::NodeRole::Learner as i32 - }) - }) - .map(|(_, idx)| *idx) - .collect(), - ) - .is_some(); + // commit), so new_commit_index yields None and the block above is skipped. + // We still reuse the single majority computation (majority.is_some()) to + // confirm quorum here. if quorum_confirmed { // Anchor deadline to send time (not ACK time) to eliminate the RTT/2 window. // Falls back to now_ms() only in tests that bypass execute_and_process_raft_rpc. @@ -1882,33 +1932,6 @@ impl RaftRoleState for LeaderState { Ok(()) } - - fn peer_replication_state( - &self, - node_id: u32, - ) -> PeerReplicationState { - self.peer_replication_state - .get(&node_id) - .copied() - .unwrap_or(PeerReplicationState::Probe) - } - - fn set_peer_replication_state( - &mut self, - node_id: u32, - state: PeerReplicationState, - ) { - metrics::gauge!( - "core.raft.peer.replication_state", - "peer_id" => node_id.to_string() - ) - .set(match state { - PeerReplicationState::Probe => 0.0, - PeerReplicationState::Replicate => 1.0, - PeerReplicationState::Snapshot => 2.0, - }); - self.peer_replication_state.insert(node_id, state); - } } /// Computes the exponential backoff delay for a snapshot push failure. @@ -1930,6 +1953,49 @@ fn snapshot_push_backoff_duration( std::time::Duration::from_millis(millis) } +/// Records an RTT sample if one is currently outstanding for this peer (#446). +/// No-op fast path: a single atomic load when no sample is pending. +fn record_rtt_sample( + sample_pending: &AtomicBool, + inflight: &Mutex>, + rpc_round_trip_ms: &metrics::Histogram, + result: &Option, +) { + if !sample_pending.load(Ordering::Acquire) { + return; + } + match result { + Some(append_entries_response::Result::Success(success)) => { + let Some(ref last_match) = success.last_match else { + return; + }; + let mut inflight = inflight.lock(); + // Match on "follower's progress has caught up to (or passed) the + // sampled index", not exact equality β€” a fast/deeply-pipelined + // follower can coalesce several acks into one report. At most one + // sample is ever in flight (`sample_pending` gates new inserts). + if let Some((sampled_index, sent_at)) = *inflight + && sampled_index <= last_match.index + { + rpc_round_trip_ms.record(sent_at.elapsed().as_secs_f64() * 1_000.0); + *inflight = None; + } + if inflight.is_none() { + sample_pending.store(false, Ordering::Release); + } + } + // Sampled batch got rejected or superseded β€” abandon this sample instead + // of leaking `sample_pending` forever (neither variant carries a + // `last_match` to key off of). + Some(append_entries_response::Result::Conflict(_)) + | Some(append_entries_response::Result::HigherTerm(_)) => { + *inflight.lock() = None; + sample_pending.store(false, Ordering::Release); + } + None => {} + } +} + impl LeaderState { // ---- Per-follower ReplicationWorker management ---------------------------------------- @@ -1973,6 +2039,7 @@ impl LeaderState { membership, retry_policies, response_compress_enabled, + send_queue_capacity, internal_event_tx, state_machine_handler, snapshot_config, @@ -1981,6 +2048,23 @@ impl LeaderState { let base_delay_ms = retry_policies.append_entries.base_delay_ms; let max_delay_ms = retry_policies.append_entries.max_delay_ms; + // RTT sampling (#446): sample-based replication RPC round-trip measurement. + // 1-in-16 attempt rate, gated to at most one outstanding sample (real rate is + // lower β€” skipped while the previous sample is still unresolved). Match is + // "index reached or passed", not exact equality β€” pipelined followers can + // coalesce several ACKs into one report. + let mut rtt_sample_counter: u64 = 0; + let rtt_sample_pending = Arc::new(AtomicBool::new(false)); + let rtt_inflight: Arc>> = Arc::new(Mutex::new(None)); + let queue_wait_ms = metrics::histogram!( + "core.raft.replication_worker.queue_wait_ms", + "peer_id" => peer_id.to_string() + ); + let rpc_round_trip_ms = metrics::histogram!( + "core.raft.replication.rpc_round_trip_ms", + "peer_id" => peer_id.to_string() + ); + // Open persistent bidi stream with backoff reconnection loop { let mut backoff_ms = base_delay_ms; @@ -1990,14 +2074,29 @@ impl LeaderState { return; } match transport - .open_replication_stream(peer_id, membership.clone(), response_compress_enabled) + .open_replication_stream( + peer_id, + membership.clone(), + response_compress_enabled, + send_queue_capacity, + ) .await { Ok(s) => { info!(peer_id, "Bidi replication stream established"); + metrics::counter!( + "core.raft.replication_worker.stream_established_total", + "peer_id" => peer_id.to_string() + ) + .increment(1); break s; } Err(e) => { + metrics::counter!( + "core.raft.replication_worker.stream_open_failed_total", + "peer_id" => peer_id.to_string() + ) + .increment(1); info!( peer_id, "Bidi stream reconnecting (backoff={}ms): {:?}", backoff_ms, e @@ -2016,14 +2115,24 @@ impl LeaderState { let stream_broken = Arc::new(AtomicBool::new(false)); // Spawn recv task to monitor ACKs and detect stream disconnection let recv_internal_event_tx = internal_event_tx.clone(); + let recv_rtt_sample_pending = rtt_sample_pending.clone(); + let recv_rtt_inflight = rtt_inflight.clone(); + let recv_rpc_round_trip_ms = rpc_round_trip_ms.clone(); let mut recv_handle = tokio::spawn(async move { loop { match stream_receiver.next().await { Some(Ok(response)) => { + record_rtt_sample( + &recv_rtt_sample_pending, + &recv_rtt_inflight, + &recv_rpc_round_trip_ms, + &response.result, + ); let _ = recv_internal_event_tx.send(InternalEvent::AppendResult { follower_id: peer_id, result: Ok(response), + sent_at: Instant::now(), }); } Some(Err(status)) => { @@ -2056,10 +2165,41 @@ impl LeaderState { } match task { - ReplicationTask::Append(request) => { - // Push batch directly into the persistent bidi stream (non-blocking) - if stream_sender.send(request).await.is_err() { - warn!(peer_id, "Bidi stream sender closed, reconnecting"); + ReplicationTask::Append(request, enqueued_at) => { + queue_wait_ms.record(enqueued_at.elapsed().as_secs_f64() * 1_000.0); + + // RTT sampling (#446): start a new RTT sample every 16th send, only + // if no sample is currently outstanding. See declaration above. + rtt_sample_counter += 1; + if rtt_sample_counter.is_multiple_of(16) + && !rtt_sample_pending.load(Ordering::Acquire) + && let Some(last_entry) = request.entries.last() + { + *rtt_inflight.lock() = Some((last_entry.index, Instant::now())); + rtt_sample_pending.store(true, Ordering::Release); + } + + // Push batch into the persistent bidi stream's bounded send channel + // (capacity = `replication_send_queue_capacity`) β€” `try_send` fails + // fast once that buffer is full, i.e. once the peer has fallen more + // than that many requests behind draining it. + let send_result = stream_sender.try_send(request); + + if let Err(err) = send_result { + let reason = match err { + tokio::sync::mpsc::error::TrySendError::Full(_) => "full", + tokio::sync::mpsc::error::TrySendError::Closed(_) => { + "closed" + } + }; + metrics::counter!( + "core.raft.replication_worker.send_failed_total", + "peer_id" => peer_id.to_string(), + "reason" => reason + ) + .increment(1); + warn!(peer_id, ?err, "Bidi stream send failed (closed or buffer full), demoting peer and reconnecting"); + stream_broken.store(true, Ordering::Release); let _ = internal_event_tx.send(InternalEvent::PeerStreamError { peer_id }); @@ -2103,7 +2243,19 @@ impl LeaderState { debug!(peer_id, "Replication worker exiting (leader stepped down)"); break; } - // Otherwise reconnect (flag persists across reconnects) + // Otherwise reconnect. abort() cancels the dead recv_handle at its next poll + // point, not at completion, so it never reaches its own reset logic (the + // Success/Conflict branches above) β€” an in-flight sample must be reset here + // instead, or rtt_sample_pending stays stuck true forever (#446). + if rtt_sample_pending.swap(false, Ordering::AcqRel) { + *rtt_inflight.lock() = None; + debug!(peer_id, "RTT sample reset after reconnect"); + metrics::counter!( + "core.raft.replication.rtt_sample_interrupted_by_reconnect", + "peer_id" => peer_id.to_string() + ) + .increment(1); + } } } @@ -2113,7 +2265,8 @@ impl LeaderState { &mut self, peer_id: u32, mut task: ReplicationTask, - cfg: ReplicationWorkerConfig, + ctx: &RaftContext, + internal_event_tx: &mpsc::UnboundedSender, ) { // Try existing worker; if dead, recover task and fall through to respawn. if let Some(handle) = self.replication_workers.get(&peer_id) { @@ -2126,7 +2279,10 @@ impl LeaderState { } } // Worker absent or just died β€” spawn and send. - let handle = Self::spawn_worker(peer_id, cfg); + let handle = Self::spawn_worker( + peer_id, + ReplicationWorkerConfig::new(ctx, internal_event_tx), + ); let _ = handle.task_tx.send(task); self.replication_workers.insert(peer_id, handle); } @@ -2220,17 +2376,27 @@ impl LeaderState { // Instant::now() calls would be redundant syscalls. let commit_ts = std::time::Instant::now(); for (_, meta) in committed { - let (start_idx, senders, wait_for_apply) = - (meta.start_idx, meta.senders, meta.wait_for_apply); + let (start_idx, senders, wait_for_apply, proposed_at) = ( + meta.start_idx, + meta.senders, + meta.wait_for_apply, + meta.proposed_at, + ); + if wait_for_apply { + let propose_to_commit_ms = + commit_ts.duration_since(proposed_at).as_secs_f64() * 1000.0; for (i, sender) in senders.into_iter().enumerate() { - let idx = start_idx + i as u64; - if let Some(t) = self.write_propose_times.get(&idx) { - let ms = t.elapsed().as_secs_f64() * 1000.0; - metrics::histogram!("core.raft.write.propose_to_commit_ms").record(ms); - } - self.write_commit_times.insert(idx, commit_ts); - self.pending_write_apply.insert(idx, sender); + metrics::histogram!("core.raft.write.propose_to_commit_ms") + .record(propose_to_commit_ms); + self.pending_write_apply.insert( + start_idx + i as u64, + PendingApply { + sender, + proposed_at, + committed_at: commit_ts, + }, + ); } } else { for sender in senders { @@ -2399,7 +2565,6 @@ impl LeaderState { &mut self, error_code: ErrorCode, ) { - self.write_propose_times.clear(); for (_, meta) in std::mem::take(&mut self.pending_client_writes) { for sender in meta.senders { let _ = sender.send(Ok(ClientResponse::client_error(error_code))); @@ -2583,6 +2748,10 @@ impl LeaderState { update: &PeerUpdate, ) { if update.success { + if let Some(match_index) = update.match_index { + self.release_in_flight_up_to(follower_id, match_index); + } + // Success: trust speculative advance β€” never regress next_index below what // the leader has already pipeline-sent. ACK confirms a lower bound only; // the leader may have sent further batches speculatively. @@ -2594,6 +2763,10 @@ impl LeaderState { // advance again. self.set_peer_replication_state(follower_id, PeerReplicationState::Replicate); } else { + // A conflict rejects the pipeline. Probe -> Probe is not a state change, + // so `set_peer_replication_state` below would not clear the window. + self.reset_in_flight(follower_id); + // Conflict: follower explicitly corrects our assumption β€” allow next_index // to retreat to the hint. Floor guard (match_index + 1) inside // update_next_index prevents retreating below confirmed entries. @@ -2826,34 +2999,46 @@ impl LeaderState { Ok(()) } - /// Calculate new submission index - fn calculate_new_commit_index( + /// Compute the majority-matched index for the current term. + /// + /// Returns `(commit_index, majority)` where `commit_index` is the base used as + /// the quorum floor, so callers apply the advance rule against the *same* base + /// the majority was computed with. + fn majority_matched_index( &self, raft_log: &Arc>, - ) -> Option { - let old_commit_index = self.commit_index(); + ) -> (u64, Option) { + let commit_index = self.commit_index(); let current_term = self.current_term(); let replication_targets = &self.cluster_metadata.replication_targets; let learner_role = d_engine_proto::common::NodeRole::Learner as i32; // Only voter peers (non-Learner) contribute to the commit quorum. - let matched_ids: Vec = self - .match_index - .iter() - .filter(|(id, _)| { - replication_targets.iter().any(|n| n.id == **id && n.role != learner_role) - }) - .map(|(_, idx)| *idx) - .collect(); + let mut matched_ids = Vec::with_capacity(self.match_index.len() + 1); + matched_ids.extend( + self.match_index + .iter() + .filter(|(id, _)| { + replication_targets.iter().any(|n| n.id == **id && n.role != learner_role) + }) + .map(|(_, idx)| *idx), + ); - let new_commit_index = - raft_log.calculate_majority_matched_index(current_term, old_commit_index, matched_ids); + let majority = + raft_log.calculate_majority_matched_index(current_term, commit_index, matched_ids); + (commit_index, majority) + } - if new_commit_index.is_some() && new_commit_index.unwrap() > old_commit_index { - new_commit_index - } else { - None - } + /// Turn a majority-matched index into the new commit index. + /// + /// Pure function: compares against the same `commit_index` that `majority` was + /// computed with. Callers must not hand-write the `>` β€” this is the single + /// source of truth for strict advance. + fn new_commit_index( + commit_index: u64, + majority: Option, + ) -> Option { + majority.filter(|m| *m > commit_index) } /// Calculate safe read index for linearizable reads. @@ -3061,6 +3246,8 @@ impl LeaderState { node_id: u32, node_config: Arc, ) -> Self { + publish_replication_config_gauges(&node_config.raft.replication); + let ReplicationConfig { rpc_append_entries_clock_in_ms, .. @@ -3103,9 +3290,9 @@ impl LeaderState { replication_workers: HashMap::new(), pending_client_writes: BTreeMap::new(), pending_commit_actions: BTreeMap::new(), - write_propose_times: HashMap::new(), - write_commit_times: HashMap::new(), peer_replication_state: HashMap::new(), + in_flight: HashMap::new(), + peer_metrics: HashMap::new(), _marker: PhantomData, } } @@ -3155,6 +3342,7 @@ impl LeaderState { senders: all_senders, wait_for_apply: any_wait_for_apply, deadline: Instant::now(), // placeholder; overwritten in Phase 2 + proposed_at: std::time::Instant::now(), // placeholder; overwritten in Phase 2 }), ) } @@ -3190,6 +3378,19 @@ impl LeaderState { // ensuring lease expires at send_ts + lease_duration_ms rather than ack_ts + lease_duration_ms. self.last_heartbeat_send_ts = now_ms(); + // Pre-compute in-flight gating decisions for all replication targets. Single + // source of truth for Phase 2 (retrieve_to_be_synced_logs_for_peers, skips entry + // prep for gated peers) and Phase 5 (dispatch gate, skips non-heartbeat sends for + // gated peers) below. Computed once and threaded through both call sites β€” if the + // two phases computed this separately, a divergence could silently block heartbeat + // delivery and stall replication (Raft liveness violation). + let peer_gating_decisions: HashMap = self + .cluster_metadata + .replication_targets + .iter() + .map(|peer| (peer.id, self.should_gate_by_inflight(peer.id))) + .collect(); + // Phase 1: write entries to local log + prepare per-peer requests (serial in Raft loop). let requests = match ctx .replication_handler() @@ -3199,6 +3400,7 @@ impl LeaderState { self.leader_state_snapshot(), &self.cluster_metadata, ctx, + &peer_gating_decisions, ) .await { @@ -3233,12 +3435,10 @@ impl LeaderState { && !meta.senders.is_empty() { let end_log_index = meta.start_idx + meta.senders.len() as u64 - 1; - let propose_ts = std::time::Instant::now(); + meta.proposed_at = std::time::Instant::now(); meta.deadline = Instant::now() + Duration::from_millis(self.node_config.raft.general_raft_timeout_duration_in_ms); - for i in 0..meta.senders.len() as u64 { - self.write_propose_times.insert(meta.start_idx + i, propose_ts); - } + self.pending_client_writes.insert(end_log_index, meta); } @@ -3300,13 +3500,6 @@ impl LeaderState { return Ok(()); } - let transport = ctx.transport.clone(); - let membership = ctx.membership.clone(); - let retry_policies = ctx.node_config.retry.clone(); - let response_compress_enabled = ctx.node_config.raft.rpc_compression.replication_response; - let state_machine_handler = ctx.state_machine_handler().clone(); - let snapshot_config = ctx.node_config.raft.snapshot.clone(); - // Phase 5: fire AppendEntries requests to per-follower workers (non-blocking). for (peer_id, request, effective_next_index) in requests.append_requests { // Leader-owned authority: never generate/dispatch an Append while this peer @@ -3317,6 +3510,16 @@ impl LeaderState { continue; } + let is_heartbeat = request.entries.is_empty(); + + // Phase 1: lightweight in-flight gate (Probe state only, window=1). + // For Replicate state, window check happens in Phase 2 (retrieve_to_be_synced_logs_for_peers). + // Gated peers get no data; heartbeats bypass the gate and never occupy a slot. + if !is_heartbeat && peer_gating_decisions.get(&peer_id).copied().unwrap_or(false) { + self.peer_metrics(peer_id).dispatch_gated_total.increment(1); + continue; + } + // #436: write next_index once. When Replicate, trust speculative advance // (always >= effective_next_index, so no separate write is needed for it). let next_index = @@ -3329,18 +3532,22 @@ impl LeaderState { error!("failed to update next_index peer={}: {:?}", peer_id, e); } + if !is_heartbeat { + let occupancy = self.in_flight_count(peer_id) as f64; + self.peer_metrics(peer_id).in_flight_at_dispatch.record(occupancy); + metrics::histogram!("core.raft.replication.entries_per_request") + .record(request.entries.len() as f64); + + if let Some(last) = request.entries.last() { + self.record_in_flight(peer_id, last.index); + } + } + self.send_to_worker_or_spawn( peer_id, - ReplicationTask::Append(request), - ReplicationWorkerConfig { - transport: transport.clone(), - membership: membership.clone(), - retry_policies: retry_policies.clone(), - response_compress_enabled, - internal_event_tx: internal_event_tx.clone(), - state_machine_handler: state_machine_handler.clone(), - snapshot_config: snapshot_config.clone(), - }, + ReplicationTask::Append(request, Instant::now()), + ctx, + internal_event_tx, ); } @@ -3371,15 +3578,8 @@ impl LeaderState { self.send_to_worker_or_spawn( peer_id, ReplicationTask::Snapshot(metadata.clone(), self.current_term()), - ReplicationWorkerConfig { - transport: transport.clone(), - membership: membership.clone(), - retry_policies: retry_policies.clone(), - response_compress_enabled, - internal_event_tx: internal_event_tx.clone(), - state_machine_handler: state_machine_handler.clone(), - snapshot_config: snapshot_config.clone(), - }, + ctx, + internal_event_tx, ); metrics::gauge!( "core.raft.peer.snapshot_in_progress", @@ -3593,6 +3793,7 @@ impl LeaderState { senders: req.senders, wait_for_apply: req.wait_for_apply_event, deadline: Instant::now(), // placeholder; overwritten in Phase 2 + proposed_at: std::time::Instant::now(), // placeholder; overwritten in Phase 2 }), ) } else { @@ -3807,10 +4008,145 @@ impl LeaderState { let response = ClientResponse::read_results(results); let _ = sender.send(Ok(response)); } + + pub(crate) fn peer_replication_state( + &self, + node_id: u32, + ) -> PeerReplicationState { + self.peer_replication_state + .get(&node_id) + .copied() + .unwrap_or(PeerReplicationState::Probe) + } + + pub(crate) fn set_peer_replication_state( + &mut self, + node_id: u32, + state: PeerReplicationState, + ) { + let previous = self.peer_replication_state.insert(node_id, state); + if previous != Some(state) { + self.reset_in_flight(node_id); + metrics::gauge!( + "core.raft.peer.replication_state", + "peer_id" => node_id.to_string() + ) + .set(match state { + PeerReplicationState::Probe => 0.0, + PeerReplicationState::Replicate => 1.0, + PeerReplicationState::Snapshot => 2.0, + }); + } + } + + /// Number of un-acknowledged AppendEntries requests currently outstanding + /// for `peer_id`. `Probe` state is gated at a fixed window of 1; + /// `Replicate` state is gated by + /// `raft.replication.max_inflight_append_requests`. + pub(super) fn in_flight_count( + &self, + peer_id: u32, + ) -> usize { + self.in_flight.get(&peer_id).map_or(0, VecDeque::len) + } + + /// Records a data request right before it is dispatched to `peer_id`'s worker. + /// `last_index` is the index of its last entry; requests go out in `next_index` + /// order, so the queue stays ascending. Never call for heartbeats. + pub(super) fn record_in_flight( + &mut self, + peer_id: u32, + last_index: u64, + ) { + let queue = self.in_flight.entry(peer_id).or_default(); + debug_assert!(queue.back().is_none_or(|&back| back <= last_index)); + queue.push_back(last_index); + let len = queue.len(); + self.publish_in_flight(peer_id, len); + } + + fn publish_in_flight( + &mut self, + peer_id: u32, + count: usize, + ) { + self.peer_metrics(peer_id).in_flight.set(count as f64); + } + + fn peer_metrics( + &mut self, + peer_id: u32, + ) -> &PeerMetrics { + self.peer_metrics.entry(peer_id).or_insert_with(|| PeerMetrics::new(peer_id)) + } + + /// Frees every request whose last entry is `<= match_index` (cumulative ACK). + /// An older or repeated `match_index` frees nothing. + pub(super) fn release_in_flight_up_to( + &mut self, + peer_id: u32, + match_index: u64, + ) { + if let Some(queue) = self.in_flight.get_mut(&peer_id) { + while queue.front().is_some_and(|&last| last <= match_index) { + queue.pop_front(); + } + let len = queue.len(); + self.publish_in_flight(peer_id, len); + } + } + + /// Drops all outstanding requests of `peer_id`; their ACKs will never arrive + /// (stream error, conflict) or can no longer be trusted. + pub(super) fn reset_in_flight( + &mut self, + peer_id: u32, + ) { + self.in_flight.entry(peer_id).or_default().clear(); + self.publish_in_flight(peer_id, 0); + } + + /// Check if this peer's in-flight window is full and should be gated. + /// + /// Used by both Phase 2 (retrieve_to_be_synced_logs_for_peers) and Phase 5 (dispatch gate) + /// to ensure consistent window management. Heartbeats always bypass this gate. + /// + /// Returns true if in_flight_count >= max_window for this peer's state. + fn should_gate_by_inflight( + &self, + peer_id: u32, + ) -> bool { + let peer_in_flight = self.in_flight_count(peer_id); + let max_window = if self.peer_replication_state(peer_id) == PeerReplicationState::Probe { + 1 + } else { + self.node_config.raft.replication.max_inflight_append_requests + }; + peer_in_flight >= max_window + } + + /// Reset `next_index[peer] = match_index[peer] + 1` after a bidi stream disconnect. + /// Ensures the next heartbeat re-sends any unACKed in-flight entries. + /// Moved here from `raft_role/mod.rs` β€” this is leader-only, the generic + /// `RaftRole`-level dispatch was the anti-pattern that hid the `in_flight` gap. + pub(crate) fn handle_peer_stream_error( + &mut self, + peer_id: u32, + ) { + if self.peer_replication_state(peer_id) == PeerReplicationState::Snapshot { + return; + } + let match_idx = self.match_index(peer_id).unwrap_or(0); + let _ = self.update_next_index(peer_id, match_idx + 1); + self.reset_in_flight(peer_id); + self.set_peer_replication_state(peer_id, PeerReplicationState::Probe); + } } impl From<&CandidateState> for LeaderState { fn from(candidate: &CandidateState) -> Self { + publish_replication_config_gauges(&candidate.node_config.raft.replication); + let ReplicationConfig { rpc_append_entries_clock_in_ms, .. @@ -3858,9 +4194,9 @@ impl From<&CandidateState> for LeaderState { replication_workers: HashMap::new(), pending_client_writes: BTreeMap::new(), pending_commit_actions: BTreeMap::new(), - write_propose_times: HashMap::new(), - write_commit_times: HashMap::new(), peer_replication_state: HashMap::new(), + in_flight: HashMap::new(), + peer_metrics: HashMap::new(), _marker: PhantomData, } } @@ -3901,6 +4237,15 @@ pub fn calculate_safe_batch_size( } } +fn publish_replication_config_gauges(cfg: &ReplicationConfig) { + metrics::gauge!("core.raft.config.max_inflight_append_requests") + .set(cfg.max_inflight_append_requests as f64); + metrics::gauge!("core.raft.config.append_entries_max_entries_per_replication") + .set(cfg.append_entries_max_entries_per_replication as f64); + metrics::gauge!("core.raft.config.replication_send_queue_capacity") + .set(cfg.replication_send_queue_capacity as f64); +} + // ============================================================================ // Test submodules β€” declared here (not in a sibling `leader_state_test` module) // so they are true descendants of this module and can see its private items @@ -3999,3 +4344,15 @@ mod state_management_test; #[cfg(test)] #[path = "leader_state_test/worker_lifecycle_test.rs"] mod worker_lifecycle_test; + +#[cfg(test)] +#[path = "leader_state_test/probe_backpressure_test.rs"] +mod probe_backpressure_test; + +#[cfg(test)] +#[path = "leader_state_test/replicate_unbounded_dispatch_test.rs"] +mod replicate_unbounded_dispatch_test; + +#[cfg(test)] +#[path = "leader_state_test/heartbeat_scenarios_test.rs"] +mod heartbeat_scenarios_test; diff --git a/d-engine-core/src/raft_role/leader_state_test/buffer_cleanup_test.rs b/d-engine-core/src/raft_role/leader_state_test/buffer_cleanup_test.rs index aa9e7847..7eadda1c 100644 --- a/d-engine-core/src/raft_role/leader_state_test/buffer_cleanup_test.rs +++ b/d-engine-core/src/raft_role/leader_state_test/buffer_cleanup_test.rs @@ -66,7 +66,7 @@ async fn test_leader_stepdown_clears_pending_write_buffer() { let mut replication_handler = crate::MockReplicationCore::new(); replication_handler .expect_prepare_batch_requests() - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); raft.ctx.storage.raft_log = Arc::new(raft_log); raft.ctx.handlers.replication_handler = replication_handler; diff --git a/d-engine-core/src/raft_role/leader_state_test/client_read_test.rs b/d-engine-core/src/raft_role/leader_state_test/client_read_test.rs index 1f940527..7eb4bfa0 100644 --- a/d-engine-core/src/raft_role/leader_state_test/client_read_test.rs +++ b/d-engine-core/src/raft_role/leader_state_test/client_read_test.rs @@ -409,7 +409,7 @@ async fn test_linearizable_read_quorum_failure() { replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Err(Error::Fatal("Quorum verification failed".to_string()))); + .returning(|_, _, _, _, _, _| Err(Error::Fatal("Quorum verification failed".to_string()))); let (_graceful_tx, graceful_rx) = watch::channel(()); @@ -489,7 +489,7 @@ async fn test_linearizable_read_quorum_success() { replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| { + .returning(|_, _, _, _, _, _| { // New architecture: prepare_batch_requests returns empty vector for read-only batches // Reads fire in Phase 3 when last_applied >= read_index Ok(crate::PrepareResult::default()) @@ -600,14 +600,13 @@ async fn test_linearizable_read_quorum_success() { async fn test_linearizable_read_encounters_higher_term() { // Given: Leader with higher term detected during prepare phase let mut replication_handler = MockReplicationCore::new(); - replication_handler - .expect_prepare_batch_requests() - .times(1) - .returning(move |_, _, _, _, _| { + replication_handler.expect_prepare_batch_requests().times(1).returning( + move |_, _, _, _, _, _| { // New architecture: Higher term detection now happens via handle_append_result // For testing, we simulate fatal error during prepare phase Err(Error::Fatal("Higher term detected".to_string())) - }); + }, + ); let expect_new_commit_index = 3; let mut raft_log = MockRaftLog::new(); @@ -869,7 +868,7 @@ async fn test_unspecified_policy_defaults_to_linearizable_read() { replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| { + .returning(|_, _, _, _, _, _| { // Unspecified policy defaults to LinearizableRead Ok(crate::PrepareResult::default()) }); @@ -1086,10 +1085,13 @@ async fn test_linearizable_read_batch_shared_quorum() { // Mock replication handler - expect exactly 1 call for the entire batch let mut replication = MockReplicationCore::new(); - replication.expect_prepare_batch_requests().times(1).returning(|_, _, _, _, _| { - // Multiple requests batched naturally - Ok(crate::PrepareResult::default()) - }); + replication + .expect_prepare_batch_requests() + .times(1) + .returning(|_, _, _, _, _, _| { + // Multiple requests batched naturally + Ok(crate::PrepareResult::default()) + }); let ctx = MockBuilder::new(shutdown_rx) .with_db_path("/tmp/test_linearizable_read_batch_shared_quorum") @@ -1175,10 +1177,13 @@ async fn test_lease_reuse_after_linearizable_read_refresh() { // Mock replication - expect only 1 call (from LinearizableRead only) let mut replication = MockReplicationCore::new(); - replication.expect_prepare_batch_requests().times(1).returning(|_, _, _, _, _| { - // Linearizable read refreshes lease (via handle_log_flushed in single-voter) - Ok(crate::PrepareResult::default()) - }); + replication + .expect_prepare_batch_requests() + .times(1) + .returning(|_, _, _, _, _, _| { + // Linearizable read refreshes lease (via handle_log_flushed in single-voter) + Ok(crate::PrepareResult::default()) + }); // MemFirst: handle_log_flushed(1) commits to last_entry_id(), must be >= durable=1. let mut raft_log = MockRaftLog::new(); @@ -1433,10 +1438,13 @@ async fn test_client_policy_override_denied() { // Mock replication handler for LinearizableRead quorum verification let mut replication = MockReplicationCore::new(); - replication.expect_prepare_batch_requests().times(1).returning(|_, _, _, _, _| { - // Single-voter cluster: reads fire immediately in Phase 3 - Ok(crate::PrepareResult::default()) - }); + replication + .expect_prepare_batch_requests() + .times(1) + .returning(|_, _, _, _, _, _| { + // Single-voter cluster: reads fire immediately in Phase 3 + Ok(crate::PrepareResult::default()) + }); let ctx = MockBuilder::new(shutdown_rx) .with_db_path("/tmp/test_client_policy_override_denied") @@ -1511,10 +1519,13 @@ async fn test_drain_single_request_no_delay() { // Mock replication for quorum verification let mut replication = MockReplicationCore::new(); - replication.expect_prepare_batch_requests().times(1).returning(|_, _, _, _, _| { - // Server enforces LinearizableRead despite client request - Ok(crate::PrepareResult::default()) - }); + replication + .expect_prepare_batch_requests() + .times(1) + .returning(|_, _, _, _, _, _| { + // Server enforces LinearizableRead despite client request + Ok(crate::PrepareResult::default()) + }); let ctx = MockBuilder::new(shutdown_rx) .with_db_path("/tmp/test_drain_single_request") @@ -1591,10 +1602,13 @@ async fn test_drain_multiple_requests_natural_batch() { // Mock replication - expect single call for entire batch let mut replication = MockReplicationCore::new(); - replication.expect_prepare_batch_requests().times(1).returning(|_, _, _, _, _| { - // Single request immediately processed - Ok(crate::PrepareResult::default()) - }); + replication + .expect_prepare_batch_requests() + .times(1) + .returning(|_, _, _, _, _, _| { + // Single request immediately processed + Ok(crate::PrepareResult::default()) + }); let ctx = MockBuilder::new(shutdown_rx) .with_db_path("/tmp/test_drain_multiple_requests") @@ -1765,7 +1779,7 @@ async fn test_linearizable_read_batch_single_quorum() { replication .expect_prepare_batch_requests() .times(1) // KEY: Single quorum check for all requests - .returning(|_, _, _, _, _| { + .returning(|_, _, _, _, _, _| { // Batch optimization: single quorum for all 5 reads Ok(crate::PrepareResult::default()) }); @@ -1912,7 +1926,7 @@ async fn test_linearizable_read_rejected_when_noop_not_committed() { replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let mut node_config = RaftNodeConfig::default(); node_config.raft.batching.max_batch_size = 1; @@ -1982,7 +1996,7 @@ async fn test_lease_read_empty_payload_verification_hangs_in_multi_node() { let mut replication = MockReplicationCore::new(); replication .expect_prepare_batch_requests() - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let ctx = MockBuilder::new(shutdown_rx) .with_db_path("/tmp/test_lease_verification_empty_payload_hangs") @@ -2124,7 +2138,7 @@ async fn test_linearizable_read_served_without_quorum_in_minority_partition() { replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let mut node_config = RaftNodeConfig::default(); node_config.raft.read_consistency.allow_client_override = true; diff --git a/d-engine-core/src/raft_role/leader_state_test/client_write_test.rs b/d-engine-core/src/raft_role/leader_state_test/client_write_test.rs index 3baaf443..319b2619 100644 --- a/d-engine-core/src/raft_role/leader_state_test/client_write_test.rs +++ b/d-engine-core/src/raft_role/leader_state_test/client_write_test.rs @@ -92,7 +92,7 @@ async fn setup_process_raft_request_test_context( replication_handler .expect_prepare_batch_requests() .times(prepare_batch_requests_expect_times) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let last_entry_id = Arc::new(AtomicU64::new(4)); let last_entry_id_clone = last_entry_id.clone(); @@ -352,7 +352,7 @@ async fn test_process_raft_request_two_consecutive_forced_sends() { replication_handler .expect_prepare_batch_requests() .times(2) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); // last_entry_id increments per call: first=4 β†’ start_index=5, second=5 β†’ start_index=6 let call_count = Arc::new(AtomicU64::new(0)); @@ -548,7 +548,7 @@ async fn test_drain_single_write_no_delay() { replication .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let last_entry_id = Arc::new(AtomicU64::new(4)); let last_entry_id_clone = last_entry_id.clone(); @@ -647,7 +647,7 @@ async fn test_drain_multiple_writes_natural_batch() { replication .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let last_entry_id = Arc::new(AtomicU64::new(4)); let last_entry_id_clone = last_entry_id.clone(); @@ -825,7 +825,7 @@ async fn test_write_batch_single_replication() { replication .expect_prepare_batch_requests() .times(1) // KEY: Single replication for all writes - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let last_entry_id = Arc::new(AtomicU64::new(4)); let last_entry_id_clone = last_entry_id.clone(); @@ -953,7 +953,7 @@ async fn test_client_write_deferred_until_sm_apply() { replication .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let last_entry_id = Arc::new(AtomicU64::new(4)); let last_entry_id_clone = last_entry_id.clone(); @@ -1040,7 +1040,7 @@ async fn test_noop_quorum_check_responds_immediately_without_sm_apply() { replication .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let last_entry_id = Arc::new(AtomicU64::new(4)); let last_entry_id_clone = last_entry_id.clone(); @@ -1121,7 +1121,7 @@ async fn test_config_change_quorum_check_responds_immediately_without_sm_apply() replication .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let last_entry_id = Arc::new(AtomicU64::new(4)); let last_entry_id_clone = last_entry_id.clone(); @@ -1249,7 +1249,7 @@ async fn test_single_voter_client_write_completes_after_log_flushed() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let mut state = LeaderState::::new(1, ctx.node_config()); state.init_cluster_metadata(&ctx.membership).await.unwrap(); @@ -1380,7 +1380,7 @@ async fn test_probe_state_does_not_speculatively_advance_next_index() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(move |_, _, _, _, _| { + .returning(move |_, _, _, _, _, _| { Ok(PrepareResult { append_requests: [( 2, @@ -1472,7 +1472,7 @@ async fn test_replicate_state_advances_next_index_speculatively() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(move |_, _, _, _, _| { + .returning(move |_, _, _, _, _, _| { Ok(PrepareResult { append_requests: [( 2, diff --git a/d-engine-core/src/raft_role/leader_state_test/commit_index_test.rs b/d-engine-core/src/raft_role/leader_state_test/commit_index_test.rs index a7811973..33c127da 100644 --- a/d-engine-core/src/raft_role/leader_state_test/commit_index_test.rs +++ b/d-engine-core/src/raft_role/leader_state_test/commit_index_test.rs @@ -20,6 +20,7 @@ use d_engine_proto::server::replication::{AppendEntriesResponse, SuccessResult}; use rand::distr::SampleString; use std::collections::VecDeque; use std::sync::Arc; +use std::sync::atomic::{AtomicU64, Ordering}; use tokio::sync::{mpsc, watch}; fn success_response( @@ -38,6 +39,15 @@ fn success_response( } } +fn voter_meta(id: u32) -> NodeMeta { + NodeMeta { + id, + address: "".into(), + status: NodeStatus::Active as i32, + role: Follower.into(), + } +} + // ── helpers ────────────────────────────────────────────────────────────────── fn write_request() -> ( @@ -109,7 +119,7 @@ async fn test_single_voter_commit_must_not_advance_before_durable() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let (req, _rx) = write_request(); let (internal_event_tx, mut internal_event_rx) = mpsc::unbounded_channel(); @@ -154,7 +164,7 @@ async fn test_single_voter_commit_advances_after_durable() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let (req, _rx) = write_request(); let (internal_event_tx, mut internal_event_rx) = mpsc::unbounded_channel(); @@ -184,7 +194,7 @@ async fn test_single_voter_commit_advances_after_durable() { /// Multi-voter path: quorum calculation (via calculate_majority_matched_index) correctly /// blocks commit when leader's durable_index=0, even if last_entry_id=1. /// -/// This path was fixed in Step 2 (buffered_raft_log.rs). This test ensures the +/// This path was fixed in Step 2 (raft_log_core.rs). This test ensures the /// MockRaftLog-based quorum mock correctly gates the commit index. #[tokio::test] async fn test_multi_voter_commit_respects_quorum_result() { @@ -228,7 +238,7 @@ async fn test_multi_voter_commit_respects_quorum_result() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); ctx.handlers .replication_handler .expect_handle_success_response() @@ -262,6 +272,85 @@ async fn test_multi_voter_commit_respects_quorum_result() { ); } +// ── handle_append_result: quorum computed exactly once per ACK ─────────────── + +/// A multi-voter leader must compute the majority-matched index exactly once per +/// AppendEntries ACK. The pre-fix code called `calculate_majority_matched_index` +/// twice in a single `handle_append_result` β€” once inside `calculate_new_commit_index` +/// and once more for the `quorum_confirmed` lease check β€” with identical inputs +/// (`current_term`, `commit_index`, and the same voter-filtered `match_index`). +/// +/// The single result must drive both outcomes: commit_index advance AND lease +/// (quorum) confirmation. This test fails if the method runs twice, or if either +/// consumer stops receiving the result. +#[tokio::test] +async fn test_handle_append_result_computes_quorum_once_per_ack() { + let (_graceful_tx, graceful_rx) = watch::channel(()); + + let mut ctx = mock_raft_context( + "/tmp/test_handle_append_result_quorum_once", + graceful_rx, + None, + ); + + // Two follower voters β†’ multi-voter; `replication_peers` must list them so + // `handle_append_result` treats peer 2 as a voter (drives the quorum path). + let mut membership = crate::MockMembership::::new(); + membership.expect_voters().returning(|| vec![voter_meta(2), voter_meta(3)]); + membership + .expect_replication_peers() + .returning(|| vec![voter_meta(2), voter_meta(3)]); + ctx.membership = Arc::new(membership); + + // Count every quorum computation; return a majority index of 1 so the single + // result is exercised by BOTH the commit path and the lease-confirmation path. + let call_count = Arc::new(AtomicU64::new(0)); + let call_count_clone = call_count.clone(); + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 1); + raft_log.expect_calculate_majority_matched_index().returning(move |_, _, _| { + call_count_clone.fetch_add(1, Ordering::Relaxed); + Some(1) + }); + ctx.storage.raft_log = Arc::new(raft_log); + + let mut state = LeaderState::::new(1, ctx.node_config.clone()); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + assert!(!state.cluster_metadata.single_voter); + + ctx.handlers + .replication_handler + .expect_handle_success_response() + .returning(|_, _, _, _| { + Ok(PeerUpdate { + match_index: Some(1), + next_index: 2, + success: true, + }) + }); + + let (internal_event_tx, _internal_event_rx) = mpsc::unbounded_channel(); + let resp = success_response(1, 1); + state.handle_append_result(2, Ok(resp), &ctx, &internal_event_tx).await.unwrap(); + + // Root-cause assertion: one ACK must trigger exactly one majority computation. + assert_eq!( + call_count.load(Ordering::Relaxed), + 1, + "calculate_majority_matched_index must run exactly once per AppendEntries ACK" + ); + // Both consumers of that single result must still observe it. + assert_eq!( + state.commit_index(), + 1, + "commit must advance from the single majority result" + ); + assert!( + state.is_lease_valid(), + "quorum confirmation (lease) must reuse the single majority result" + ); +} + // ── handle_log_flushed: Leader commit advances on LogFlushed (#313 P0) ──────── /// Single-voter leader: LogFlushed(1) triggers commit_index to advance to 1. diff --git a/d-engine-core/src/raft_role/leader_state_test/deadline_test.rs b/d-engine-core/src/raft_role/leader_state_test/deadline_test.rs index fe09422f..18652029 100644 --- a/d-engine-core/src/raft_role/leader_state_test/deadline_test.rs +++ b/d-engine-core/src/raft_role/leader_state_test/deadline_test.rs @@ -65,6 +65,7 @@ fn write_metadata( senders: vec![tx], wait_for_apply: false, deadline, + proposed_at: std::time::Instant::now(), }; (meta, rx) } diff --git a/d-engine-core/src/raft_role/leader_state_test/event_handling_test.rs b/d-engine-core/src/raft_role/leader_state_test/event_handling_test.rs index 8bff40bb..a5da5f0c 100644 --- a/d-engine-core/src/raft_role/leader_state_test/event_handling_test.rs +++ b/d-engine-core/src/raft_role/leader_state_test/event_handling_test.rs @@ -524,7 +524,7 @@ async fn test_handle_client_propose_success() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let mut raft_log = MockRaftLog::new(); raft_log.expect_last_entry_id().returning(|| 4); context.storage.raft_log = Arc::new(raft_log); @@ -578,7 +578,7 @@ async fn test_handle_client_read_linearizable_failure() { replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Err(Error::Fatal("".to_string()))); + .returning(|_, _, _, _, _, _| Err(Error::Fatal("".to_string()))); // Initializing Shutdown Signal let (_graceful_tx, graceful_rx) = watch::channel(()); @@ -645,7 +645,7 @@ async fn test_handle_client_read_linearizable_success() { replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); // Mock state machine handler: read_from_state_machine called when ApplyCompleted fires. let mut state_machine_handler = MockStateMachineHandler::new(); @@ -768,7 +768,7 @@ async fn test_handle_client_read_encounters_higher_term() { replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); // Initializing Shutdown Signal let (_graceful_tx, graceful_rx) = watch::channel(()); @@ -914,7 +914,7 @@ async fn test_drain_read_buffer_clears_pending_reads_on_stepdown() { replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let (_graceful_tx, graceful_rx) = watch::channel(()); let mut node_config = RaftNodeConfig::default(); diff --git a/d-engine-core/src/raft_role/leader_state_test/heartbeat_scenarios_test.rs b/d-engine-core/src/raft_role/leader_state_test/heartbeat_scenarios_test.rs new file mode 100644 index 00000000..b7c22777 --- /dev/null +++ b/d-engine-core/src/raft_role/leader_state_test/heartbeat_scenarios_test.rs @@ -0,0 +1,1236 @@ +//! Test heartbeat scenarios when in-flight windows are full or empty. +//! +//! Background: the fix for #446 introduces a per-peer in-flight counter to prevent +//! the leader from repeatedly re-offering the same un-acknowledged entries. This creates +//! an architectural question: when the leader wants to send a genuine heartbeat (no new +//! data to replicate), but the peer's in-flight window is already full, what should happen? +//! +//! **Core principle**: heartbeats must never be blocked by window pressure. A heartbeat is +//! semantically an empty-entries request; an empty entries list occupies zero window slots. +//! Therefore: +//! - A *true* heartbeat (entries built by `prepare_batch_requests` are absent, so +//! `build_append_request` produces empty entries) must always be sendable. +//! - A *substitute* heartbeat (entries were withheld because the window is full, so +//! `prepare_batch_requests` did not populate `peer_entries` for this peer) is also safe +//! to send as empty entries, from the follower's point of view. +//! +//! The distinction matters for observability and correctness reasoning: we must verify +//! that even when a Replicate-state peer's window is at its limit (N un-acked requests), +//! the leader still emits a heartbeat that cycle (possibly empty) to keep the peer from +//! timing out and initiating an election. +//! +//! These tests verify the contract: +//! 1. True heartbeat always succeeds, even with a full window. +//! 2. Substitute heartbeat (window full) also gets sent, preserving liveness. +//! 3. Multiple in-flight requests for Replicate state can coexist (window > 1). +//! 4. Probe-state peers are still limited to exactly one in-flight request (window = 1). +//! 5. Receiving an ACK decrements exactly one in-flight slot, not all of them. + +use std::sync::Arc; + +use d_engine_proto::common::{NodeRole::Follower, NodeStatus}; +use d_engine_proto::server::cluster::NodeMeta; +use d_engine_proto::server::replication::{ + AppendEntriesResponse, ConflictResult, append_entries_response, +}; +use tokio::sync::{mpsc, watch}; +use tracing_test::traced_test; + +use crate::MockMembership; +use crate::MockRaftLog; +use crate::event::InternalEvent; +use crate::raft_role::leader_state::LeaderState; +use crate::raft_role::role_state::{PeerReplicationState, RaftRoleState}; +use crate::test_utils::MetricsCapture; +use crate::test_utils::mock::{MockTypeConfig, mock_raft_context}; + +fn two_peer_membership() -> MockMembership { + let peers = vec![ + NodeMeta { + id: 2, + address: String::new(), + status: NodeStatus::Active as i32, + role: Follower.into(), + }, + NodeMeta { + id: 3, + address: String::new(), + status: NodeStatus::Active as i32, + role: Follower.into(), + }, + ]; + let peers2 = peers.clone(); + let mut m = MockMembership::new(); + m.expect_is_single_node_cluster().returning(|| false); + m.expect_voters().returning(move || peers.clone()); + m.expect_replication_peers().returning(move || peers2.clone()); + m +} + +/// Verify that a peer in Replicate state can track multiple un-acknowledged requests +/// simultaneously (window > 1). This is the throughput-critical path: without multi-slot +/// windows, Replicate degrades to Probe's RTT-serialized behavior. +#[tokio::test] +#[traced_test] +async fn test_replicate_multiple_in_flight_requests_coexist() { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut ctx = mock_raft_context( + "/tmp/test_replicate_multiple_in_flight_coexist", + graceful_rx, + None, + ); + ctx.membership = Arc::new(two_peer_membership()); + + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 0); + raft_log.expect_flush().returning(|| Ok(())); + raft_log.expect_save_hard_state().returning(|_| Ok(())); + ctx.storage.raft_log = Arc::new(raft_log); + + let mut state = LeaderState::::new(1, ctx.node_config.clone()); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + + // Place peer 2 in Replicate state (not Probe). + state.set_peer_replication_state(2, PeerReplicationState::Replicate); + state.next_index.insert(2, 1); + + // Verify initial in-flight count is 0. + assert_eq!( + state.in_flight_count(2), + 0, + "Replicate-state peer must start with zero in-flight requests" + ); + + // Simulate dispatching multiple un-acknowledged requests. + state.record_in_flight(2, 10); + assert_eq!(state.in_flight_count(2), 1); + + state.record_in_flight(2, 20); + assert_eq!(state.in_flight_count(2), 2); + + state.record_in_flight(2, 30); + assert_eq!(state.in_flight_count(2), 3); + + // Contract: for a configurable window size (default 256), this should still be allowed. + // (Actual window gating happens in Phase 5 / prepare_batch_requests; this test verifies + // the counter itself accumulates correctly.) +} + +/// An ACK releases every outstanding request whose last entry is `<= match_index` (cumulative), +/// and nothing else. An older or repeated ACK frees nothing. +#[tokio::test] +#[traced_test] +async fn test_replicate_ack_releases_all_requests_up_to_its_index() { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut ctx = mock_raft_context( + "/tmp/test_replicate_ack_releases_one_slot", + graceful_rx, + None, + ); + ctx.membership = Arc::new(two_peer_membership()); + + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 0); + raft_log.expect_flush().returning(|| Ok(())); + raft_log.expect_save_hard_state().returning(|_| Ok(())); + ctx.storage.raft_log = Arc::new(raft_log); + + let mut state = LeaderState::::new(1, ctx.node_config.clone()); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + + state.set_peer_replication_state(2, PeerReplicationState::Replicate); + state.next_index.insert(2, 1); + + // 5 un-acknowledged requests, last entry indexes 10..=50. + for last_index in [10, 20, 30, 40, 50] { + state.record_in_flight(2, last_index); + } + assert_eq!(state.in_flight_count(2), 5, "Setup: 5 requests in-flight"); + + state.release_in_flight_up_to(2, 10); + assert_eq!( + state.in_flight_count(2), + 4, + "ACK 10 covers only the first request" + ); + + state.release_in_flight_up_to(2, 40); + assert_eq!( + state.in_flight_count(2), + 1, + "ACK 40 is cumulative: it covers the requests ending at 20, 30 and 40" + ); + + state.release_in_flight_up_to(2, 30); + assert_eq!(state.in_flight_count(2), 1, "an older ACK frees nothing"); + + state.release_in_flight_up_to(2, 40); + assert_eq!(state.in_flight_count(2), 1, "a repeated ACK frees nothing"); +} + +/// Verify that when a state transition occurs (e.g., Replicate β†’ Probe due to a conflict), +/// the in-flight counter is reset to 0. The old state's un-acknowledged requests are now +/// invalid (they were sent with a stale `next_index`), and the new state must start fresh. +#[tokio::test] +#[traced_test] +async fn test_replicate_to_probe_transition_resets_in_flight() { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut ctx = mock_raft_context("/tmp/test_replicate_to_probe_reset", graceful_rx, None); + ctx.membership = Arc::new(two_peer_membership()); + + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 0); + raft_log.expect_flush().returning(|| Ok(())); + raft_log.expect_save_hard_state().returning(|_| Ok(())); + ctx.storage.raft_log = Arc::new(raft_log); + + let mut state = LeaderState::::new(1, ctx.node_config.clone()); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + + state.set_peer_replication_state(2, PeerReplicationState::Replicate); + state.next_index.insert(2, 100); + + // Peer 2 has 10 un-acknowledged requests in Replicate state. + for i in 1..=10u64 { + state.record_in_flight(2, i * 10); + } + assert_eq!(state.in_flight_count(2), 10); + + // Conflict received: transition back to Probe, which resets all in-flight. + state.set_peer_replication_state(2, PeerReplicationState::Probe); + assert_eq!( + state.in_flight_count(2), + 0, + "State transition to Probe must reset in-flight counter to 0" + ); + assert_eq!( + state.peer_replication_state(2), + PeerReplicationState::Probe, + "Peer should now be in Probe state" + ); +} + +/// Verify that Probe state is still limited to at most one in-flight request (window = 1). +/// The window=1 constraint for Probe is enforced in Phase 5 (leader_state.rs dispatch loop); +/// this test verifies the in-flight counter itself can track Probe's semantics correctly +/// (one recorded request, released again by an ACK covering it). +#[tokio::test] +#[traced_test] +async fn test_probe_window_one_semantic() { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut ctx = mock_raft_context("/tmp/test_probe_window_one", graceful_rx, None); + ctx.membership = Arc::new(two_peer_membership()); + + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 0); + raft_log.expect_flush().returning(|| Ok(())); + raft_log.expect_save_hard_state().returning(|_| Ok(())); + ctx.storage.raft_log = Arc::new(raft_log); + + let mut state = LeaderState::::new(1, ctx.node_config.clone()); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + + state.set_peer_replication_state(2, PeerReplicationState::Probe); + state.next_index.insert(2, 1); + + // Probe starts with zero in-flight. + assert_eq!(state.in_flight_count(2), 0); + + // Dispatch the first (and only allowed) probe request. + state.record_in_flight(2, 10); + assert_eq!( + state.in_flight_count(2), + 1, + "Probe-state peer has exactly one in-flight request" + ); + + // Phase 5 should NOT dispatch another request until this one is ACK'd + // (that gate is in Phase 5; here we just verify the counter tracks 1). + + // Receive the ACK, releasing the slot. + state.release_in_flight_up_to(2, 10); + assert_eq!( + state.in_flight_count(2), + 0, + "Probe ACK releases the single slot" + ); +} + +/// Verify that heartbeats (empty entries) do not increment the in-flight counter. +/// This is a contract test: when Phase 5 detects `entries.is_empty()`, it must NOT +/// call `record_in_flight`. The window should only track *data* requests, not keepalive. +#[tokio::test] +#[traced_test] +async fn test_heartbeat_does_not_occupy_window() { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut ctx = mock_raft_context("/tmp/test_heartbeat_no_window", graceful_rx, None); + ctx.membership = Arc::new(two_peer_membership()); + + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 0); + raft_log.expect_flush().returning(|| Ok(())); + raft_log.expect_save_hard_state().returning(|_| Ok(())); + ctx.storage.raft_log = Arc::new(raft_log); + + let mut state = LeaderState::::new(1, ctx.node_config.clone()); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + + state.set_peer_replication_state(2, PeerReplicationState::Replicate); + state.next_index.insert(2, 100); + + // Simulate the window being full (256 un-acked requests). + // In reality, these would be actual data requests; we just set the counter. + for last_index in 1..=256u64 { + state.record_in_flight(2, last_index); + } + assert_eq!(state.in_flight_count(2), 256, "Window is now full"); + + // Phase 5 logic: if `is_heartbeat` (empty entries), do NOT check the window gate, + // and do NOT call `record_in_flight`. This test documents that heartbeat + // does not occupy a slot even when the window is full. + // The actual gate enforcement is in Phase 5; here we just verify that + // `record_in_flight` should NOT be called for heartbeats. + + assert_eq!( + state.in_flight_count(2), + 256, + "A heartbeat never recorded a request, so the window remains at 256 (full)" + ); +} + +/// Releasing on an empty window is a no-op, and the window is still usable afterwards. +#[tokio::test] +#[traced_test] +async fn test_release_on_empty_window_is_a_noop() { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut ctx = mock_raft_context("/tmp/test_release_on_empty_window", graceful_rx, None); + ctx.membership = Arc::new(two_peer_membership()); + + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 0); + raft_log.expect_flush().returning(|| Ok(())); + raft_log.expect_save_hard_state().returning(|_| Ok(())); + ctx.storage.raft_log = Arc::new(raft_log); + + let mut state = LeaderState::::new(1, ctx.node_config.clone()); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + + state.set_peer_replication_state(2, PeerReplicationState::Replicate); + state.next_index.insert(2, 1); + + assert_eq!(state.in_flight_count(2), 0); + + state.release_in_flight_up_to(2, 100); + assert_eq!( + state.in_flight_count(2), + 0, + "an ACK with nothing outstanding frees nothing" + ); + + state.record_in_flight(2, 101); + assert_eq!( + state.in_flight_count(2), + 1, + "the window is still usable afterwards" + ); +} + +/// Combines two contracts that must both hold at once for `Replicate` state (window > 1): +/// a stale-term response must not release *any* slot, and a real (term-matching) response +/// must release exactly the requests it covers β€” not all outstanding slots, and not zero. +/// +/// This matters specifically for Replicate because the failure mode is different from Probe's +/// (window=1): with multiple requests in flight, an over-eager release (e.g. accidentally +/// resetting instead of releasing by index) would silently let through a burst of new sends the +/// window was supposed to prevent, while an under-release would eventually starve the peer of +/// any new data once real ACKs stop being able to keep up. +/// +/// # Scenario +/// - Peer 2 in `Replicate` state, 3 un-acknowledged requests ending at indexes 3, 6 and 9 +/// (in_flight = 3). +/// - A stale-term response arrives (term 0 < leader_term 1) β€” must release nothing (in_flight +/// stays 3). +/// - A real, term-matching success response for index 3 arrives β€” must release exactly the +/// request ending at 3 (in_flight becomes 2), proving the earlier stale response didn't +/// already consume it. +#[tokio::test] +#[traced_test] +async fn test_replicate_stale_response_releases_nothing_real_ack_releases_covered_requests() { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut ctx = mock_raft_context( + "/tmp/test_replicate_stale_releases_nothing_real_releases_one", + graceful_rx, + None, + ); + ctx.membership = Arc::new(two_peer_membership()); + + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 3); + raft_log.expect_flush().returning(|| Ok(())); + raft_log.expect_save_hard_state().returning(|_| Ok(())); + raft_log.expect_calculate_majority_matched_index().returning(|_, _, _| Some(3)); + ctx.storage.raft_log = Arc::new(raft_log); + + ctx.handlers + .replication_handler + .expect_handle_success_response() + .returning(|_, _, _, _| { + Ok(crate::PeerUpdate { + match_index: Some(3), + next_index: 4, + success: true, + }) + }); + + let mut state = LeaderState::::new(1, ctx.node_config.clone()); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + state.set_peer_replication_state(2, PeerReplicationState::Replicate); + state.next_index.insert(2, 1); + + // Set up 3 outstanding requests, as Phase 5 dispatch would have left them. + state.record_in_flight(2, 3); + state.record_in_flight(2, 6); + state.record_in_flight(2, 9); + assert_eq!(state.in_flight_count(2), 3, "setup: 3 requests in flight"); + + let (internal_event_tx, _internal_event_rx) = mpsc::unbounded_channel::(); + + // Stale response: term 0 < leader_term 1 β€” must not touch any slot. + state + .handle_append_result( + 2, + Ok(AppendEntriesResponse { + node_id: 2, + term: 0, + result: None, + }), + &ctx, + &internal_event_tx, + ) + .await + .unwrap(); + assert_eq!( + state.in_flight_count(2), + 3, + "a stale-term response must release zero slots, not one and not all three" + ); + + // Real response: term matches β€” must release exactly one slot. + state + .handle_append_result( + 2, + Ok(AppendEntriesResponse { + node_id: 2, + term: 1, + result: Some(append_entries_response::Result::Success( + d_engine_proto::server::replication::SuccessResult { + last_match: Some(d_engine_proto::common::LogId { term: 1, index: 3 }), + }, + )), + }), + &ctx, + &internal_event_tx, + ) + .await + .unwrap(); + assert_eq!( + state.in_flight_count(2), + 2, + "the real response must release exactly one slot β€” the stale response above must not \ + have already consumed it, and this one must not release the other two" + ); +} + +/// `ConflictResult` differs from `SuccessResult` here: both are term-matching resolutions (so +/// neither is stale-ignored), but conflict additionally transitions the peer to `Probe` +/// (`update_peer_index`'s conflict branch β†’ `set_peer_replication_state(Probe)`), and that +/// transition unconditionally resets the whole in-flight window to 0 β€” not a partial release. This is intentional: a conflict means the leader's whole speculative pipeline +/// for this peer was built on a wrong assumption, so every other request still in flight is +/// also suspect, not just the one this response answers. Documented here so this isn't +/// mistaken for a bug symmetrical to the success case. +#[tokio::test] +#[traced_test] +async fn test_replicate_conflict_response_resets_all_inflight_via_probe_transition() { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut ctx = mock_raft_context( + "/tmp/test_replicate_conflict_releases_one_slot", + graceful_rx, + None, + ); + ctx.membership = Arc::new(two_peer_membership()); + + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 0); + raft_log.expect_flush().returning(|| Ok(())); + raft_log.expect_save_hard_state().returning(|_| Ok(())); + ctx.storage.raft_log = Arc::new(raft_log); + + ctx.handlers + .replication_handler + .expect_handle_conflict_response() + .returning(|_, _, _, _| { + Ok(crate::PeerUpdate { + match_index: None, + next_index: 1, + success: false, + }) + }); + + let mut state = LeaderState::::new(1, ctx.node_config.clone()); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + state.set_peer_replication_state(2, PeerReplicationState::Replicate); + state.next_index.insert(2, 1); + + state.record_in_flight(2, 3); + state.record_in_flight(2, 6); + assert_eq!(state.in_flight_count(2), 2, "setup: 2 requests in flight"); + + let (internal_event_tx, _internal_event_rx) = mpsc::unbounded_channel::(); + + state + .handle_append_result( + 2, + Ok(AppendEntriesResponse { + node_id: 2, + term: 1, + result: Some(append_entries_response::Result::Conflict(ConflictResult { + conflict_term: None, + conflict_index: Some(1), + })), + }), + &ctx, + &internal_event_tx, + ) + .await + .unwrap(); + assert_eq!( + state.in_flight_count(2), + 0, + "conflict must reset in-flight to 0 via the Probe transition, not decrement by one β€” \ + the whole speculative pipeline for this peer is now suspect, not just this request" + ); + assert_eq!( + state.peer_replication_state(2), + PeerReplicationState::Probe, + "conflict must move the peer to Probe" + ); +} + +/// `handle_peer_stream_error` (the consumer of a failed/full worker send) must demote a +/// `Replicate` peer to `Probe`, drop every outstanding in-flight slot, and rewind `next_index` +/// to `match_index + 1` so the next probe re-sends whatever the follower never received. +/// +/// # Scenario +/// - Peer 2 in `Replicate`, `match_index = 5`, speculative `next_index = 10`, 3 requests in flight. +/// - `handle_peer_stream_error(2)`. +/// - Expected: `Probe`, in_flight 0, `next_index = 6`. +#[tokio::test] +#[traced_test] +async fn test_peer_stream_error_demotes_replicate_peer_and_clears_inflight() { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut ctx = mock_raft_context( + "/tmp/test_peer_stream_error_demotes_replicate_peer", + graceful_rx, + None, + ); + ctx.membership = Arc::new(two_peer_membership()); + + let mut state = LeaderState::::new(1, ctx.node_config.clone()); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + + state.set_peer_replication_state(2, PeerReplicationState::Replicate); + state.match_index.insert(2, 5); + state.next_index.insert(2, 10); + state.record_in_flight(2, 7); + state.record_in_flight(2, 8); + state.record_in_flight(2, 9); + assert_eq!(state.in_flight_count(2), 3, "setup: 3 requests in flight"); + + state.handle_peer_stream_error(2); + + assert_eq!(state.peer_replication_state(2), PeerReplicationState::Probe); + assert_eq!( + state.in_flight_count(2), + 0, + "every outstanding slot is suspect once the send path failed" + ); + assert_eq!( + state.next_index.get(&2).copied(), + Some(6), + "next_index must rewind to match_index + 1 so unACKed entries are re-sent" + ); +} + +/// Pins a deliberate design decision in `set_peer_replication_state`: the "only reset on a +/// transition" check compares against `peer_replication_state.insert`'s return value (was this +/// peer ever explicitly recorded before?), not against the peer's implicit state (`Probe` by +/// default when never recorded). This means the *first* explicit call for a given peer always +/// resets in-flight, even if the state being set matches what the implicit default already was. +/// +/// # Why this is the chosen behavior, not a bug +/// Owner's call: correctness must not depend on tracking "was this the peer's first-ever +/// explicit state write" separately from the `HashMap` itself β€” `insert`'s return value already +/// tells us that for free. The cost is this one corner case (first write happens to match the +/// implicit default) triggers a reset that, by pure state-equality, wasn't strictly necessary β€” +/// but every current call site only reaches this in ways that are harmless (a peer's first +/// explicit state write coincides with it having at most one outstanding request, since +/// multiple in-flight requires already being in `Replicate`, which requires an explicit +/// insert). Accepting this corner case keeps the implementation simple and avoids a second, +/// separate notion of "current state" to keep in sync with the `HashMap`. +/// +/// This test exists so that if a future refactor "simplifies" this into comparing against +/// `peer_replication_state()`'s implicit-default-aware accessor instead, it fails loudly instead +/// of silently changing behavior. +#[tokio::test] +#[traced_test] +async fn test_first_explicit_state_write_resets_inflight_even_if_state_unchanged() { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut ctx = mock_raft_context( + "/tmp/test_first_explicit_state_write_resets_inflight", + graceful_rx, + None, + ); + ctx.membership = Arc::new(two_peer_membership()); + + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 0); + raft_log.expect_flush().returning(|| Ok(())); + raft_log.expect_save_hard_state().returning(|_| Ok(())); + ctx.storage.raft_log = Arc::new(raft_log); + + let mut state = LeaderState::::new(1, ctx.node_config.clone()); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + + // Peer 2 has never had set_peer_replication_state called for it β€” peer_replication_state + // (the HashMap) has no entry, so its state is only implicitly Probe via the default in + // peer_replication_state()'s accessor. + assert_eq!( + state.peer_replication_state(2), + PeerReplicationState::Probe, + "setup: peer 2's state must be the implicit default, not an explicitly recorded one" + ); + + // Simulate a request already dispatched to this never-explicitly-recorded peer (this is + // exactly what Phase 5 does for a peer's first-ever probe β€” record_in_flight does not + // require set_peer_replication_state to have been called first). + state.record_in_flight(2, 1); + assert_eq!(state.in_flight_count(2), 1, "setup: one request in flight"); + + // The first-ever explicit call for peer 2, setting it to Probe β€” the same value its + // implicit default already was. Per the chosen design, this still resets in-flight, because + // the check is "was this HashMap entry ever written before", not "did the state value + // change". + state.set_peer_replication_state(2, PeerReplicationState::Probe); + + assert_eq!( + state.in_flight_count(2), + 0, + "the first explicit write for a peer always resets in-flight, by design β€” the check is \ + against HashMap::insert's return value (was there a prior entry?), not against the \ + peer's implicit state" + ); +} + +/// Boundary behavior of the window gate, which Phase 2 and Phase 5 both consume: one below the +/// limit is open, at the limit is closed, and a released slot reopens it. +/// +/// # Scenario +/// - `max_inflight_append_requests = 3`. `Replicate`: 0/2 in flight open, 3 closed, back to 2 open. +/// - `Probe` is fixed at a window of 1 regardless of the configured value. +#[tokio::test] +#[traced_test] +async fn test_window_gate_boundaries_for_replicate_and_probe() { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut ctx = mock_raft_context("/tmp/test_window_gate_boundaries", graceful_rx, None); + ctx.membership = Arc::new(two_peer_membership()); + + let mut cfg = (*ctx.node_config).clone(); + cfg.raft.replication.max_inflight_append_requests = 3; + let mut state = LeaderState::::new(1, Arc::new(cfg)); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + + state.set_peer_replication_state(2, PeerReplicationState::Replicate); + assert!(!state.should_gate_by_inflight(2), "0 in flight: open"); + state.record_in_flight(2, 10); + state.record_in_flight(2, 20); + assert!(!state.should_gate_by_inflight(2), "2 of 3 in flight: open"); + state.record_in_flight(2, 30); + assert!(state.should_gate_by_inflight(2), "3 of 3 in flight: closed"); + state.release_in_flight_up_to(2, 10); + assert!( + !state.should_gate_by_inflight(2), + "a released slot reopens the gate" + ); + + assert!( + !state.should_gate_by_inflight(3), + "Probe with 0 in flight: open" + ); + state.record_in_flight(3, 10); + assert!( + state.should_gate_by_inflight(3), + "Probe is fixed at window 1, regardless of the configured Replicate window" + ); +} + +// ── Gauge emission tests ────────────────────────────────────────────────────── + +/// `core.raft.peer.in_flight` must reflect the current count immediately after +/// each increment and after a decrement. +#[tokio::test] +#[traced_test] +async fn test_in_flight_gauge_emitted() { + let capture = MetricsCapture::new(); + // set_default_local_recorder stays active until guard is dropped; works on + // current_thread tokio tests where there is no thread switching across awaits. + let _guard = metrics::set_default_local_recorder(&capture); + + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut ctx = mock_raft_context("/tmp/test_in_flight_gauge_emitted", graceful_rx, None); + ctx.membership = Arc::new(two_peer_membership()); + + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 0); + raft_log.expect_flush().returning(|| Ok(())); + raft_log.expect_save_hard_state().returning(|_| Ok(())); + ctx.storage.raft_log = Arc::new(raft_log); + + let mut state = LeaderState::::new(1, ctx.node_config.clone()); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + state.set_peer_replication_state(2, PeerReplicationState::Replicate); + + state.record_in_flight(2, 10); + state.record_in_flight(2, 20); + assert_eq!( + capture.gauge("core.raft.peer.in_flight", &[("peer_id", "2")]), + Some(2.0), + "gauge must be 2 after two dispatches" + ); + + state.release_in_flight_up_to(2, 10); + assert_eq!( + capture.gauge("core.raft.peer.in_flight", &[("peer_id", "2")]), + Some(1.0), + "gauge must drop to 1 after an ACK covering the first request" + ); +} + +/// `core.raft.peer.match_index` must be emitted when `update_peer_index` advances +/// the match index via a success response. +#[tokio::test] +#[traced_test] +async fn test_match_index_gauge_emitted_on_success_ack() { + let capture = MetricsCapture::new(); + let _guard = metrics::set_default_local_recorder(&capture); + + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut ctx = mock_raft_context("/tmp/test_match_index_gauge_emitted", graceful_rx, None); + ctx.membership = Arc::new(two_peer_membership()); + + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 5); + raft_log.expect_flush().returning(|| Ok(())); + raft_log.expect_save_hard_state().returning(|_| Ok(())); + raft_log.expect_calculate_majority_matched_index().returning(|_, _, _| Some(5)); + ctx.storage.raft_log = Arc::new(raft_log); + + ctx.handlers + .replication_handler + .expect_handle_success_response() + .returning(|_, _, _, _| { + Ok(crate::PeerUpdate { + match_index: Some(5), + next_index: 6, + success: true, + }) + }); + + let mut state = LeaderState::::new(1, ctx.node_config.clone()); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + state.set_peer_replication_state(2, PeerReplicationState::Replicate); + state.next_index.insert(2, 1); + state.record_in_flight(2, 5); + + let (internal_event_tx, _rx) = mpsc::unbounded_channel::(); + state + .handle_append_result( + 2, + Ok(AppendEntriesResponse { + node_id: 2, + term: 1, + result: Some(append_entries_response::Result::Success( + d_engine_proto::server::replication::SuccessResult { + last_match: Some(d_engine_proto::common::LogId { term: 1, index: 5 }), + }, + )), + }), + &ctx, + &internal_event_tx, + ) + .await + .unwrap(); + + assert_eq!( + capture.gauge("core.raft.peer.match_index", &[("peer_id", "2")]), + Some(5.0), + "match_index gauge must be set to 5 after a success ACK for index=5" + ); +} + +/// `core.raft.peer.last_ack_timestamp_seconds` must be set (non-zero) after a +/// term-matching response, and must NOT be set for a stale-term response. +#[tokio::test] +#[traced_test] +async fn test_last_ack_timestamp_gauge_emitted_only_on_non_stale_ack() { + let capture = MetricsCapture::new(); + let _guard = metrics::set_default_local_recorder(&capture); + + { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut ctx = mock_raft_context("/tmp/test_last_ack_ts_gauge_emitted", graceful_rx, None); + ctx.membership = Arc::new(two_peer_membership()); + + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 0); + raft_log.expect_flush().returning(|| Ok(())); + raft_log.expect_save_hard_state().returning(|_| Ok(())); + raft_log.expect_calculate_majority_matched_index().returning(|_, _, _| Some(3)); + ctx.storage.raft_log = Arc::new(raft_log); + + ctx.handlers.replication_handler.expect_handle_success_response().returning( + |_, _, _, _| { + Ok(crate::PeerUpdate { + match_index: Some(3), + next_index: 4, + success: true, + }) + }, + ); + + let mut state = LeaderState::::new(1, ctx.node_config.clone()); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + state.set_peer_replication_state(2, PeerReplicationState::Replicate); + state.next_index.insert(2, 1); + state.record_in_flight(2, 3); + + let (internal_event_tx, _rx) = mpsc::unbounded_channel::(); + + // Stale response (term 0 < leader term 1): gauge must NOT be written yet. + state + .handle_append_result( + 2, + Ok(AppendEntriesResponse { + node_id: 2, + term: 0, + result: None, + }), + &ctx, + &internal_event_tx, + ) + .await + .unwrap(); + assert_eq!( + capture.gauge( + "core.raft.peer.last_ack_timestamp_seconds", + &[("peer_id", "2")] + ), + None, + "stale-term response must not set the timestamp gauge" + ); + + // Real response (term 1 = leader term): gauge must be written as a positive unix timestamp. + state + .handle_append_result( + 2, + Ok(AppendEntriesResponse { + node_id: 2, + term: 1, + result: Some(append_entries_response::Result::Success( + d_engine_proto::server::replication::SuccessResult { + last_match: Some(d_engine_proto::common::LogId { term: 1, index: 3 }), + }, + )), + }), + &ctx, + &internal_event_tx, + ) + .await + .unwrap(); + + let ts = capture + .gauge( + "core.raft.peer.last_ack_timestamp_seconds", + &[("peer_id", "2")], + ) + .expect("timestamp gauge must be set after a real ACK"); + assert!( + ts > 1_700_000_000.0, + "timestamp must be a plausible unix epoch value, got {ts}" + ); + } +} + +/// A stream error loses every request still queued on that stream, so no response will ever +/// come back for them. The window must be cleared even when the peer is *already* Probe +/// (the Probe->Probe write is a no-op for `set_peer_replication_state`, so it cannot be +/// relied on to do the reset). +#[tokio::test] +#[traced_test] +async fn test_stream_error_clears_inflight_when_peer_is_already_probe() { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut ctx = mock_raft_context("/tmp/test_stream_error_probe_probe", graceful_rx, None); + ctx.membership = Arc::new(two_peer_membership()); + + let mut state = LeaderState::::new(1, ctx.node_config.clone()); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + + state.set_peer_replication_state(2, PeerReplicationState::Probe); + state.record_in_flight(2, 1); + assert_eq!( + state.in_flight_count(2), + 1, + "setup: one probe request outstanding" + ); + + state.handle_peer_stream_error(2); + + assert_eq!( + state.in_flight_count(2), + 0, + "the outstanding request died with the stream; its slot must be released" + ); +} + +/// A success response reporting an index below every outstanding request (what a heartbeat +/// reply carries) must not release any data slot; a later response covering them must. +#[tokio::test] +#[traced_test] +async fn test_success_response_below_outstanding_requests_releases_nothing() { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut ctx = mock_raft_context("/tmp/test_success_below_outstanding", graceful_rx, None); + ctx.membership = Arc::new(two_peer_membership()); + + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 20); + raft_log.expect_flush().returning(|| Ok(())); + raft_log.expect_save_hard_state().returning(|_| Ok(())); + raft_log.expect_calculate_majority_matched_index().returning(|_, _, _| Some(5)); + ctx.storage.raft_log = Arc::new(raft_log); + + ctx.handlers.replication_handler.expect_handle_success_response().returning( + |_, _, success, _| { + let match_index = success.last_match.as_ref().map(|l| l.index); + Ok(crate::PeerUpdate { + match_index, + next_index: match_index.unwrap_or(0) + 1, + success: true, + }) + }, + ); + + let mut state = LeaderState::::new(1, ctx.node_config.clone()); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + state.set_peer_replication_state(2, PeerReplicationState::Replicate); + state.next_index.insert(2, 1); + state.record_in_flight(2, 10); + state.record_in_flight(2, 20); + + let (internal_event_tx, _rx) = mpsc::unbounded_channel::(); + let success_at = |index: u64| AppendEntriesResponse { + node_id: 2, + term: 1, + result: Some(append_entries_response::Result::Success( + d_engine_proto::server::replication::SuccessResult { + last_match: Some(d_engine_proto::common::LogId { term: 1, index }), + }, + )), + }; + + state + .handle_append_result(2, Ok(success_at(5)), &ctx, &internal_event_tx) + .await + .unwrap(); + assert_eq!( + state.in_flight_count(2), + 2, + "a reply at index 5 confirms nothing the two outstanding requests (10, 20) carry" + ); + + state + .handle_append_result(2, Ok(success_at(20)), &ctx, &internal_event_tx) + .await + .unwrap(); + assert_eq!( + state.in_flight_count(2), + 0, + "a reply at index 20 covers both requests" + ); +} + +/// A conflict rejects the request. A `Probe` peer stays `Probe`, so the state-change reset does +/// not fire: `update_peer_index` must clear the window itself or the gate stays closed forever. +#[tokio::test] +#[traced_test] +async fn test_probe_conflict_response_clears_window() { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut ctx = mock_raft_context("/tmp/test_probe_conflict_clears_window", graceful_rx, None); + ctx.membership = Arc::new(two_peer_membership()); + + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 0); + raft_log.expect_flush().returning(|| Ok(())); + raft_log.expect_save_hard_state().returning(|_| Ok(())); + ctx.storage.raft_log = Arc::new(raft_log); + + ctx.handlers + .replication_handler + .expect_handle_conflict_response() + .returning(|_, _, _, _| { + Ok(crate::PeerUpdate { + match_index: None, + next_index: 1, + success: false, + }) + }); + + let mut state = LeaderState::::new(1, ctx.node_config.clone()); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + state.set_peer_replication_state(2, PeerReplicationState::Probe); + state.next_index.insert(2, 1); + state.record_in_flight(2, 1); + assert!( + state.should_gate_by_inflight(2), + "setup: the Probe window (1) is full" + ); + + let (internal_event_tx, _rx) = mpsc::unbounded_channel::(); + state + .handle_append_result( + 2, + Ok(AppendEntriesResponse { + node_id: 2, + term: 1, + result: Some(append_entries_response::Result::Conflict(ConflictResult { + conflict_term: None, + conflict_index: Some(1), + })), + }), + &ctx, + &internal_event_tx, + ) + .await + .unwrap(); + + assert_eq!( + state.in_flight_count(2), + 0, + "the rejected probe no longer occupies the window" + ); + assert!( + !state.should_gate_by_inflight(2), + "the next probe may be sent" + ); +} + +/// Builds a leader with peer 2 in `Replicate` and three requests in flight (last indexes 3, 6, 9), +/// feeds it one `handle_append_result` input, and returns the window afterwards. +/// +/// If `conflict_handler_fails` is set, the conflict handler returns an error instead of a +/// `PeerUpdate`, as a future change to it might. +async fn window_after_response( + response: crate::Result, + conflict_handler_fails: bool, +) -> (usize, PeerReplicationState) { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut ctx = mock_raft_context("/tmp/test_window_after_response", graceful_rx, None); + ctx.membership = Arc::new(two_peer_membership()); + + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 9); + raft_log.expect_flush().returning(|| Ok(())); + raft_log.expect_save_hard_state().returning(|_| Ok(())); + raft_log.expect_calculate_majority_matched_index().returning(|_, _, _| Some(1)); + ctx.storage.raft_log = Arc::new(raft_log); + + ctx.handlers.replication_handler.expect_handle_success_response().returning( + |_, _, success, _| { + let match_index = success.last_match.as_ref().map(|l| l.index); + Ok(crate::PeerUpdate { + match_index, + next_index: match_index.unwrap_or(0) + 1, + success: true, + }) + }, + ); + ctx.handlers.replication_handler.expect_handle_conflict_response().returning( + move |_, _, _, _| { + if conflict_handler_fails { + Err(crate::ReplicationError::HigherTerm(9).into()) + } else { + Ok(crate::PeerUpdate { + match_index: None, + next_index: 1, + success: false, + }) + } + }, + ); + + let mut state = LeaderState::::new(1, ctx.node_config.clone()); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + state.set_peer_replication_state(2, PeerReplicationState::Replicate); + state.next_index.insert(2, 10); + for last_index in [3, 6, 9] { + state.record_in_flight(2, last_index); + } + + let (internal_event_tx, _rx) = mpsc::unbounded_channel::(); + let _ = state.handle_append_result(2, response, &ctx, &internal_event_tx).await; + (state.in_flight_count(2), state.peer_replication_state(2)) +} + +fn response_with( + term: u64, + result: Option, +) -> crate::Result { + Ok(AppendEntriesResponse { + node_id: 2, + term, + result, + }) +} + +fn success_at(index: u64) -> Option { + Some(append_entries_response::Result::Success( + d_engine_proto::server::replication::SuccessResult { + last_match: Some(d_engine_proto::common::LogId { term: 1, index }), + }, + )) +} + +/// Every shape of response `handle_append_result` can receive, and what it must do to the window. +/// A new early return that skips the window handling shows up here as a changed row. +/// The leader term is 1; the window starts as [3, 6, 9]. +#[tokio::test] +#[traced_test] +async fn test_every_response_shape_has_a_defined_window_effect() { + use PeerReplicationState::{Probe, Replicate}; + let conflict = Some(append_entries_response::Result::Conflict(ConflictResult { + conflict_term: None, + conflict_index: Some(1), + })); + let closed = Err(crate::Error::System(crate::SystemError::Network( + crate::NetworkError::ResponseChannelClosed, + ))); + + let cases: Vec<( + &str, + crate::Result, + usize, + PeerReplicationState, + )> = vec![ + ( + "success covering one request", + response_with(1, success_at(3)), + 2, + Replicate, + ), + ( + "success covering two requests", + response_with(1, success_at(6)), + 1, + Replicate, + ), + ( + "success covering all requests", + response_with(1, success_at(9)), + 0, + Replicate, + ), + ( + "success below every request (heartbeat-like)", + response_with(1, success_at(1)), + 3, + Replicate, + ), + ("conflict", response_with(1, conflict), 0, Probe), + ("no result variant", response_with(1, None), 0, Replicate), + ( + "HigherTerm variant that is not higher (late rejection)", + response_with(1, Some(append_entries_response::Result::HigherTerm(1))), + 3, + Replicate, + ), + ("stale term", response_with(0, success_at(9)), 3, Replicate), + ("superseded response channel", closed, 0, Replicate), + ]; + + for (name, response, expected_count, expected_state) in cases { + let (count, state) = window_after_response(response, false).await; + assert_eq!(count, expected_count, "window after: {name}"); + assert_eq!(state, expected_state, "peer state after: {name}"); + } +} + +/// The conflict handler failing must not leave the rejected pipeline counted as in flight: +/// a stuck window means the peer never receives data again. +#[tokio::test] +#[traced_test] +async fn test_conflict_handler_error_does_not_leave_the_window_stuck() { + let conflict = Some(append_entries_response::Result::Conflict(ConflictResult { + conflict_term: None, + conflict_index: Some(1), + })); + let (count, _) = window_after_response(response_with(1, conflict), true).await; + assert_eq!( + count, 0, + "an error while handling a conflict must still clear the window" + ); +} + +/// A follower can report an index beyond the leader's own log (for example a divergent tail +/// left by an earlier leader). The leader must never record a `match_index` past its own +/// last entry. +#[tokio::test] +#[traced_test] +async fn test_match_index_never_exceeds_the_leaders_own_last_entry() { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut ctx = mock_raft_context( + "/tmp/test_match_index_bounded_by_leader_log", + graceful_rx, + None, + ); + ctx.membership = Arc::new(two_peer_membership()); + + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 20); + raft_log.expect_flush().returning(|| Ok(())); + raft_log.expect_save_hard_state().returning(|_| Ok(())); + raft_log.expect_calculate_majority_matched_index().returning(|_, _, _| Some(1)); + ctx.storage.raft_log = Arc::new(raft_log); + + ctx.handlers.replication_handler.expect_handle_success_response().returning( + |_, _, success, _| { + let match_index = success.last_match.as_ref().map(|l| l.index); + Ok(crate::PeerUpdate { + match_index, + next_index: match_index.unwrap_or(0) + 1, + success: true, + }) + }, + ); + + let mut state = LeaderState::::new(1, ctx.node_config.clone()); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + state.set_peer_replication_state(2, PeerReplicationState::Replicate); + + let (internal_event_tx, _rx) = mpsc::unbounded_channel::(); + let response = response_with(1, success_at(50)); + state.handle_append_result(2, response, &ctx, &internal_event_tx).await.unwrap(); + + assert!( + state.match_index_for_test(2) <= 20, + "the leader's log ends at 20, but peer 2's match_index was recorded as {}", + state.match_index_for_test(2) + ); +} diff --git a/d-engine-core/src/raft_role/leader_state_test/membership_change_test.rs b/d-engine-core/src/raft_role/leader_state_test/membership_change_test.rs index ada6c1cd..f371633f 100644 --- a/d-engine-core/src/raft_role/leader_state_test/membership_change_test.rs +++ b/d-engine-core/src/raft_role/leader_state_test/membership_change_test.rs @@ -103,7 +103,7 @@ async fn test_join_cluster_precondition_checks() { .handlers .replication_handler .expect_prepare_batch_requests() - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let mut state = LeaderState::::new(1, raft_context.node_config.clone()); @@ -192,7 +192,7 @@ async fn test_join_cluster_creates_correct_config_change() { .handlers .replication_handler .expect_prepare_batch_requests() - .returning(move |payloads, _, _, _, _| { + .returning(move |payloads, _, _, _, _, _| { // Capture payloads for validation captured_clone.lock().extend(payloads.clone()); // Return empty to let test proceed without waiting for commit @@ -324,7 +324,7 @@ async fn test_join_cluster_triggers_verification() { .replication_handler .expect_prepare_batch_requests() .times(1) // Must be called exactly once - .returning(|_, _, _, _, _| { + .returning(|_, _, _, _, _, _| { // Verification triggered successfully Ok(crate::PrepareResult::default()) }); @@ -459,7 +459,7 @@ async fn test_handle_join_cluster_quorum_failed() { .replication_handler .expect_prepare_batch_requests() .times(..) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let mut state = LeaderState::::new(1, context.node_config.clone()); @@ -560,7 +560,7 @@ async fn test_handle_join_cluster_quorum_error() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Err(Error::Fatal("Simulated quorum error".to_string()))); + .returning(|_, _, _, _, _, _| Err(Error::Fatal("Simulated quorum error".to_string()))); let mut state = LeaderState::::new(1, context.node_config.clone()); @@ -730,7 +730,7 @@ mod stale_learner_tests { // Mock replication handler to capture payloads ctx.handlers.replication_handler.expect_prepare_batch_requests().returning( - move |payloads, _, _, _, _| { + move |payloads, _, _, _, _, _| { captured_clone.lock().extend(payloads.clone()); Ok(crate::PrepareResult::default()) // Return empty, test focuses on config creation }, @@ -1186,7 +1186,7 @@ mod pending_promotion_tests { let capture_clone = captured_payloads.clone(); let mut replication_handler = MockReplicationCore::::new(); replication_handler.expect_prepare_batch_requests().times(..).returning( - move |payloads, _, _, _, _| { + move |payloads, _, _, _, _, _| { capture_clone.lock().extend(payloads.clone()); Ok(crate::PrepareResult::default()) }, @@ -1336,7 +1336,7 @@ mod pending_promotion_tests { .handlers .replication_handler .expect_prepare_batch_requests() - .returning(move |payloads, _, _, _, _| { + .returning(move |payloads, _, _, _, _, _| { captured_clone.lock().extend(payloads.clone()); Ok(crate::PrepareResult::default()) // Return empty, test focuses on config creation }); @@ -1613,7 +1613,7 @@ mod zombie_purge_tests { .replication_handler .expect_prepare_batch_requests() .times(0) // Must NOT be called β€” zombie detection is warn-only - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let mut raft_log = MockRaftLog::new(); raft_log.expect_last_entry_id().returning(|| 10); @@ -1652,7 +1652,7 @@ mod zombie_purge_tests { .replication_handler .expect_prepare_batch_requests() .times(0) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let mut raft_log = MockRaftLog::new(); raft_log.expect_last_entry_id().returning(|| 10); @@ -1694,7 +1694,7 @@ mod zombie_purge_tests { .replication_handler .expect_prepare_batch_requests() .times(0) // Must NOT be called β€” node is already removed - .returning(move |payloads, _, _, _, _| { + .returning(move |payloads, _, _, _, _, _| { captured_clone.lock().extend(payloads.clone()); Ok(crate::PrepareResult::default()) }); diff --git a/d-engine-core/src/raft_role/leader_state_test/pending_lease_reads_test.rs b/d-engine-core/src/raft_role/leader_state_test/pending_lease_reads_test.rs index fe924282..75ae9a9c 100644 --- a/d-engine-core/src/raft_role/leader_state_test/pending_lease_reads_test.rs +++ b/d-engine-core/src/raft_role/leader_state_test/pending_lease_reads_test.rs @@ -89,7 +89,7 @@ async fn setup_multi_voter_expired_lease( let mut replication_handler = MockReplicationCore::new(); replication_handler .expect_prepare_batch_requests() - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let mut raft_log = MockRaftLog::new(); raft_log.expect_last_entry_id().returning(|| 10); @@ -261,7 +261,7 @@ async fn test_single_voter_lease_read_served_immediately_on_expired_lease() { let mut replication_handler = MockReplicationCore::new(); replication_handler .expect_prepare_batch_requests() - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let mut raft_log = MockRaftLog::new(); raft_log.expect_last_entry_id().returning(|| 5); diff --git a/d-engine-core/src/raft_role/leader_state_test/pending_reads_test.rs b/d-engine-core/src/raft_role/leader_state_test/pending_reads_test.rs index 689f6e45..d0075b78 100644 --- a/d-engine-core/src/raft_role/leader_state_test/pending_reads_test.rs +++ b/d-engine-core/src/raft_role/leader_state_test/pending_reads_test.rs @@ -206,7 +206,7 @@ async fn test_multi_voter_fast_path_linear_read_is_queued() { replication .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let (mut state, context, internal_event_tx, _internal_event_rx) = setup_multi_voter( "/tmp/test_multi_voter_fast_path_linear_read_is_queued", @@ -572,7 +572,7 @@ async fn test_linearizable_read_served_immediately_with_valid_lease_in_multi_vot replication .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let (mut state, context, internal_event_tx, _internal_event_rx) = setup_multi_voter( "/tmp/test_linearizable_read_served_immediately_with_valid_lease_in_multi_voter", @@ -644,7 +644,7 @@ async fn test_linearizable_read_queued_when_sm_behind_despite_valid_lease() { replication .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let (mut state, context, internal_event_tx, _internal_event_rx) = setup_multi_voter( "/tmp/test_linearizable_read_queued_when_sm_behind_despite_valid_lease", @@ -714,7 +714,7 @@ async fn test_linearizable_read_not_served_when_lease_expired_in_multi_voter() { replication .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let (mut state, context, internal_event_tx, _internal_event_rx) = setup_multi_voter( "/tmp/test_linearizable_read_not_served_when_lease_expired_in_multi_voter", diff --git a/d-engine-core/src/raft_role/leader_state_test/probe_backpressure_test.rs b/d-engine-core/src/raft_role/leader_state_test/probe_backpressure_test.rs new file mode 100644 index 00000000..5b61051b --- /dev/null +++ b/d-engine-core/src/raft_role/leader_state_test/probe_backpressure_test.rs @@ -0,0 +1,1040 @@ +//! Test for the missing per-peer in-flight gate during `PeerReplicationState::Probe`. +//! +//! Background (see `446-expert-q-probe-backpressure-fix-8020.md` in the product-design repo): +//! every reference Raft implementation limits a `Probe`-state +//! peer to at most one outstanding (unacknowledged) `AppendEntries` request. d-engine's +//! `PeerReplicationState::Probe`/`Replicate` only controls whether `next_index` is optimistically +//! advanced before sending β€” it never checks whether the peer already has a request in flight. +//! Left unchecked, the leader keeps re-sending `prev_log_index=0` probes to a peer whose first +//! attempt hasn't been acknowledged yet, and each one forces the follower to wipe and rebuild its +//! entire log (`RaftLogCore::reset`), which is the root cause of the throughput collapse +//! this ticket investigated. +//! +//! This test is intentionally RED until the in-flight gate is implemented in +//! `execute_and_process_raft_rpc` (Phase 5, `leader_state.rs`). It does not assert *how* the gate +//! is implemented β€” only the externally observable contract: a peer with one unacknowledged +//! request must not receive a second one. + +use std::collections::VecDeque; +use std::sync::Arc; + +use bytes::Bytes; +use d_engine_proto::common::{Entry, EntryPayload, NodeRole::Follower, NodeStatus}; +use d_engine_proto::server::cluster::NodeMeta; +use d_engine_proto::server::replication::{ + AppendEntriesRequest, AppendEntriesResponse, ConflictResult, append_entries_response, +}; +use tokio::sync::{mpsc, watch}; +use tracing_test::traced_test; + +use crate::MockMembership; +use crate::MockRaftLog; +use crate::RaftRequestWithSignal; +use crate::event::InternalEvent; +use crate::maybe_clone_oneshot::{MaybeCloneOneshot, RaftOneshot}; +use crate::network::PeerUpdate; +use crate::raft_role::leader_state::LeaderState; +use crate::raft_role::role_state::{PeerReplicationState, RaftRoleState}; +use crate::test_utils::MetricsCapture; +use crate::test_utils::mock::{MockTypeConfig, mock_raft_context}; + +/// Two-voter membership (peers 2 & 3) so the cluster is multi-voter β€” a single-voter leader +/// short-circuits Phase 5 entirely (no peer work to gate), which would make this test vacuous. +fn two_peer_membership() -> MockMembership { + let peers = vec![ + NodeMeta { + id: 2, + address: String::new(), + status: NodeStatus::Active as i32, + role: Follower.into(), + }, + NodeMeta { + id: 3, + address: String::new(), + status: NodeStatus::Active as i32, + role: Follower.into(), + }, + ]; + let peers2 = peers.clone(); + let mut m = MockMembership::new(); + m.expect_is_single_node_cluster().returning(|| false); + m.expect_voters().returning(move || peers.clone()); + m.expect_replication_peers().returning(move || peers2.clone()); + m +} + +/// A minimal `AppendEntriesRequest` stub β€” its contents don't matter, only whether Phase 5 +/// forwards a request to the peer's worker channel at all. +fn stub_request() -> AppendEntriesRequest { + AppendEntriesRequest::default() +} + +/// A non-empty probe (one entry) β€” used where the empty-heartbeat vs non-empty-probe +/// distinction matters for the gate: a heartbeat must neither block nor set the gate. +fn stub_probe_request() -> AppendEntriesRequest { + AppendEntriesRequest { + entries: vec![Entry { + index: 1, + term: 1, + payload: None, + }], + ..AppendEntriesRequest::default() + } +} + +/// A single one-entry write batch, matching the shape `process_batch` expects. +fn one_entry_batch() -> VecDeque { + let (tx, _rx) = >::new(); + let req = RaftRequestWithSignal { + id: "test".into(), + payloads: vec![EntryPayload::command(Bytes::from_static(b"cmd"))], + senders: vec![tx], + wait_for_apply_event: false, + }; + VecDeque::from(vec![req]) +} + +/// Scenario (release direction β€” the half the gate must also get right): +/// - Batch 1 is dispatched to peer 2: its first `Probe`, correctly limited to one outstanding +/// request. +/// - The follower answers that probe with a CONFLICT (reject), not a success. A reject is not +/// evidence the peer is caught up, so `update_peer_index`'s conflict branch retreats +/// `next_index` to the conflict hint and leaves peer 2 in `Probe`. +/// - Batch 2 is processed afterwards. The response to batch 1 has *arrived*, so peer 2 no longer +/// has an outstanding request and the gate must release: the corrected probe must be +/// dispatched. +/// +/// # Expected +/// `Probe` means "at most one unacknowledged `AppendEntries` at a time", not "at most one ever". A response must +/// re-arm the gate, never latch it shut: a latched `Probe` peer receives nothing further β€” not +/// even heartbeats, since they share this dispatch path β€” while the frozen follower times out +/// into candidacy and the leader cannot reach quorum. +/// +/// # Current behavior (why this test is RED) +/// The gate is set on dispatch and only cleared by `handle_peer_stream_error` (a bidi stream +/// disconnect). A peer whose probe was rejected stays `Probe` with the latch closed forever, so +/// batch 2 dispatches nothing. +#[tokio::test] +#[traced_test] +async fn test_probe_peer_dispatches_next_probe_after_reject() { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut ctx = mock_raft_context( + "/tmp/test_probe_peer_dispatches_next_probe_after_reject", + graceful_rx, + None, + ); + + ctx.membership = Arc::new(two_peer_membership()); + + // Both batches offer a request for peer 2, so every dispatch decision is Phase 5's own. + ctx.handlers + .replication_handler + .expect_prepare_batch_requests() + .times(2) + .returning(|_, _, _, _, _, _| { + Ok(crate::PrepareResult { + append_requests: vec![(2, stub_probe_request(), 1)], + snapshot_targets: vec![], + }) + }); + ctx.handlers + .replication_handler + .expect_handle_conflict_response() + .returning(|_, _, _, _| { + Ok(PeerUpdate { + match_index: None, + next_index: 1, + success: false, + }) + }); + + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 0); + raft_log.expect_flush().returning(|| Ok(())); + raft_log.expect_save_hard_state().returning(|_| Ok(())); + ctx.storage.raft_log = Arc::new(raft_log); + + let mut state = LeaderState::::new(1, ctx.node_config.clone()); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + + let (task_tx, mut task_rx) = mpsc::unbounded_channel(); + state.replication_workers.insert( + 2, + super::ReplicationWorkerHandle { + task_tx, + snapshot_failure_count: 0, + snapshot_next_retry_at: None, + }, + ); + + let (internal_event_tx, _internal_event_rx) = mpsc::unbounded_channel::(); + + // Batch 1: peer 2's first probe β€” nothing outstanding, so it must be dispatched. + state.process_batch(one_entry_batch(), &internal_event_tx, &ctx).await.unwrap(); + assert!( + task_rx.try_recv().is_ok(), + "first batch must reach peer 2's worker β€” it had no outstanding request" + ); + + // Peer 2 rejects that probe: the leader retreats next_index and keeps the peer in `Probe`, + // because a reject says nothing about the peer being caught up. + let reject = AppendEntriesResponse { + node_id: 2, + term: 1, + result: Some(append_entries_response::Result::Conflict(ConflictResult { + conflict_term: None, + conflict_index: Some(1), + })), + }; + state + .handle_append_result(2, Ok(reject), &ctx, &internal_event_tx) + .await + .unwrap(); + assert_eq!( + state.peer_replication_state(2), + PeerReplicationState::Probe, + "a rejected probe must leave the peer in `Probe` β€” it is not caught up" + ); + + // Batch 2: the response to batch 1 already arrived, so the gate must release and the + // corrected probe must go out. + state.process_batch(one_entry_batch(), &internal_event_tx, &ctx).await.unwrap(); + assert!( + task_rx.try_recv().is_ok(), + "the probe was answered (reject), so the peer has nothing in flight β€” the leader must \ + re-probe with the corrected next_index instead of latching the peer shut forever" + ); +} + +/// Scenario: +/// - Peer 2 has a worker whose channel this test holds directly (no real transport/network +/// involved), so what actually got dispatched can be checked synchronously β€” no dependency on +/// background task scheduling, so this test cannot flake on timing. +/// - `prepare_batch_requests` is mocked to unconditionally offer a request for peer 2 on every +/// call, simulating "there's always more to replicate" regardless of ack status β€” this isolates +/// the assertion to Phase 5's own dispatch decision, which is where the missing gate belongs. +/// - Batch 1 is processed and dispatched β€” this is correct: peer 2 starts in `Probe` with nothing +/// outstanding, so it must receive its first probe. +/// - Batch 2 is processed *without* `handle_append_result` ever being called for peer 2's first +/// request β€” i.e. the leader has not (and cannot have) learned whether the first probe was +/// acknowledged. `next_index`/`match_index`/`peer_replication_state` are therefore still exactly +/// what they were after batch 1. +/// +/// # Expected (once the fix lands) +/// Batch 2 must NOT produce a second dispatch to peer 2's worker: a `Probe`-state peer with an +/// unacknowledged request in flight must wait for that response before being sent to again. +/// +/// # Current behavior (why this test is RED today) +/// `execute_and_process_raft_rpc`'s Phase 5 loop sends to every peer in `append_requests` +/// unconditionally β€” `PeerReplicationState` only gates whether `next_index` is optimistically +/// advanced beforehand, not whether sending is allowed at all. So batch 2 dispatches anyway. +#[tokio::test] +#[traced_test] +async fn test_probe_peer_with_pending_ack_receives_no_second_dispatch() { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut ctx = mock_raft_context( + "/tmp/test_probe_peer_with_pending_ack_receives_no_second_dispatch", + graceful_rx, + None, + ); + + ctx.membership = Arc::new(two_peer_membership()); + + ctx.handlers + .replication_handler + .expect_prepare_batch_requests() + .times(2) + .returning(|_, _, _, _, _, _| { + Ok(crate::PrepareResult { + append_requests: vec![(2, stub_probe_request(), 1)], + snapshot_targets: vec![], + }) + }); + + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 0); + raft_log.expect_flush().returning(|| Ok(())); + raft_log.expect_save_hard_state().returning(|_| Ok(())); + ctx.storage.raft_log = Arc::new(raft_log); + + let mut state = LeaderState::::new(1, ctx.node_config.clone()); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + + // Inject peer 2's worker handle directly and keep the channel's receiver in this test β€” this + // is what makes the dispatch count observable synchronously, without any async worker task or + // transport mock (`send_to_worker_or_spawn` finds this handle and reuses it, so the real + // worker-spawn path β€” the only place that would touch `ctx.transport` β€” is never exercised). + // `ReplicationWorkerHandle`/`ReplicationTask` are private to `leader_state`, visible here only + // because this test module nests under it β€” same access pattern `inject_dead_worker_for_test` + // already relies on for the sibling `worker_lifecycle_test.rs` file. + let (task_tx, mut task_rx) = mpsc::unbounded_channel(); + state.replication_workers.insert( + 2, + super::ReplicationWorkerHandle { + task_tx, + snapshot_failure_count: 0, + snapshot_next_retry_at: None, + }, + ); + + let (internal_event_tx, _internal_event_rx) = mpsc::unbounded_channel::(); + + // Batch 1: peer 2 starts in `Probe` with nothing in flight β€” must be dispatched. + state.process_batch(one_entry_batch(), &internal_event_tx, &ctx).await.unwrap(); + assert!( + task_rx.try_recv().is_ok(), + "first batch must reach peer 2's worker β€” it had no outstanding request" + ); + + // Batch 2: peer 2's first request has not been acknowledged (handle_append_result was never + // called), so it is still awaiting a response. + state.process_batch(one_entry_batch(), &internal_event_tx, &ctx).await.unwrap(); + assert!( + task_rx.try_recv().is_err(), + "peer 2 already has an unacknowledged request in flight β€” a Probe-state peer must not \ + receive a second AppendEntries until the first is acked. See \ + 446-expert-q-probe-backpressure-fix-8020.md for the etcd/raft and openraft references." + ); +} + +/// Shared setup for the gate open/close tests below: a two-peer cluster with peer 2's worker +/// channel handed to the test, so dispatch is observable synchronously without a real transport. +async fn setup_gate_harness( + path: &str +) -> ( + crate::raft_context::RaftContext, + LeaderState, + mpsc::UnboundedReceiver, + mpsc::UnboundedSender, +) { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut ctx = mock_raft_context(path, graceful_rx, None); + ctx.membership = Arc::new(two_peer_membership()); + + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 0); + raft_log.expect_flush().returning(|| Ok(())); + raft_log.expect_save_hard_state().returning(|_| Ok(())); + ctx.storage.raft_log = Arc::new(raft_log); + + let mut state = LeaderState::::new(1, ctx.node_config.clone()); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + + let (task_tx, task_rx) = mpsc::unbounded_channel(); + state.replication_workers.insert( + 2, + super::ReplicationWorkerHandle { + task_tx, + snapshot_failure_count: 0, + snapshot_next_retry_at: None, + }, + ); + + let (internal_event_tx, _internal_event_rx) = mpsc::unbounded_channel::(); + (ctx, state, task_rx, internal_event_tx) +} + +/// An unparseable response (no `result` variant) must still reopen the gate: the request is no +/// longer in flight, even though the leader learned nothing usable from it. Latching here would +/// freeze the peer exactly like the reject case. +#[tokio::test] +#[traced_test] +async fn test_probe_peer_dispatches_next_probe_after_unparseable_response() { + let (mut ctx, mut state, mut task_rx, internal_event_tx) = + setup_gate_harness("/tmp/test_probe_peer_dispatches_next_probe_after_unparseable_response") + .await; + + ctx.handlers + .replication_handler + .expect_prepare_batch_requests() + .times(2) + .returning(|_, _, _, _, _, _| { + Ok(crate::PrepareResult { + append_requests: vec![(2, stub_probe_request(), 1)], + snapshot_targets: vec![], + }) + }); + + state.process_batch(one_entry_batch(), &internal_event_tx, &ctx).await.unwrap(); + assert!(task_rx.try_recv().is_ok(), "first probe must be dispatched"); + + state + .handle_append_result( + 2, + Ok(AppendEntriesResponse { + node_id: 2, + term: 1, + result: None, + }), + &ctx, + &internal_event_tx, + ) + .await + .unwrap(); + + state.process_batch(one_entry_batch(), &internal_event_tx, &ctx).await.unwrap(); + assert!( + task_rx.try_recv().is_ok(), + "an unparseable response still resolves the outstanding probe β€” the leader must re-probe" + ); +} + +/// An empty (heartbeat) dispatch must NOT set the gate, otherwise a heartbeat would occupy the +/// "probe in flight" slot and block the next real probe. +#[tokio::test] +#[traced_test] +async fn test_heartbeat_does_not_latch_gate() { + let (mut ctx, mut state, mut task_rx, internal_event_tx) = + setup_gate_harness("/tmp/test_heartbeat_does_not_latch_gate").await; + + let calls = Arc::new(std::sync::atomic::AtomicUsize::new(0)); + ctx.handlers + .replication_handler + .expect_prepare_batch_requests() + .times(2) + .returning(move |_, _, _, _, _, _| { + let n = calls.fetch_add(1, std::sync::atomic::Ordering::SeqCst); + let req = if n == 0 { + stub_request() + } else { + stub_probe_request() + }; + Ok(crate::PrepareResult { + append_requests: vec![(2, req, 1)], + snapshot_targets: vec![], + }) + }); + + // Batch 1 is an empty heartbeat β€” it must be dispatched but leave the gate open. + state.process_batch(one_entry_batch(), &internal_event_tx, &ctx).await.unwrap(); + assert!(task_rx.try_recv().is_ok(), "heartbeat must be dispatched"); + + // Batch 2 is a real probe β€” it must not be blocked by a latch the heartbeat never set. + state.process_batch(one_entry_batch(), &internal_event_tx, &ctx).await.unwrap(); + assert!( + task_rx.try_recv().is_ok(), + "an empty heartbeat must not latch the gate β€” the following probe must go out" + ); +} + +/// While a non-empty probe is in flight, an empty heartbeat must still be dispatched: the gate +/// throttles probes only, never heartbeats β€” this is the liveness backstop that unfreezes a peer +/// whose probe response was lost or unparseable. +#[tokio::test] +#[traced_test] +async fn test_heartbeat_bypasses_gate_while_probe_in_flight() { + let (mut ctx, mut state, mut task_rx, internal_event_tx) = + setup_gate_harness("/tmp/test_heartbeat_bypasses_gate_while_probe_in_flight").await; + + let calls = Arc::new(std::sync::atomic::AtomicUsize::new(0)); + ctx.handlers + .replication_handler + .expect_prepare_batch_requests() + .times(2) + .returning(move |_, _, _, _, _, _| { + let n = calls.fetch_add(1, std::sync::atomic::Ordering::SeqCst); + let req = if n == 0 { + stub_probe_request() + } else { + stub_request() + }; + Ok(crate::PrepareResult { + append_requests: vec![(2, req, 1)], + snapshot_targets: vec![], + }) + }); + + // Batch 1 is a non-empty probe β€” it latches the gate. + state.process_batch(one_entry_batch(), &internal_event_tx, &ctx).await.unwrap(); + assert!(task_rx.try_recv().is_ok(), "first probe must be dispatched"); + + // Batch 2 is an empty heartbeat β€” it must bypass the gate and still be dispatched. + state.process_batch(one_entry_batch(), &internal_event_tx, &ctx).await.unwrap(); + assert!( + task_rx.try_recv().is_ok(), + "a heartbeat must bypass the probe gate β€” it is the liveness backstop, not a probe" + ); +} + +/// A stale-term response belongs to an older request, not the outstanding probe, so it must NOT +/// reopen the gate. Liveness is instead recovered by the heartbeat backstop (see the bypass test), +/// mirroring etcd's `MaybeUpdate(n <= Match)` early-return-without-resume. +#[tokio::test] +#[traced_test] +async fn test_stale_term_response_keeps_gate_latched() { + let (mut ctx, mut state, mut task_rx, internal_event_tx) = + setup_gate_harness("/tmp/test_stale_term_response_keeps_gate_latched").await; + + ctx.handlers + .replication_handler + .expect_prepare_batch_requests() + .times(2) + .returning(|_, _, _, _, _, _| { + Ok(crate::PrepareResult { + append_requests: vec![(2, stub_probe_request(), 1)], + snapshot_targets: vec![], + }) + }); + + state.process_batch(one_entry_batch(), &internal_event_tx, &ctx).await.unwrap(); + assert!(task_rx.try_recv().is_ok(), "first probe must be dispatched"); + + // term 0 < leader_term 1 β†’ stale, ignored without clearing the gate. + state + .handle_append_result( + 2, + Ok(AppendEntriesResponse { + node_id: 2, + term: 0, + result: None, + }), + &ctx, + &internal_event_tx, + ) + .await + .unwrap(); + + state.process_batch(one_entry_batch(), &internal_event_tx, &ctx).await.unwrap(); + assert!( + task_rx.try_recv().is_err(), + "a stale-term response must not reopen the gate β€” the outstanding probe is still in flight" + ); +} + +/// Completes the story `test_stale_term_response_keeps_gate_latched` deliberately stops short +/// of: a stale response must not release the slot, but the *real* (term-matching) response for +/// the same outstanding request must still release it once it arrives. Without this, "stale +/// responses don't release" could be (mis)implemented as "nothing ever releases". +/// +/// # Scenario +/// - Batch 1 dispatched to peer 2 β€” gate latches (Probe, window=1). +/// - A stale-term response arrives (term 0 < leader_term 1) β€” ignored, gate stays latched +/// (same setup as the sibling test above). +/// - The *real* response for that outstanding probe arrives (term 1, a reject) β€” this is the +/// response Phase 5 actually sent the probe to elicit, so it must release the slot. +/// - Batch 2 must now dispatch β€” proves the slot was released by the real response, not +/// permanently stuck after the stale one was ignored. +#[tokio::test] +#[traced_test] +async fn test_real_response_after_stale_still_releases_the_gate() { + let (mut ctx, mut state, mut task_rx, internal_event_tx) = + setup_gate_harness("/tmp/test_real_response_after_stale_still_releases_the_gate").await; + + ctx.handlers + .replication_handler + .expect_prepare_batch_requests() + .times(2) + .returning(|_, _, _, _, _, _| { + Ok(crate::PrepareResult { + append_requests: vec![(2, stub_probe_request(), 1)], + snapshot_targets: vec![], + }) + }); + ctx.handlers + .replication_handler + .expect_handle_conflict_response() + .returning(|_, _, _, _| { + Ok(PeerUpdate { + match_index: None, + next_index: 1, + success: false, + }) + }); + + state.process_batch(one_entry_batch(), &internal_event_tx, &ctx).await.unwrap(); + assert!(task_rx.try_recv().is_ok(), "first probe must be dispatched"); + assert_eq!( + state.in_flight_count(2), + 1, + "setup: probe dispatch must occupy the one Probe-window slot" + ); + + // Stale response: term 0 < leader_term 1 β€” must not release the slot. + state + .handle_append_result( + 2, + Ok(AppendEntriesResponse { + node_id: 2, + term: 0, + result: None, + }), + &ctx, + &internal_event_tx, + ) + .await + .unwrap(); + assert_eq!( + state.in_flight_count(2), + 1, + "a stale-term response must not release the slot" + ); + + // The real response to the original probe: term matches, a reject (conflict) β€” this must + // release the slot regardless of accept/reject, because it genuinely resolves the request + // Phase 5 was tracking. + state + .handle_append_result( + 2, + Ok(AppendEntriesResponse { + node_id: 2, + term: 1, + result: Some(append_entries_response::Result::Conflict(ConflictResult { + conflict_term: None, + conflict_index: Some(1), + })), + }), + &ctx, + &internal_event_tx, + ) + .await + .unwrap(); + assert_eq!( + state.in_flight_count(2), + 0, + "the real (term-matching) response must release the slot the stale one could not" + ); + + state.process_batch(one_entry_batch(), &internal_event_tx, &ctx).await.unwrap(); + assert!( + task_rx.try_recv().is_ok(), + "the slot was released by the real response, so the corrected probe must go out" + ); +} + +/// `record_in_flight` must never fire for a heartbeat dispatch, in `Replicate` state exactly +/// as much as in `Probe` state β€” the "heartbeats occupy zero window slots" contract is +/// state-independent. Without this, repeated heartbeats to an idle Replicate-state peer would +/// silently fill its window (default 256) and eventually start gating genuine data sends, even +/// though nothing was ever un-acknowledged. +/// +/// # Scenario +/// - Peer 2 in `Replicate` state (window = configured `max_inflight_append_requests`). +/// - Ten consecutive heartbeat batches (empty entries) are dispatched. +/// - `in_flight_count` must remain 0 throughout β€” heartbeats never occupied a slot. +#[tokio::test] +#[traced_test] +async fn test_heartbeat_in_replicate_state_does_not_occupy_window() { + let (mut ctx, mut state, mut task_rx, internal_event_tx) = + setup_gate_harness("/tmp/test_heartbeat_in_replicate_state_does_not_occupy_window").await; + + state.set_peer_replication_state(2, PeerReplicationState::Replicate); + state.next_index.insert(2, 1); + + ctx.handlers + .replication_handler + .expect_prepare_batch_requests() + .times(10) + .returning(|_, _, _, _, _, _| { + Ok(crate::PrepareResult { + // Empty entries: build_append_request's natural heartbeat fallback shape. + append_requests: vec![(2, stub_request(), 1)], + snapshot_targets: vec![], + }) + }); + + for i in 0..10 { + state.process_batch(one_entry_batch(), &internal_event_tx, &ctx).await.unwrap(); + assert!( + task_rx.try_recv().is_ok(), + "heartbeat #{i} must still be dispatched" + ); + assert_eq!( + state.in_flight_count(2), + 0, + "heartbeat #{i} must not occupy a window slot in Replicate state" + ); + } +} + +/// Ledger invariant for any "dispatched but lost" path (stream torn down, send buffer full, +/// or a future drop-on-unreachable): Phase 5 writes `next_index` and `in_flight` *before* the +/// request leaves the leader, so recovering from a lost request must rewind both. +/// +/// # Scenario +/// - Peer 2 in `Replicate`, `match_index = 5`. One data request (1 entry, effective next 6) is +/// dispatched: `next_index` advances optimistically to 7 and `in_flight` becomes 1. +/// - The request is lost: `handle_peer_stream_error(2)`. +/// - Expected: `Probe`, `in_flight = 0`, `next_index = match_index + 1 = 6`, so the same range is +/// offered again instead of being skipped. +#[tokio::test] +#[traced_test] +async fn test_lost_dispatch_rewinds_next_index_and_in_flight() { + let (mut ctx, mut state, mut task_rx, internal_event_tx) = + setup_gate_harness("/tmp/test_lost_dispatch_rewinds_next_index_and_in_flight").await; + + ctx.handlers + .replication_handler + .expect_prepare_batch_requests() + .times(1) + .returning(|_, _, _, _, _, _| { + Ok(crate::PrepareResult { + append_requests: vec![(2, stub_probe_request(), 6)], + snapshot_targets: vec![], + }) + }); + + state.set_peer_replication_state(2, PeerReplicationState::Replicate); + state.match_index.insert(2, 5); + state.next_index.insert(2, 6); + + state.process_batch(one_entry_batch(), &internal_event_tx, &ctx).await.unwrap(); + assert!( + task_rx.try_recv().is_ok(), + "the data request must be dispatched" + ); + assert_eq!( + state.next_index.get(&2).copied(), + Some(7), + "setup: Replicate advances next_index optimistically at dispatch" + ); + assert_eq!( + state.in_flight_count(2), + 1, + "setup: dispatch occupies one slot" + ); + + state.handle_peer_stream_error(2); + + assert_eq!(state.peer_replication_state(2), PeerReplicationState::Probe); + assert_eq!( + state.in_flight_count(2), + 0, + "the lost request's slot must be released, or the gate stays closed" + ); + assert_eq!( + state.next_index.get(&2).copied(), + Some(6), + "next_index must rewind to match_index + 1 so the lost range is offered again" + ); +} + +/// Two-peer harness: both peers' worker channels are held by the test, so each peer's dispatch is +/// observable independently. +async fn setup_two_worker_harness( + path: &str +) -> ( + crate::raft_context::RaftContext, + LeaderState, + mpsc::UnboundedReceiver, + mpsc::UnboundedReceiver, + mpsc::UnboundedSender, +) { + let (ctx, mut state, task_rx2, internal_event_tx) = setup_gate_harness(path).await; + let (task_tx3, task_rx3) = mpsc::unbounded_channel(); + state.replication_workers.insert( + 3, + super::ReplicationWorkerHandle { + task_tx: task_tx3, + snapshot_failure_count: 0, + snapshot_next_retry_at: None, + }, + ); + (ctx, state, task_rx2, task_rx3, internal_event_tx) +} + +/// One peer's closed window must not block another peer's dispatch. +/// +/// # Scenario +/// - Peer 2 is `Probe` (window 1) with its one probe unanswered; peer 3 is `Replicate`. +/// - Every batch offers a data request to both peers. +/// - Expected: batch 1 reaches both; batch 2 is withheld from peer 2 (window full) but still +/// reaches peer 3. +#[tokio::test] +#[traced_test] +async fn test_gated_peer_does_not_block_other_peer() { + let (mut ctx, mut state, mut rx2, mut rx3, internal_event_tx) = + setup_two_worker_harness("/tmp/test_gated_peer_does_not_block_other_peer").await; + + ctx.handlers + .replication_handler + .expect_prepare_batch_requests() + .times(2) + .returning(|_, _, _, _, _, _| { + Ok(crate::PrepareResult { + append_requests: vec![(2, stub_probe_request(), 1), (3, stub_probe_request(), 1)], + snapshot_targets: vec![], + }) + }); + + state.set_peer_replication_state(3, PeerReplicationState::Replicate); + + state.process_batch(one_entry_batch(), &internal_event_tx, &ctx).await.unwrap(); + assert!(rx2.try_recv().is_ok(), "batch 1 must reach peer 2"); + assert!(rx3.try_recv().is_ok(), "batch 1 must reach peer 3"); + + state.process_batch(one_entry_batch(), &internal_event_tx, &ctx).await.unwrap(); + assert!( + rx2.try_recv().is_err(), + "peer 2's window is full, batch 2 must be withheld from it" + ); + assert!( + rx3.try_recv().is_ok(), + "peer 2 being gated must not stop peer 3 from receiving batch 2" + ); +} + +/// A stream failure on one peer must leave the other peer's ledger untouched. +/// +/// # Scenario +/// - Peer 2 and peer 3 are both `Replicate` with a request in flight; `next_index` recorded. +/// - `handle_peer_stream_error(2)`. +/// - Expected: peer 2 is demoted and cleared; peer 3 keeps its state, in-flight count and +/// `next_index`. +#[tokio::test] +#[traced_test] +async fn test_stream_error_on_one_peer_leaves_other_peer_untouched() { + let (_ctx, mut state, _rx2, _rx3, _tx) = + setup_two_worker_harness("/tmp/test_stream_error_on_one_peer_leaves_other_untouched").await; + + for peer in [2_u32, 3] { + state.set_peer_replication_state(peer, PeerReplicationState::Replicate); + state.match_index.insert(peer, 5); + state.next_index.insert(peer, 9); + state.record_in_flight(peer, 6); + } + + state.handle_peer_stream_error(2); + + assert_eq!(state.peer_replication_state(2), PeerReplicationState::Probe); + assert_eq!(state.in_flight_count(2), 0); + assert_eq!( + state.peer_replication_state(3), + PeerReplicationState::Replicate, + "peer 3 must not be demoted by peer 2's failure" + ); + assert_eq!( + state.in_flight_count(3), + 1, + "peer 3's in-flight slot must survive" + ); + assert_eq!( + state.next_index.get(&3).copied(), + Some(9), + "peer 3's next_index must not be rewound" + ); +} + +/// End-to-end window behavior in `Replicate`: several requests may be in flight without any ACK +/// (not serialized to one), and the dispatch that would exceed the window is withheld. +/// +/// # Scenario +/// - `max_inflight_append_requests = 3`, peer 2 in `Replicate`, every batch offers a data request. +/// - Batches 1..3 are all dispatched with no response in between (in flight 1, 2, 3). +/// - Batch 4 would be the fourth outstanding request: it must be withheld, in flight stays 3. +#[tokio::test] +#[traced_test] +async fn test_replicate_dispatches_up_to_window_then_withholds() { + let (mut ctx, mut state, mut task_rx, internal_event_tx) = + setup_gate_harness("/tmp/test_replicate_dispatches_up_to_window_then_withholds").await; + + let mut cfg = (*ctx.node_config).clone(); + cfg.raft.replication.max_inflight_append_requests = 3; + state.node_config = Arc::new(cfg); + + ctx.handlers + .replication_handler + .expect_prepare_batch_requests() + .times(4) + .returning(|_, _, _, _, _, _| { + Ok(crate::PrepareResult { + append_requests: vec![(2, stub_probe_request(), 1)], + snapshot_targets: vec![], + }) + }); + + state.set_peer_replication_state(2, PeerReplicationState::Replicate); + state.next_index.insert(2, 1); + + for expected in 1..=3usize { + state.process_batch(one_entry_batch(), &internal_event_tx, &ctx).await.unwrap(); + assert!( + task_rx.try_recv().is_ok(), + "request {expected} must be dispatched without waiting for an ACK" + ); + assert_eq!(state.in_flight_count(2), expected); + } + + state.process_batch(one_entry_batch(), &internal_event_tx, &ctx).await.unwrap(); + assert!( + task_rx.try_recv().is_err(), + "the fourth outstanding request exceeds the window and must be withheld" + ); + assert_eq!( + state.in_flight_count(2), + 3, + "a withheld request must not occupy a slot" + ); +} + +/// Window=3, four dispatch attempts: the first three are recorded (occupancy before each +/// dispatch is 0,1,2, one entry each), the fourth is withheld and is counted as gated but +/// must not add a histogram sample (it never dispatched). +#[tokio::test] +#[traced_test] +async fn test_dispatch_metrics_record_occupancy_entries_and_gated_count() { + let capture = MetricsCapture::new(); + let _guard = metrics::set_default_local_recorder(&capture); + + let (mut ctx, mut state, mut task_rx, internal_event_tx) = + setup_gate_harness("/tmp/test_dispatch_metrics_record_occupancy").await; + + let mut cfg = (*ctx.node_config).clone(); + cfg.raft.replication.max_inflight_append_requests = 3; + state.node_config = Arc::new(cfg); + + ctx.handlers + .replication_handler + .expect_prepare_batch_requests() + .times(4) + .returning(|_, _, _, _, _, _| { + Ok(crate::PrepareResult { + append_requests: vec![(2, stub_probe_request(), 1)], + snapshot_targets: vec![], + }) + }); + + state.set_peer_replication_state(2, PeerReplicationState::Replicate); + state.next_index.insert(2, 1); + + for _ in 0..4 { + state.process_batch(one_entry_batch(), &internal_event_tx, &ctx).await.unwrap(); + } + assert!(task_rx.try_recv().is_ok(), "sanity: dispatches happened"); + + assert_eq!( + capture.histogram("core.raft.peer.in_flight_at_dispatch", &[("peer_id", "2")]), + vec![0.0, 1.0, 2.0], + "occupancy is sampled before each of the 3 real dispatches; the withheld one adds nothing" + ); + assert_eq!( + capture.histogram("core.raft.replication.entries_per_request", &[]), + vec![1.0, 1.0, 1.0], + "one sample per real dispatch, equal to the request's entry count" + ); + assert_eq!( + capture.counter("core.raft.peer.dispatch_gated_total", &[("peer_id", "2")]), + 1, + "exactly the fourth attempt was withheld by the window" + ); +} + +/// A heartbeat (empty entries) occupies no window slot, so it must neither be counted as +/// gated when the window is full nor contribute to the dispatch histograms. +#[tokio::test] +#[traced_test] +async fn test_heartbeat_emits_no_dispatch_metrics_even_when_window_full() { + let capture = MetricsCapture::new(); + let _guard = metrics::set_default_local_recorder(&capture); + + let (mut ctx, mut state, _task_rx, internal_event_tx) = + setup_gate_harness("/tmp/test_heartbeat_emits_no_dispatch_metrics").await; + + let mut cfg = (*ctx.node_config).clone(); + cfg.raft.replication.max_inflight_append_requests = 1; + state.node_config = Arc::new(cfg); + + ctx.handlers + .replication_handler + .expect_prepare_batch_requests() + .times(1) + .returning(|_, _, _, _, _, _| { + Ok(crate::PrepareResult { + append_requests: vec![(2, AppendEntriesRequest::default(), 1)], + snapshot_targets: vec![], + }) + }); + + state.set_peer_replication_state(2, PeerReplicationState::Replicate); + state.next_index.insert(2, 1); + state.record_in_flight(2, 1); + assert!(state.should_gate_by_inflight(2), "setup: window is full"); + + state.process_batch(one_entry_batch(), &internal_event_tx, &ctx).await.unwrap(); + + assert_eq!( + capture.counter("core.raft.peer.dispatch_gated_total", &[("peer_id", "2")]), + 0, + "a heartbeat bypasses the gate, so it is not a gated dispatch" + ); + assert!( + capture + .histogram("core.raft.peer.in_flight_at_dispatch", &[("peer_id", "2")]) + .is_empty(), + "heartbeats are not window dispatches" + ); + assert!( + capture.histogram("core.raft.replication.entries_per_request", &[]).is_empty(), + "an empty heartbeat must not skew the entries-per-request distribution" + ); +} + +/// The two config gauges are the reference line for reading occupancy / batch-size +/// distributions; they must carry the values the leader was actually built with. +#[tokio::test] +#[traced_test] +async fn test_config_gauges_reflect_configured_values() { + let capture = MetricsCapture::new(); + let _guard = metrics::set_default_local_recorder(&capture); + + let (_graceful_tx, graceful_rx) = watch::channel(()); + let ctx = mock_raft_context("/tmp/test_config_gauges_reflect_values", graceful_rx, None); + let mut cfg = (*ctx.node_config).clone(); + cfg.raft.replication.max_inflight_append_requests = 7; + cfg.raft.replication.append_entries_max_entries_per_replication = 42; + cfg.raft.replication.replication_send_queue_capacity = 900; + + let _state = LeaderState::::new(1, Arc::new(cfg)); + + assert_eq!( + capture.gauge("core.raft.config.max_inflight_append_requests", &[]), + Some(7.0) + ); + assert_eq!( + capture.gauge( + "core.raft.config.append_entries_max_entries_per_replication", + &[] + ), + Some(42.0) + ); + assert_eq!( + capture.gauge("core.raft.config.replication_send_queue_capacity", &[]), + Some(900.0) + ); +} + +/// Real elections build the leader via `From<&CandidateState>`, not `LeaderState::new`, +/// so the config gauges must be set on that path too or they never reach a running cluster. +#[tokio::test] +#[traced_test] +async fn test_config_gauges_are_set_when_leader_is_built_from_candidate() { + let capture = MetricsCapture::new(); + let _guard = metrics::set_default_local_recorder(&capture); + + let (_graceful_tx, graceful_rx) = watch::channel(()); + let ctx = mock_raft_context("/tmp/test_config_gauges_from_candidate", graceful_rx, None); + let mut cfg = (*ctx.node_config).clone(); + cfg.raft.replication.max_inflight_append_requests = 7; + cfg.raft.replication.append_entries_max_entries_per_replication = 42; + cfg.raft.replication.replication_send_queue_capacity = 900; + + let candidate = + crate::raft_role::candidate_state::CandidateState::::new(1, Arc::new(cfg)); + let _leader = LeaderState::::from(&candidate); + + assert_eq!( + capture.gauge("core.raft.config.max_inflight_append_requests", &[]), + Some(7.0), + "production builds the leader via From<&CandidateState>; gauge must be set there" + ); + assert_eq!( + capture.gauge( + "core.raft.config.append_entries_max_entries_per_replication", + &[] + ), + Some(42.0) + ); + assert_eq!( + capture.gauge("core.raft.config.replication_send_queue_capacity", &[]), + Some(900.0) + ); +} diff --git a/d-engine-core/src/raft_role/leader_state_test/replicate_unbounded_dispatch_test.rs b/d-engine-core/src/raft_role/leader_state_test/replicate_unbounded_dispatch_test.rs new file mode 100644 index 00000000..1555b6f5 --- /dev/null +++ b/d-engine-core/src/raft_role/leader_state_test/replicate_unbounded_dispatch_test.rs @@ -0,0 +1,180 @@ +//! Characterizes `LeaderState`'s Phase 5 dispatch loop in isolation: it has NO in-flight limit of +//! its own for `Replicate` state (unlike `Probe`, see `probe_backpressure_test.rs`) and applies +//! `next_index = effective_next_index + entries.len()` to whatever `prepare_batch_requests` hands +//! it, on every batch, unconditionally. +//! +//! Background (full root-cause chain: `replication-worker-backpressure-deadlock-expert-q.md` and +//! `446-replication-stall-case-study-2026-09-28.md` in the product-design repo): a real deadlock +//! reproduced on embedded-bench (100K writes, 3-node localhost) traced to `ReplicationHandler` +//! (`retrieve_to_be_synced_logs_for_peers` / `build_append_request`) repeatedly re-offering the +//! same un-acked range β€” 8955 rebuilds for 28 real acks in one capture β€” which starved the peer's +//! own worker task of scheduling (CPU spent re-fetching/re-cloning the same entries, not network +//! I/O) and spiraled into a permanent stall. +//! +//! **Important scoping correction**: the fix belongs in `ReplicationHandler`, upstream of this +//! layer β€” it must stop *offering* a duplicate un-acked range at all (checked before touching the +//! log, mirroring `raft-rs`'s `Progress::is_paused()` ahead of `maybe_send_append`'s entry fetch). +//! Once that lands, `build_append_request` naturally falls back to an empty (heartbeat) request +//! for a peer with nothing new to offer (`entries_per_peer.remove(..).unwrap_or_default()`), and +//! Phase 5's existing formula correctly no-ops on that (`entries.len() == 0` advances nothing) β€” +//! **no change needed here**. This test's assertions (Phase 5 forwards whatever it's given, +//! unconditionally) describe a real but *different* property than the bug, and should stay GREEN +//! even after the real fix lands β€” it is not the regression test for the fix. That test belongs +//! next to `ReplicationHandler` (`replication_handler_test/`) once the in-flight representation +//! is implemented β€” this file's mocked `prepare_batch_requests` bypasses that logic entirely, by +//! design, to isolate Phase 5's own (correct, by-design) "trust the input" contract. + +use std::collections::VecDeque; +use std::sync::Arc; + +use bytes::Bytes; +use d_engine_proto::common::{Entry, EntryPayload, NodeRole::Follower, NodeStatus}; +use d_engine_proto::server::cluster::NodeMeta; +use d_engine_proto::server::replication::AppendEntriesRequest; +use tokio::sync::{mpsc, watch}; +use tracing_test::traced_test; + +use crate::MockMembership; +use crate::MockRaftLog; +use crate::RaftRequestWithSignal; +use crate::event::InternalEvent; +use crate::maybe_clone_oneshot::{MaybeCloneOneshot, RaftOneshot}; +use crate::raft_role::leader_state::LeaderState; +use crate::raft_role::role_state::{PeerReplicationState, RaftRoleState}; +use crate::test_utils::mock::{MockTypeConfig, mock_raft_context}; + +/// Two-voter membership (peers 2 & 3) β€” same shape as `probe_backpressure_test.rs`'s, duplicated +/// here rather than shared: these helpers are private to their own test module. +fn two_peer_membership() -> MockMembership { + let peers = vec![ + NodeMeta { + id: 2, + address: String::new(), + status: NodeStatus::Active as i32, + role: Follower.into(), + }, + NodeMeta { + id: 3, + address: String::new(), + status: NodeStatus::Active as i32, + role: Follower.into(), + }, + ]; + let peers2 = peers.clone(); + let mut m = MockMembership::new(); + m.expect_is_single_node_cluster().returning(|| false); + m.expect_voters().returning(move || peers.clone()); + m.expect_replication_peers().returning(move || peers2.clone()); + m +} + +/// A non-empty (non-heartbeat) request β€” one entry, so `next_index` advances by exactly 1 per +/// dispatch, making the "raced ahead by ROUNDS" assertion below exact, not just "some drift". +fn stub_probe_request() -> AppendEntriesRequest { + AppendEntriesRequest { + entries: vec![Entry { + index: 1, + term: 1, + payload: None, + }], + ..AppendEntriesRequest::default() + } +} + +/// A single one-entry write batch, matching the shape `process_batch` expects. +fn one_entry_batch() -> VecDeque { + let (tx, _rx) = >::new(); + let req = RaftRequestWithSignal { + id: "test".into(), + payloads: vec![EntryPayload::command(Bytes::from_static(b"cmd"))], + senders: vec![tx], + wait_for_apply_event: false, + }; + VecDeque::from(vec![req]) +} + +/// 50 back-to-back batches, peer 2 in `Replicate`, `handle_append_result` never called for it β€” +/// simulating a worker that accepted every task into its (unbounded) queue but has delivered and +/// had acknowledged literally none of them, exactly what a worker wedged inside a full +/// bounded-128 `stream_sender.send().await` looks like from the leader's side. +#[tokio::test] +#[traced_test] +async fn test_replicate_peer_dispatch_and_next_index_are_unbounded_without_any_ack() { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut ctx = mock_raft_context( + "/tmp/test_replicate_peer_dispatch_and_next_index_are_unbounded_without_any_ack", + graceful_rx, + None, + ); + ctx.membership = Arc::new(two_peer_membership()); + + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 0); + raft_log.expect_flush().returning(|| Ok(())); + raft_log.expect_save_hard_state().returning(|_| Ok(())); + ctx.storage.raft_log = Arc::new(raft_log); + + let mut state = LeaderState::::new(1, ctx.node_config.clone()); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + + // Peer 2's worker channel held directly by the test β€” dispatch is observable synchronously, + // no real transport/worker task involved (same pattern as probe_backpressure_test.rs). + let (task_tx, mut task_rx) = mpsc::unbounded_channel(); + state.replication_workers.insert( + 2, + super::ReplicationWorkerHandle { + task_tx, + snapshot_failure_count: 0, + snapshot_next_retry_at: None, + }, + ); + + // The steady state a peer settles into after its first real ACK β€” where this bug lives. + state.set_peer_replication_state(2, PeerReplicationState::Replicate); + state.next_index.insert(2, 1); + + let (internal_event_tx, _internal_event_rx) = mpsc::unbounded_channel::(); + + const ROUNDS: u64 = 50; + ctx.handlers + .replication_handler + .expect_prepare_batch_requests() + .times(ROUNDS as usize) + .returning(|_, _, leader_state_snapshot, _, _, _| { + // Read the CURRENT next_index for peer 2, same as the real + // `prepare_batch_requests` does via `leader_state_snapshot.next_index` β€” this is + // what makes next_index compound round over round below, faithfully reproducing + // the real feedback loop instead of a fixed stand-in value. + let effective_next_index = + leader_state_snapshot.next_index.get(&2).copied().unwrap_or(1); + Ok(crate::PrepareResult { + append_requests: vec![(2, stub_probe_request(), effective_next_index)], + snapshot_targets: vec![], + }) + }); + + for _ in 0..ROUNDS { + state.process_batch(one_entry_batch(), &internal_event_tx, &ctx).await.unwrap(); + } + + let mut dispatched = 0u64; + while task_rx.try_recv().is_ok() { + dispatched += 1; + } + assert_eq!( + dispatched, ROUNDS, + "a Replicate-state peer accepted every one of {ROUNDS} un-acked dispatches β€” there is no \ + in-flight limit at all for Replicate (unlike Probe), so a wedged worker never causes \ + back-pressure on the dispatch loop" + ); + + // stub_probe_request() carries 1 entry, so next_index advances by 1 per round purely from + // dispatch β€” zero acknowledgments were ever processed. + assert_eq!( + state.next_index.get(&2), + Some(&(1 + ROUNDS)), + "next_index advanced by one dispatch's worth of entries on every single round with zero \ + real acknowledgments β€” the leader's belief about this peer's progress has no bound tying \ + it to what has actually been confirmed delivered" + ); +} diff --git a/d-engine-core/src/raft_role/leader_state_test/replication_test.rs b/d-engine-core/src/raft_role/leader_state_test/replication_test.rs index 4479e71c..0fef8c40 100644 --- a/d-engine-core/src/raft_role/leader_state_test/replication_test.rs +++ b/d-engine-core/src/raft_role/leader_state_test/replication_test.rs @@ -175,7 +175,7 @@ async fn test_process_batch_quorum_achieved() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let last_entry_id = Arc::new(AtomicU64::new(4)); let last_entry_id_clone = last_entry_id.clone(); @@ -255,7 +255,7 @@ async fn test_process_batch_quorum_failed_verifiable() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let mut raft_log = MockRaftLog::new(); raft_log.expect_last_entry_id().returning(|| 4); @@ -316,7 +316,7 @@ async fn test_process_batch_quorum_non_verifiable_failure() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let mut raft_log = MockRaftLog::new(); raft_log.expect_last_entry_id().returning(|| 4); @@ -370,7 +370,7 @@ async fn test_process_batch_higher_term() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let mut raft_log = MockRaftLog::new(); raft_log.expect_last_entry_id().returning(|| 4); @@ -446,7 +446,7 @@ async fn test_process_batch_partial_timeouts() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let mut raft_log = MockRaftLog::new(); raft_log.expect_last_entry_id().returning(|| 4); @@ -497,7 +497,7 @@ async fn test_process_batch_all_timeout() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let mut raft_log = MockRaftLog::new(); raft_log.expect_last_entry_id().returning(|| 4); @@ -548,7 +548,7 @@ async fn test_process_batch_fatal_error() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Err(Error::Fatal("Storage failure".to_string()))); + .returning(|_, _, _, _, _, _| Err(Error::Fatal("Storage failure".to_string()))); let mut raft_log = MockRaftLog::new(); raft_log.expect_last_entry_id().returning(|| 4); @@ -667,7 +667,7 @@ async fn setup_commit_index_test_context( /// /// # When /// - process_batch is called (no peers β†’ no replication requests sent) -/// - handle_log_flushed(7) is called to simulate the async LogFlushed event from BufferedRaftLog +/// - handle_log_flushed(7) is called to simulate the async LogFlushed event from RaftLogCore /// /// # Then /// - Commit index advances to 7 (driven by the simulated LogFlushed event) @@ -687,7 +687,7 @@ async fn test_single_node_cluster_commit_index() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let last_entry_id = Arc::new(AtomicU64::new(6)); let last_entry_id_clone = last_entry_id.clone(); @@ -708,8 +708,8 @@ async fn test_single_node_cluster_commit_index() { .await; assert!(result.is_ok()); - // Simulate the async LogFlushed event that BufferedRaftLog's batch_processor fires - // after fsync. MemFirst: set last_entry_id=7 before flush. + // Simulate the async LogFlushed event that RaftLogCore emits after fsync. + // Set last_entry_id=7 before flush. last_entry_id.store(7, Ordering::Relaxed); context .state @@ -763,7 +763,7 @@ async fn test_multi_node_cluster_empty_peer_updates_commit_index() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let mut raft_log = MockRaftLog::new(); raft_log.expect_last_entry_id().returning(|| 9); @@ -831,7 +831,7 @@ async fn test_multi_node_cluster_with_peer_updates_commit_index() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); context .raft_context .handlers @@ -909,7 +909,7 @@ async fn test_multi_node_cluster_with_peer_updates_commit_index() { /// /// # When /// - execute_request_immediately is called -/// - handle_log_flushed(5) is called to simulate the async LogFlushed event from BufferedRaftLog +/// - handle_log_flushed(5) is called to simulate the async LogFlushed event from RaftLogCore /// /// # Then /// - Commit index advances to 5 @@ -937,7 +937,7 @@ async fn test_verify_internal_quorum_success() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let last_entry_id = Arc::new(AtomicU64::new(4)); let last_entry_id_clone = last_entry_id.clone(); @@ -1015,7 +1015,7 @@ async fn test_verify_internal_quorum_verifiable_failure() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let mut raft_log = MockRaftLog::new(); raft_log.expect_last_entry_id().returning(|| 4); @@ -1078,7 +1078,7 @@ async fn test_verify_internal_quorum_non_verifiable_failure() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let mut raft_log = MockRaftLog::new(); raft_log.expect_last_entry_id().returning(|| 4); @@ -1168,7 +1168,7 @@ async fn test_verify_internal_quorum_partial_timeouts() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let mut raft_log = MockRaftLog::new(); raft_log.expect_last_entry_id().returning(|| 4); @@ -1226,7 +1226,7 @@ async fn test_verify_internal_quorum_all_timeouts() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let mut raft_log = MockRaftLog::new(); raft_log.expect_last_entry_id().returning(|| 4); @@ -1291,7 +1291,7 @@ async fn test_verify_internal_quorum_higher_term() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let mut raft_log = MockRaftLog::new(); raft_log.expect_last_entry_id().returning(|| 4); @@ -1367,7 +1367,7 @@ async fn test_verify_internal_quorum_critical_failure() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Err(Error::Fatal("Storage failure".to_string()))); + .returning(|_, _, _, _, _, _| Err(Error::Fatal("Storage failure".to_string()))); let mut raft_log = MockRaftLog::new(); raft_log.expect_last_entry_id().returning(|| 4); @@ -1436,7 +1436,7 @@ async fn test_execute_and_process_raft_rpc_multi_node_empty_peer_updates() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| Ok(crate::PrepareResult::default())); + .returning(|_, _, _, _, _, _| Ok(crate::PrepareResult::default())); let mut raft_log = MockRaftLog::new(); // Sentinel: if single-node path were taken, commit would jump to 10 (wrong) @@ -1609,6 +1609,7 @@ async fn test_handle_append_result_two_node_quorum_achieved() { }); let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 3); raft_log.expect_calculate_majority_matched_index().returning(|_, _, _| Some(3)); raft_context.storage.raft_log = Arc::new(raft_log); @@ -1674,6 +1675,7 @@ async fn test_handle_append_result_three_node_quorum_all_peers() { }); let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 10); raft_log .expect_calculate_majority_matched_index() .returning( @@ -1762,6 +1764,7 @@ async fn test_handle_append_result_three_node_quorum_partial_timeout() { }); let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 10); raft_log.expect_calculate_majority_matched_index().returning(|_, _, _| Some(10)); context.raft_context.storage.raft_log = Arc::new(raft_log); @@ -1842,6 +1845,7 @@ async fn test_handle_append_result_five_node_quorum_majority() { }); let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 10); raft_log .expect_calculate_majority_matched_index() .returning( @@ -1976,6 +1980,7 @@ async fn test_handle_append_result_three_node_no_quorum_single_peer_insufficient // calculate_majority returns None β€” peer's match_index too low for new commit let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 4); raft_log.expect_calculate_majority_matched_index().returning(|_, _, _| None); context.raft_context.storage.raft_log = Arc::new(raft_log); @@ -2056,6 +2061,7 @@ async fn test_handle_append_result_five_node_no_quorum_minority() { // 1 peer (2/5 nodes) is minority β€” calculate_majority returns None let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 10); raft_log.expect_calculate_majority_matched_index().returning(|_, _, _| None); context.raft_context.storage.raft_log = Arc::new(raft_log); @@ -2338,8 +2344,14 @@ async fn setup_state_with_next_and_match( }); let (_graceful_tx, graceful_rx) = watch::channel(()); - let mut context_inner = - MockBuilder::new(graceful_rx).with_replication_handler(rep).build_context(); + // The conflict hint carries `next_index`; the leader must own a log at least that + // long so `bounded_by_leader_log` doesn't clamp the hint below itself. + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(move || next_index); + let mut context_inner = MockBuilder::new(graceful_rx) + .with_replication_handler(rep) + .with_raft_log(raft_log) + .build_context(); // Set up membership with peer 2 and 3 as voters (same as setup_commit_index_test_context). let mut membership = MockMembership::new(); diff --git a/d-engine-core/src/raft_role/leader_state_test/single_voter_commit_test.rs b/d-engine-core/src/raft_role/leader_state_test/single_voter_commit_test.rs index 3b9b8d07..2f6e7ec1 100644 --- a/d-engine-core/src/raft_role/leader_state_test/single_voter_commit_test.rs +++ b/d-engine-core/src/raft_role/leader_state_test/single_voter_commit_test.rs @@ -1,17 +1,16 @@ //! Single-Voter Commit Path Tests //! -//! Regression tests for the MemFirst single-voter commit path in `handle_log_flushed`. +//! RPO=0 (#446): `handle_log_flushed` single-voter branch must commit to `durable`, not +//! `last_entry_id()` β€” a single-voter cluster has no majority to fall back on, so if the +//! leader itself hasn't fsynced an entry, there is no copy anywhere safe from power loss. //! -//! ## Bug History -//! `fix #329` changed `handle_log_flushed` single-voter branch to commit to `durable` -//! instead of `last_entry_id()`. This placed IO thread latency on the commit critical -//! path, causing a ~3x latency regression in 3-node embedded bench (1731Β΅s vs ~566Β΅s). -//! -//! ## MemFirst Single-Voter Invariant -//! `LogFlushed(durable)` is an IO checkpoint. Commit must advance to `last_entry_id()` -//! β€” not just `durable` β€” to allow pipelining across IO batch boundaries. -//! This matches the multi-voter path where the leader contributes `last_entry_id()` to -//! quorum (not `durable_index`). +//! ## Superseded design (kept as history, do not resurrect) +//! `fix #329` changed this branch to commit to `durable` instead of `last_entry_id()`, +//! then reverted it after measuring a ~3x latency regression in 3-node embedded bench +//! (1731Β΅s vs ~566Β΅s) β€” IO thread latency landed on the commit critical path. That +//! regression is real and will resurface here. RPO=0 makes paying it mandatory for +//! single-voter clusters β€” there is no majority to absorb the risk the old design +//! accepted. use crate::MockMembership; use crate::MockRaftLog; @@ -61,16 +60,16 @@ async fn setup_single_voter( (state, ctx, last_entry_id) } -/// MemFirst single-voter: `handle_log_flushed` must commit to `last_entry_id`, not `durable`. +/// RPO=0: `handle_log_flushed` must commit to `durable`, not `last_entry_id`. /// -/// Simulates: IO batch flushed entries 1-3 (`durable=3`), but entries 4-5 arrived -/// in memory during the flush (`last_entry_id=5`). MemFirst: commit must advance -/// to 5 (all in-memory entries), not stall at 3 (only persisted entries). +/// Simulates: entries 4-5 arrived in memory (`last_entry_id=5`) but the IO batch has +/// only flushed entries 1-3 so far (`durable=3`). Commit must stay at 3 β€” entries 4-5 +/// aren't crash-safe yet, and a single-voter cluster has no other copy to fall back on. /// -/// This test FAILS if `handle_log_flushed` uses `durable` for commit -/// (the `fix #329` regression that caused +617Β΅s avg latency in 3-node embedded bench). +/// This test FAILS if `handle_log_flushed` still uses `last_entry_id` for commit (the +/// old MemFirst behavior, since revoked β€” RPO=0 makes single-voter durability mandatory). #[tokio::test] -async fn test_single_voter_commit_uses_last_entry_id_not_durable() { +async fn test_single_voter_commit_uses_durable_not_last_entry_id() { // last_entry_id=5: entries 4-5 arrived in memory during the IO flush of 1-3 let (mut state, ctx, _last_entry_id) = setup_single_voter(5).await; let (internal_event_tx, _internal_event_rx) = mpsc::unbounded_channel(); @@ -80,13 +79,13 @@ async fn test_single_voter_commit_uses_last_entry_id_not_durable() { assert_eq!( state.commit_index(), - 5, - "MemFirst single-voter: commit must use last_entry_id=5, not durable=3. \ - Using durable puts IO latency on the commit critical path." + 3, + "RPO=0: commit must use durable=3, not last_entry_id=5 β€” entries 4-5 aren't \ + fsynced yet, and single-voter has no majority to fall back on" ); } -/// After IO catches up (durable == last_entry_id), commit equals last_entry_id. +/// After IO catches up (durable == last_entry_id), commit equals durable. #[tokio::test] async fn test_single_voter_commit_when_durable_equals_last_entry_id() { let (mut state, ctx, _last_entry_id) = setup_single_voter(5).await; @@ -94,20 +93,16 @@ async fn test_single_voter_commit_when_durable_equals_last_entry_id() { state.handle_log_flushed(5, &ctx, &internal_event_tx).await; - assert_eq!( - state.commit_index(), - 5, - "commit must advance to last_entry_id=5 when durable=5" - ); + assert_eq!(state.commit_index(), 5, "commit must advance to durable=5"); } -/// Pipelining across multiple IO batches: each flush triggers commit to current last_entry_id. +/// Commit tracks `durable` across IO batches, not the in-memory tail. /// -/// Simulates rapid writes where IO batches lag behind in-memory log: -/// - Flush 1: IO flushed 1-3, log has 1-7 in memory β†’ commit=7 -/// - Flush 2: IO flushed 4-7, log has 1-10 in memory β†’ commit=10 +/// Simulates rapid writes where the in-memory log runs ahead of what's fsynced: +/// - Flush 1: IO flushed 1-3 (durable=3), memory has 1-7 β†’ commit=3, not 7 +/// - Flush 2: IO flushed 4-7 (durable=7), memory now has 1-10 β†’ commit=7, not 10 #[tokio::test] -async fn test_single_voter_pipelining_across_io_batches() { +async fn test_single_voter_commit_tracks_durable_not_memory_tail() { let (mut state, ctx, last_entry_id) = setup_single_voter(7).await; let (internal_event_tx, _internal_event_rx) = mpsc::unbounded_channel(); @@ -115,8 +110,8 @@ async fn test_single_voter_pipelining_across_io_batches() { state.handle_log_flushed(3, &ctx, &internal_event_tx).await; assert_eq!( state.commit_index(), - 7, - "commit must advance to last_entry_id=7" + 3, + "commit must stay at durable=3 β€” entries 4-7 aren't fsynced yet" ); // IO batch 2: flushed 4-7, memory now has 1-10 @@ -124,14 +119,14 @@ async fn test_single_voter_pipelining_across_io_batches() { state.handle_log_flushed(7, &ctx, &internal_event_tx).await; assert_eq!( state.commit_index(), - 10, - "commit must advance to last_entry_id=10" + 7, + "commit must advance to durable=7, not the in-memory tail (10)" ); } -/// No-op flush: last_entry_id == commit_index means nothing new to commit. +/// No-op flush: durable == commit_index means nothing new is safe to commit yet. #[tokio::test] -async fn test_single_voter_no_commit_when_nothing_new() { +async fn test_single_voter_no_commit_when_nothing_new_durable() { let (mut state, ctx, _last_entry_id) = setup_single_voter(3).await; let (internal_event_tx, _internal_event_rx) = mpsc::unbounded_channel(); @@ -139,11 +134,11 @@ async fn test_single_voter_no_commit_when_nothing_new() { state.handle_log_flushed(3, &ctx, &internal_event_tx).await; assert_eq!(state.commit_index(), 3); - // Second flush with same last_entry_id=3: no new entries β†’ no commit advance + // Second flush with same durable=3: nothing new is fsynced β†’ no commit advance state.handle_log_flushed(3, &ctx, &internal_event_tx).await; assert_eq!( state.commit_index(), 3, - "commit must not advance when last_entry_id == commit_index" + "commit must not advance when durable == commit_index" ); } diff --git a/d-engine-core/src/raft_role/leader_state_test/snapshot_worker_test.rs b/d-engine-core/src/raft_role/leader_state_test/snapshot_worker_test.rs index a2f31812..593eebd3 100644 --- a/d-engine-core/src/raft_role/leader_state_test/snapshot_worker_test.rs +++ b/d-engine-core/src/raft_role/leader_state_test/snapshot_worker_test.rs @@ -167,7 +167,7 @@ fn open_stream_capturing_appends() -> ( ) { let (append_tx, append_rx) = mpsc::channel::(128); let mut transport = MockTransport::::new(); - transport.expect_open_replication_stream().returning(move |_, _, _| { + transport.expect_open_replication_stream().returning(move |_, _, _, _| { let stream = futures::stream::pending().boxed(); Ok(crate::ReplicationStream { sender: append_tx.clone(), @@ -202,7 +202,7 @@ async fn test_worker_handles_snapshot_task_and_emits_completed_event() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| { + .returning(|_, _, _, _, _, _| { Ok(crate::replication::PrepareResult { append_requests: vec![], snapshot_targets: vec![2], @@ -211,7 +211,7 @@ async fn test_worker_handles_snapshot_task_and_emits_completed_event() { let mut transport = MockTransport::::new(); transport.expect_send_snapshot().times(1).returning(|_, _, _, _, _, _| Ok(())); - transport.expect_open_replication_stream().returning(|_, _, _| { + transport.expect_open_replication_stream().returning(|_, _, _, _| { let (tx, _rx) = mpsc::channel(128); let stream = futures::stream::empty().boxed(); Ok(crate::ReplicationStream { @@ -265,7 +265,7 @@ async fn test_phase6_skips_redispatch_when_snapshot_already_in_flight() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| { + .returning(|_, _, _, _, _, _| { Ok(crate::replication::PrepareResult { append_requests: vec![], snapshot_targets: vec![2], @@ -275,7 +275,7 @@ async fn test_phase6_skips_redispatch_when_snapshot_already_in_flight() { let mut transport = MockTransport::::new(); // Must NEVER be called β€” the leader-side check must skip dispatch entirely. transport.expect_send_snapshot().times(0); - transport.expect_open_replication_stream().returning(|_, _, _| { + transport.expect_open_replication_stream().returning(|_, _, _, _| { let (tx, _rx) = mpsc::channel(128); let stream = futures::stream::empty().boxed(); Ok(crate::ReplicationStream { @@ -316,7 +316,7 @@ async fn test_phase6_snapshot_dispatch_sets_snapshot_state() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| { + .returning(|_, _, _, _, _, _| { Ok(crate::replication::PrepareResult { append_requests: vec![], snapshot_targets: vec![2], @@ -325,7 +325,7 @@ async fn test_phase6_snapshot_dispatch_sets_snapshot_state() { let mut transport = MockTransport::::new(); transport.expect_send_snapshot().times(1).returning(|_, _, _, _, _, _| Ok(())); - transport.expect_open_replication_stream().returning(|_, _, _| { + transport.expect_open_replication_stream().returning(|_, _, _, _| { let (tx, _rx) = mpsc::channel(128); let stream = futures::stream::empty().boxed(); Ok(crate::ReplicationStream { @@ -372,7 +372,7 @@ async fn test_phase5_skips_append_and_leaves_next_index_untouched_while_peer_in_ .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| { + .returning(|_, _, _, _, _, _| { Ok(crate::replication::PrepareResult { append_requests: vec![(2, stub_append_request(), 1)], snapshot_targets: vec![], @@ -417,28 +417,22 @@ async fn test_phase5_skips_append_and_leaves_next_index_untouched_while_peer_in_ /// - Expected: the request reaches the worker's bidi stream unmodified. #[tokio::test] async fn test_worker_forwards_any_append_task_it_is_given_without_inspecting_state() { - let (mut state, ctx) = new_leader_and_ctx( + let (mut state, mut ctx) = new_leader_and_ctx( "/tmp/test_worker_forwards_any_append_task_it_is_given_without_inspecting_state", ); state.init_cluster_metadata(&ctx.membership).await.unwrap(); state.set_peer_replication_state(2, PeerReplicationState::Snapshot); let (transport, mut append_rx) = open_stream_capturing_appends(); + ctx.transport = Arc::new(transport); let (internal_event_tx, _internal_event_rx) = mpsc::unbounded_channel(); // Bypasses process_batch/Phase 5 on purpose β€” see doc comment above. state.send_to_worker_or_spawn( 2, - super::ReplicationTask::Append(stub_append_request()), - super::ReplicationWorkerConfig { - transport: Arc::new(transport), - membership: ctx.membership.clone(), - retry_policies: ctx.node_config.retry.clone(), - response_compress_enabled: ctx.node_config.raft.rpc_compression.replication_response, - internal_event_tx, - state_machine_handler: ctx.state_machine_handler().clone(), - snapshot_config: ctx.node_config.raft.snapshot.clone(), - }, + super::ReplicationTask::Append(stub_append_request(), tokio::time::Instant::now()), + &ctx, + &internal_event_tx, ); let received = tokio::time::timeout(Duration::from_millis(200), append_rx.recv()) @@ -477,7 +471,7 @@ async fn test_snapshot_completion_success_returns_to_probe_and_allows_append() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| { + .returning(|_, _, _, _, _, _| { Ok(crate::replication::PrepareResult { append_requests: vec![], snapshot_targets: vec![2], @@ -501,7 +495,7 @@ async fn test_snapshot_completion_success_returns_to_probe_and_allows_append() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| { + .returning(|_, _, _, _, _, _| { Ok(crate::replication::PrepareResult { append_requests: vec![(2, stub_append_request(), 1)], snapshot_targets: vec![], @@ -538,7 +532,7 @@ async fn test_snapshot_completion_failure_returns_to_probe_with_backoff() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| { + .returning(|_, _, _, _, _, _| { Ok(crate::replication::PrepareResult { append_requests: vec![], snapshot_targets: vec![2], @@ -549,7 +543,7 @@ async fn test_snapshot_completion_failure_returns_to_probe_with_backoff() { .expect_send_snapshot() .times(1) .returning(|_, _, _, _, _, _| Ok(())); - dispatch_transport.expect_open_replication_stream().returning(|_, _, _| { + dispatch_transport.expect_open_replication_stream().returning(|_, _, _, _| { let (tx, _rx) = mpsc::channel(128); let stream = futures::stream::empty().boxed(); Ok(crate::ReplicationStream { @@ -637,7 +631,7 @@ async fn test_backoff_window_blocks_redispatch_even_when_classification_still_re .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| { + .returning(|_, _, _, _, _, _| { Ok(crate::replication::PrepareResult { append_requests: vec![], snapshot_targets: vec![2], @@ -645,7 +639,7 @@ async fn test_backoff_window_blocks_redispatch_even_when_classification_still_re }); let mut transport = MockTransport::::new(); transport.expect_send_snapshot().times(1).returning(|_, _, _, _, _, _| Ok(())); - transport.expect_open_replication_stream().returning(|_, _, _| { + transport.expect_open_replication_stream().returning(|_, _, _, _| { let (tx, _rx) = mpsc::channel(128); let stream = futures::stream::empty().boxed(); Ok(crate::ReplicationStream { @@ -672,7 +666,7 @@ async fn test_backoff_window_blocks_redispatch_even_when_classification_still_re // backoff window from the failure above has not expired yet. let mut transport2 = MockTransport::::new(); transport2.expect_send_snapshot().times(0); // must NOT be called again β€” still in backoff - transport2.expect_open_replication_stream().returning(|_, _, _| { + transport2.expect_open_replication_stream().returning(|_, _, _, _| { let (tx, _rx) = mpsc::channel(128); let stream = futures::stream::empty().boxed(); Ok(crate::ReplicationStream { @@ -685,7 +679,7 @@ async fn test_backoff_window_blocks_redispatch_even_when_classification_still_re .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| { + .returning(|_, _, _, _, _, _| { Ok(crate::replication::PrepareResult { append_requests: vec![], snapshot_targets: vec![2], diff --git a/d-engine-core/src/raft_role/leader_state_test/stale_learner_deadline_test.rs b/d-engine-core/src/raft_role/leader_state_test/stale_learner_deadline_test.rs index f917721e..14a9d309 100644 --- a/d-engine-core/src/raft_role/leader_state_test/stale_learner_deadline_test.rs +++ b/d-engine-core/src/raft_role/leader_state_test/stale_learner_deadline_test.rs @@ -176,7 +176,7 @@ async fn test_tick_removes_stale_learner_when_deadline_fires() { .replication_handler .expect_prepare_batch_requests() .times(..) - .returning(move |payloads, _, _, _, _| { + .returning(move |payloads, _, _, _, _, _| { captured_clone.lock().extend(payloads.clone()); Ok(crate::PrepareResult::default()) }); diff --git a/d-engine-core/src/raft_role/leader_state_test/worker_lifecycle_test.rs b/d-engine-core/src/raft_role/leader_state_test/worker_lifecycle_test.rs index a51c3e72..d2717208 100644 --- a/d-engine-core/src/raft_role/leader_state_test/worker_lifecycle_test.rs +++ b/d-engine-core/src/raft_role/leader_state_test/worker_lifecycle_test.rs @@ -18,6 +18,7 @@ use d_engine_proto::server::cluster::NodeMeta; use d_engine_proto::server::replication::{AppendEntriesRequest, AppendEntriesResponse}; use futures::StreamExt; use tokio::sync::{mpsc, watch}; +use tokio_stream::wrappers::UnboundedReceiverStream; use tracing_test::traced_test; use crate::MockMembership; @@ -61,6 +62,20 @@ fn stub_request() -> AppendEntriesRequest { AppendEntriesRequest::default() } +/// Like `stub_request()`, but with one real `Entry` β€” needed wherever the code +/// under test reads `request.entries.last()` (e.g. RTT sampling), which a +/// default/empty request can never satisfy. +fn stub_request_with_entry(index: u64) -> AppendEntriesRequest { + AppendEntriesRequest { + entries: vec![d_engine_proto::common::Entry { + index, + term: 1, + payload: None, + }], + ..Default::default() + } +} + /// A minimal `AppendEntriesResponse` with `success = true`. fn stub_response() -> AppendEntriesResponse { AppendEntriesResponse::default() @@ -106,7 +121,7 @@ async fn test_worker_spawned_on_first_request() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| { + .returning(|_, _, _, _, _, _| { Ok(crate::PrepareResult { append_requests: vec![(2, stub_request(), 1)], snapshot_targets: vec![], @@ -119,7 +134,7 @@ async fn test_worker_spawned_on_first_request() { // worker count assertion, not the transport call count. let mut transport = MockTransport::::new(); // Worker opens bidi stream at startup (new behavior in #345) - transport.expect_open_replication_stream().returning(|_, _, _| { + transport.expect_open_replication_stream().returning(|_, _, _, _| { let (req_tx, mut req_rx) = tokio::sync::mpsc::channel(128); let (resp_tx, resp_rx) = tokio::sync::mpsc::channel(128); @@ -193,7 +208,7 @@ async fn test_worker_reused_on_subsequent_requests() { .replication_handler .expect_prepare_batch_requests() .times(2) - .returning(|_, _, _, _, _| { + .returning(|_, _, _, _, _, _| { Ok(crate::PrepareResult { append_requests: vec![(2, stub_request(), 1)], snapshot_targets: vec![], @@ -204,7 +219,7 @@ async fn test_worker_reused_on_subsequent_requests() { // because the worker runs in background and may not have executed by teardown. let mut transport = MockTransport::::new(); // Worker opens bidi stream at startup (new behavior in #345) - transport.expect_open_replication_stream().returning(|_, _, _| { + transport.expect_open_replication_stream().returning(|_, _, _, _| { let (req_tx, mut req_rx) = tokio::sync::mpsc::channel(128); let (resp_tx, resp_rx) = tokio::sync::mpsc::channel(128); @@ -274,7 +289,7 @@ async fn test_worker_rebuilt_after_death() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| { + .returning(|_, _, _, _, _, _| { Ok(crate::PrepareResult { append_requests: vec![(2, stub_request(), 1)], snapshot_targets: vec![], @@ -285,7 +300,7 @@ async fn test_worker_rebuilt_after_death() { // indirectly via AppendResult on internal_event_rx rather than mock call count. let mut transport = MockTransport::::new(); // Worker opens bidi stream at startup (new behavior in #345) - transport.expect_open_replication_stream().returning(|_, _, _| { + transport.expect_open_replication_stream().returning(|_, _, _, _| { let (req_tx, mut req_rx) = tokio::sync::mpsc::channel(128); let (resp_tx, resp_rx) = tokio::sync::mpsc::channel(128); @@ -374,7 +389,7 @@ async fn test_replication_worker_exits_when_handle_dropped() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| { + .returning(|_, _, _, _, _, _| { Ok(crate::PrepareResult { append_requests: vec![(2, stub_request(), 1)], snapshot_targets: vec![], @@ -389,7 +404,7 @@ async fn test_replication_worker_exits_when_handle_dropped() { let (first_attempt_tx, mut first_attempt_rx) = tokio::sync::mpsc::channel::<()>(1); let mut transport = MockTransport::::new(); - transport.expect_open_replication_stream().returning(move |_, _, _| { + transport.expect_open_replication_stream().returning(move |_, _, _, _| { let n = call_count_clone.fetch_add(1, Ordering::SeqCst) + 1; if n == 1 { // Signal test that worker reached the reconnect loop. @@ -462,7 +477,7 @@ async fn test_worker_reconnects_on_stream_recv_error() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| { + .returning(|_, _, _, _, _, _| { Ok(crate::PrepareResult { append_requests: vec![(2, stub_request(), 1)], snapshot_targets: vec![], @@ -473,7 +488,7 @@ async fn test_worker_reconnects_on_stream_recv_error() { let call_count_clone = Arc::clone(&call_count); let mut transport = MockTransport::::new(); - transport.expect_open_replication_stream().returning(move |_, _, _| { + transport.expect_open_replication_stream().returning(move |_, _, _, _| { let n = call_count_clone.fetch_add(1, Ordering::SeqCst) + 1; if n == 1 { // First open succeeds, but the stream immediately errors. @@ -556,7 +571,7 @@ async fn test_replication_worker_sender_closed_emits_peer_stream_error() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| { + .returning(|_, _, _, _, _, _| { Ok(crate::PrepareResult { append_requests: vec![(2, stub_request(), 1)], snapshot_targets: vec![], @@ -568,18 +583,21 @@ async fn test_replication_worker_sender_closed_emits_peer_stream_error() { let open_count = Arc::new(AtomicUsize::new(0)); let open_count_clone = Arc::clone(&open_count); let mut transport = MockTransport::::new(); - transport.expect_open_replication_stream().times(..).returning(move |_, _, _| { - let attempt = open_count_clone.fetch_add(1, Ordering::SeqCst) + 1; - let (sender, send_receiver) = tokio::sync::mpsc::channel::(128); - if attempt == 1 { - drop(send_receiver); - } - // Keep the ACK stream pending so the biased select! reaches the Append arm - // (attempt 1) and the worker parks (later attempts). - let receiver = - futures::stream::pending::>().boxed(); - Ok(crate::ReplicationStream { sender, receiver }) - }); + transport + .expect_open_replication_stream() + .times(..) + .returning(move |_, _, _, _| { + let attempt = open_count_clone.fetch_add(1, Ordering::SeqCst) + 1; + let (sender, send_receiver) = tokio::sync::mpsc::channel::(128); + if attempt == 1 { + drop(send_receiver); + } + // Keep the ACK stream pending so the biased select! reaches the Append arm + // (attempt 1) and the worker parks (later attempts). + let receiver = + futures::stream::pending::>().boxed(); + Ok(crate::ReplicationStream { sender, receiver }) + }); ctx.transport = Arc::new(transport); let mut raft_log = MockRaftLog::new(); @@ -604,6 +622,82 @@ async fn test_replication_worker_sender_closed_emits_peer_stream_error() { ); } +/// A full (not closed) bidi send buffer must also surface as `PeerStreamError` instead of +/// parking the worker: the peer is alive but not draining, so the leader has to stop trusting +/// its optimistic pipeline for this peer rather than silently wedge behind the buffer. +/// +/// # Covered branch +/// `stream_sender.try_send(request)` returning `Full`. The sender's receiver stays alive and +/// undrained, with the channel pre-filled to capacity, so the failure is `Full`, not `Closed`. +/// With a blocking `send().await` this test would time out instead of seeing the event. +#[tokio::test] +#[traced_test] +async fn test_replication_worker_send_buffer_full_emits_peer_stream_error() { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut ctx = mock_raft_context( + "/tmp/test_replication_worker_send_buffer_full_emits_peer_stream_error", + graceful_rx, + None, + ); + + ctx.membership = Arc::new(two_peer_membership()); + + ctx.handlers + .replication_handler + .expect_prepare_batch_requests() + .times(1) + .returning(|_, _, _, _, _, _| { + Ok(crate::PrepareResult { + append_requests: vec![(2, stub_request(), 1)], + snapshot_targets: vec![], + }) + }); + + // First stream: capacity-1 channel, already full, receiver kept alive (never drained). + // Later reconnects park on a pending ACK stream so the worker stops reconnecting. + let open_count = Arc::new(AtomicUsize::new(0)); + let open_count_clone = Arc::clone(&open_count); + let held_receivers = Arc::new(std::sync::Mutex::new(Vec::new())); + let held_receivers_clone = Arc::clone(&held_receivers); + let mut transport = MockTransport::::new(); + transport + .expect_open_replication_stream() + .times(..) + .returning(move |_, _, _, _| { + let attempt = open_count_clone.fetch_add(1, Ordering::SeqCst) + 1; + let (sender, send_receiver) = tokio::sync::mpsc::channel::(1); + if attempt == 1 { + sender.try_send(AppendEntriesRequest::default()).unwrap(); + } + held_receivers_clone.lock().unwrap().push(send_receiver); + let receiver = + futures::stream::pending::>().boxed(); + Ok(crate::ReplicationStream { sender, receiver }) + }); + ctx.transport = Arc::new(transport); + + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 0); + raft_log.expect_flush().returning(|| Ok(())); + raft_log.expect_save_hard_state().returning(|_| Ok(())); + ctx.storage.raft_log = Arc::new(raft_log); + + let mut state = LeaderState::::new(1, ctx.node_config.clone()); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + + let (internal_event_tx, mut internal_event_rx) = mpsc::unbounded_channel(); + state.process_batch(one_entry_batch(), &internal_event_tx, &ctx).await.unwrap(); + + let event = tokio::time::timeout(std::time::Duration::from_secs(2), internal_event_rx.recv()) + .await + .expect("timed out β€” worker parked on a full send buffer instead of failing fast") + .expect("event channel closed before PeerStreamError"); + assert!( + matches!(event, InternalEvent::PeerStreamError { peer_id: 2 }), + "expected PeerStreamError for peer 2, got {event:?}" + ); +} + /// Worker forwards a successful stream ACK as `AppendResult`. /// /// # Scenario @@ -628,7 +722,7 @@ async fn test_worker_forwards_append_result_on_recv() { .replication_handler .expect_prepare_batch_requests() .times(1) - .returning(|_, _, _, _, _| { + .returning(|_, _, _, _, _, _| { Ok(crate::PrepareResult { append_requests: vec![(2, stub_request(), 1)], snapshot_targets: vec![], @@ -639,7 +733,7 @@ async fn test_worker_forwards_append_result_on_recv() { let call_count_clone = Arc::clone(&call_count); let mut transport = MockTransport::::new(); - transport.expect_open_replication_stream().returning(move |_, _, _| { + transport.expect_open_replication_stream().returning(move |_, _, _, _| { let n = call_count_clone.fetch_add(1, Ordering::SeqCst) + 1; if n == 1 { // First open succeeds and delivers one successful ACK. @@ -685,9 +779,189 @@ async fn test_worker_forwards_append_result_on_recv() { event, InternalEvent::AppendResult { follower_id: 2, - result: Ok(_) + result: Ok(_), + .. } ), "expected AppendResult for peer 2, got {event:?}" ); } + +/// Reconnecting mid-sample must reset `rtt_sample_pending`, or RTT sampling +/// permanently stops for the rest of this worker's lifetime (#446). +/// +/// # Scenario +/// - 16 sends land on the sampling boundary exactly once (the 16th), starting +/// an RTT sample that never resolves β€” the mock stream stays pending, no ACK. +/// - The stream then errors, forcing `recv_handle.abort()` + reconnect. +/// +/// # Guarantee checked +/// The reconnect path must reset the interrupted sample (logged as "RTT sample +/// reset after reconnect"), not leave `rtt_sample_pending` stuck `true` forever. +#[tokio::test] +#[traced_test] +async fn test_rtt_sample_reset_on_reconnect_after_interrupted_sample() { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut ctx = mock_raft_context( + "/tmp/test_rtt_sample_reset_on_reconnect_after_interrupted_sample", + graceful_rx, + None, + ); + + ctx.membership = Arc::new(two_peer_membership()); + + let prepare_count = Arc::new(AtomicUsize::new(0)); + let prepare_count_clone = Arc::clone(&prepare_count); + ctx.handlers + .replication_handler + .expect_prepare_batch_requests() + .times(16) + .returning(move |_, _, _, _, _, _| { + let n = prepare_count_clone.fetch_add(1, Ordering::SeqCst) as u64 + 1; + Ok(crate::PrepareResult { + append_requests: vec![(2, stub_request_with_entry(n), 1)], + snapshot_targets: vec![], + }) + }); + + let open_count = Arc::new(AtomicUsize::new(0)); + let open_count_clone = Arc::clone(&open_count); + let (break_tx, break_rx) = + mpsc::unbounded_channel::>(); + // `returning` takes an `FnMut`, so it must be able to run more than once β€” + // wrap the receiver so only the first (successful) call can move it out. + let break_rx = std::sync::Mutex::new(Some(break_rx)); + + let mut transport = MockTransport::::new(); + transport.expect_open_replication_stream().returning(move |_, _, _, _| { + let n = open_count_clone.fetch_add(1, Ordering::SeqCst) + 1; + if n == 1 { + let (req_tx, mut req_rx) = mpsc::channel::(128); + // Drain requests so all 16 sends succeed β€” this test exercises the + // sample-reset-on-reconnect path, not a send failure. + tokio::spawn(async move { while req_rx.recv().await.is_some() {} }); + // Stays pending (no ACK, so the sample started on send #16 never + // resolves) until the test pushes the break signal below. + let rx = break_rx + .lock() + .unwrap() + .take() + .expect("stream should only open successfully once"); + let stream = UnboundedReceiverStream::new(rx).boxed(); + Ok(crate::ReplicationStream { + sender: req_tx, + receiver: stream, + }) + } else { + // Reconnect fails so the worker enters backoff instead of spinning. + Err(crate::NetworkError::ConnectError("peer unreachable".into()).into()) + } + }); + ctx.transport = Arc::new(transport); + + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 0); + raft_log.expect_flush().returning(|| Ok(())); + raft_log.expect_save_hard_state().returning(|_| Ok(())); + ctx.storage.raft_log = Arc::new(raft_log); + + let mut state = LeaderState::::new(1, ctx.node_config.clone()); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + + let (internal_event_tx, _internal_event_rx) = mpsc::unbounded_channel(); + + // 16 sends: the 16th lands on the sampling boundary and starts a sample + // that never resolves (mock stream stays pending, no ACK). A new peer + // starts in Probe state, which allows only one in-flight request at a + // time (leader_state.rs:3384) β€” since this test never sends a real ACK, + // that flag would never clear naturally, so it's reset by hand here to + // let all 16 attempts actually reach the worker. + for _ in 0..16 { + state.reset_in_flight(2); + state.process_batch(one_entry_batch(), &internal_event_tx, &ctx).await.unwrap(); + } + + // Give the worker time to process all 16 sends before breaking the stream. + tokio::time::sleep(std::time::Duration::from_millis(100)).await; + + // Break the stream mid-sample to force recv_handle.abort() + reconnect. + let _ = break_tx.send(Err(tonic::Status::internal("stream broken"))); + + // Give the worker time to detect the break, abort, and reconnect. + tokio::time::sleep(std::time::Duration::from_millis(200)).await; + + assert!( + logs_contain("RTT sample reset after reconnect"), + "reconnecting mid-sample must reset the interrupted RTT sample" + ); +} + +/// The queue capacity configured in `replication_send_queue_capacity` must be the value the +/// worker hands to the transport when it opens the stream β€” otherwise the knob is decorative. +#[tokio::test] +#[traced_test] +async fn test_worker_opens_stream_with_configured_send_queue_capacity() { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut ctx = mock_raft_context( + "/tmp/test_worker_opens_stream_with_configured_capacity", + graceful_rx, + None, + ); + ctx.membership = Arc::new(two_peer_membership()); + + let mut cfg = (*ctx.node_config).clone(); + cfg.raft.replication.replication_send_queue_capacity = 777; + ctx.node_config = Arc::new(cfg); + + ctx.handlers + .replication_handler + .expect_prepare_batch_requests() + .times(1) + .returning(|_, _, _, _, _, _| { + Ok(crate::PrepareResult { + append_requests: vec![(2, stub_request(), 1)], + snapshot_targets: vec![], + }) + }); + + let seen_capacity = Arc::new(AtomicUsize::new(0)); + let seen = seen_capacity.clone(); + let mut transport = MockTransport::::new(); + transport.expect_open_replication_stream().returning(move |_, _, _, capacity| { + seen.store(capacity, Ordering::SeqCst); + let (req_tx, mut req_rx) = tokio::sync::mpsc::channel(capacity); + let (resp_tx, resp_rx) = tokio::sync::mpsc::channel(8); + tokio::spawn(async move { + while req_rx.recv().await.is_some() { + let _ = resp_tx.send(Ok(stub_response())).await; + } + }); + Ok(crate::ReplicationStream { + sender: req_tx, + receiver: tokio_stream::wrappers::ReceiverStream::new(resp_rx).boxed(), + }) + }); + ctx.transport = Arc::new(transport); + + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 0); + raft_log.expect_flush().returning(|| Ok(())); + raft_log.expect_save_hard_state().returning(|_| Ok(())); + ctx.storage.raft_log = Arc::new(raft_log); + + let mut state = LeaderState::::new(1, ctx.node_config.clone()); + state.init_cluster_metadata(&ctx.membership).await.unwrap(); + + let (internal_event_tx, _rx) = mpsc::unbounded_channel(); + state.process_batch(one_entry_batch(), &internal_event_tx, &ctx).await.unwrap(); + + let deadline = tokio::time::Instant::now() + std::time::Duration::from_secs(2); + while seen_capacity.load(Ordering::SeqCst) == 0 && tokio::time::Instant::now() < deadline { + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + } + assert_eq!( + seen_capacity.load(Ordering::SeqCst), + 777, + "the worker must pass replication_send_queue_capacity to open_replication_stream" + ); +} diff --git a/d-engine-core/src/raft_role/learner_state.rs b/d-engine-core/src/raft_role/learner_state.rs index 01c735b1..7d5dcaf3 100644 --- a/d-engine-core/src/raft_role/learner_state.rs +++ b/d-engine-core/src/raft_role/learner_state.rs @@ -22,6 +22,7 @@ use crate::alias::MOF; use crate::cluster_printer::print_learner_join_success; use crate::cluster_printer::print_learner_promoted_to_voter; use crate::cluster_printer::print_role_transition_line; +use crate::role_state::PendingAcks; use crate::role_state::schedule_and_execute_purge; use async_trait::async_trait; use d_engine_proto::common::LogId; @@ -35,6 +36,7 @@ use d_engine_proto::server::cluster::LeaderDiscoveryResponse; use d_engine_proto::server::election::VoteResponse; use d_engine_proto::server::election::VotedFor; use d_engine_proto::server::storage::SnapshotMetadata; +use std::collections::BTreeMap; use std::fmt::Debug; use std::marker::PhantomData; use std::sync::Arc; @@ -88,6 +90,10 @@ pub struct LearnerState { /// reflected in the latest snapshot. pub last_purged_index: Option, + /// AppendEntries responses withheld pending this node's own durable_index. + /// See `role_state::PendingAcks`. + pending_append_acks: PendingAcks, + // -- Snapshot Management -- /// Prevents concurrent snapshot creation /// @@ -514,6 +520,10 @@ impl RaftRoleState for LearnerState { fn pending_purge_upto_mut(&mut self) -> Option<&mut Option> { Some(&mut self.pending_purge_upto) } + + fn pending_append_acks_mut(&mut self) -> Option<&mut PendingAcks> { + Some(&mut self.pending_append_acks) + } } impl LearnerState { @@ -537,6 +547,7 @@ impl LearnerState { shared_state: SharedState::new(node_id, None, None), last_purged_index: None, snapshot_in_progress: AtomicBool::new(false), + pending_append_acks: BTreeMap::new(), node_config, _marker: PhantomData, pending_purge_upto: None, @@ -646,6 +657,7 @@ impl From<&FollowerState> for LearnerState { shared_state: follower_state.shared_state.clone(), node_config: follower_state.node_config.clone(), snapshot_in_progress: AtomicBool::new(false), + pending_append_acks: BTreeMap::new(), last_purged_index: follower_state.last_purged_index, pending_purge_upto: follower_state.pending_purge_upto, _marker: PhantomData, @@ -658,6 +670,7 @@ impl From<&CandidateState> for LearnerState { shared_state: candidate_state.shared_state.clone(), node_config: candidate_state.node_config.clone(), snapshot_in_progress: AtomicBool::new(false), + pending_append_acks: BTreeMap::new(), last_purged_index: candidate_state.last_purged_index, pending_purge_upto: None, _marker: PhantomData, diff --git a/d-engine-core/src/raft_role/learner_state_test.rs b/d-engine-core/src/raft_role/learner_state_test.rs index 7eaf7bc7..07d32edb 100644 --- a/d-engine-core/src/raft_role/learner_state_test.rs +++ b/d-engine-core/src/raft_role/learner_state_test.rs @@ -363,6 +363,16 @@ async fn test_learner_handles_append_entries_success() { assert_eq!(state.current_term(), leader_term); assert_eq!(state.commit_index(), expected_commit); + // RPO=0 (#446): the success ACK is withheld until durable_index reaches the claimed index. + assert!( + resp_rx.try_recv().is_err(), + "ACK must be withheld until durable_index catches up" + ); + + // Simulate LogFlushed: durable_index advances to 1, releasing the withheld ACK. + let (flush_tx, _flush_rx) = mpsc::unbounded_channel(); + state.handle_log_flushed(1, &context, &flush_tx).await; + let response = resp_rx.recv().await.unwrap().unwrap(); assert!(response.is_success()); } @@ -1726,63 +1736,6 @@ async fn test_apply_completed_respects_snapshot_disabled_config() { ); } -// ============================================================================ -// MemFirst ACK Tests -// ============================================================================ - -/// Learner ACKs leader immediately after memory write (MemFirst). -#[tokio::test] -async fn test_learner_acks_immediately_after_memory_write() { - let (_graceful_tx, graceful_rx) = watch::channel(()); - let (mut context, _temp_dir) = mock_raft_context_with_temp(graceful_rx, None); - - let leader_term = 2u64; - let appended_index = 5u64; - - let mut replication_handler = crate::MockReplicationCore::new(); - replication_handler.expect_handle_append_entries().returning(move |_, _, _| { - Ok(crate::AppendResponseWithUpdates { - response: d_engine_proto::server::replication::AppendEntriesResponse::success( - 1, - leader_term, - Some(LogId { - term: leader_term, - index: appended_index, - }), - ), - commit_index_update: None, - }) - }); - context.handlers.replication_handler = replication_handler; - context.membership = Arc::new(MockMembership::new()); - - let mut state = LearnerState::::new(1, context.node_config.clone()); - state.update_current_term(leader_term); - - let append_request = d_engine_proto::server::replication::AppendEntriesRequest { - term: leader_term, - leader_id: 2, - prev_log_index: 0, - prev_log_term: 0, - entries: vec![], - leader_commit_index: 0, - }; - let (resp_tx, mut resp_rx) = MaybeCloneOneshot::new(); - let inbound_event = InboundEvent::AppendEntries(append_request, vec![resp_tx]); - let (internal_event_tx, _internal_event_rx) = mpsc::unbounded_channel(); - - assert!( - state - .handle_inbound_event(inbound_event, &context, internal_event_tx) - .await - .is_ok() - ); - - // MemFirst: ACK sent immediately - let response = resp_rx.try_recv().expect("ACK must be sent immediately after memory write"); - assert!(response.unwrap().is_success()); -} - /// Spawns a fake Worker that answers exactly one `InstallSnapshot` command with `result`, /// then exits. Returns a `StateMachineCommandSender` wired to it β€” stands in for the real /// `StateMachineWorker`, which isn't running in these role-layer unit tests. diff --git a/d-engine-core/src/raft_role/mod.rs b/d-engine-core/src/raft_role/mod.rs index b7737611..257d22e6 100644 --- a/d-engine-core/src/raft_role/mod.rs +++ b/d-engine-core/src/raft_role/mod.rs @@ -16,6 +16,8 @@ mod follower_state_test; #[cfg(test)] mod learner_state_test; #[cfg(test)] +mod pending_ack_test; +#[cfg(test)] mod role_state_test; use std::collections::HashMap; @@ -48,7 +50,7 @@ use super::InternalEvent; use super::RaftContext; use crate::Result; use crate::TypeConfig; -use crate::role_state::PeerReplicationState; +use crate::role_state::PendingAcks; /// The role state focuses solely on its own logic /// and does not directly manipulate the underlying storage or network. @@ -374,6 +376,34 @@ impl RaftRole { pub(crate) fn become_learner(&self) -> Result> { self.state().become_learner() } + /// Move the withheld-ACK queue out of the current role before a transition. + /// Only Follower and Learner keep one; every other role yields an empty map. + /// + /// A withheld ACK describes this node's durable log, not its role. Dropping it + /// on a `Learner -> Follower` promotion would strand the leader waiting on a + /// response that never arrives (#446). + pub(crate) fn take_pending_acks(&mut self) -> PendingAcks { + self.state_mut() + .pending_append_acks_mut() + .map(std::mem::take) + .unwrap_or_default() + } + + /// Install a carried withheld-ACK queue into the role a transition produced. + /// Follower and Learner adopt it; any other role cannot hold it, so its + /// entries are failed with a conflict response. + pub(crate) fn restore_pending_acks( + &mut self, + acks: PendingAcks, + ) { + let node_id = self.state().node_id(); + let current_term = self.state().current_term(); + match self.state_mut().pending_append_acks_mut() { + Some(queue) => *queue = acks, + None => role_state::reject_pending_acks(acks, node_id, current_term), + } + } + pub fn current_term(&self) -> u64 { self.state().current_term() } @@ -386,28 +416,6 @@ impl RaftRole { self.state_mut().init_peers_next_index_and_match_index(last_entry_id, peer_ids) } - /// Reset `next_index[peer] = match_index[peer] + 1` after a bidi stream disconnect. - /// Ensures the next heartbeat re-sends any unACKed in-flight entries. - pub(crate) fn handle_peer_stream_error( - &mut self, - peer_id: u32, - ) { - // The bidi stream only carries AppendEntries. While this peer is in Snapshot - // state, an error on this stream says nothing about the independent - // connection the snapshot transfer runs on, so it has no authority to act - // (mirrors etcd raft.go MsgUnreachable: only BecomeProbe() when StateReplicate). - if self.state().peer_replication_state(peer_id) == PeerReplicationState::Snapshot { - return; - } - let match_idx = self.state().match_index(peer_id).unwrap_or(0); - let _ = self.state_mut().update_next_index(peer_id, match_idx + 1); - - // #436: stream is down, we don't know what (if anything) the peer received β€” - // stop trusting speculative advance (etcd: BecomeProbe on MsgUnreachable). - self.state_mut() - .set_peer_replication_state(peer_id, PeerReplicationState::Probe); - } - pub(crate) async fn handle_zombie_detected( &mut self, node_id: u32, diff --git a/d-engine-core/src/raft_role/pending_ack_test.rs b/d-engine-core/src/raft_role/pending_ack_test.rs new file mode 100644 index 00000000..c0f4ce48 --- /dev/null +++ b/d-engine-core/src/raft_role/pending_ack_test.rs @@ -0,0 +1,223 @@ +//! Tests for #446's withheld-AppendEntries-ACK primitives: +//! +//! - `RaftRoleState::resolve_pending_acks` β€” the release / reject decision; +//! - `RaftRole::{take,restore}_pending_acks` β€” carrying the queue across a role +//! transition instead of dropping it. +//! +//! These drive bare role structs (no `RaftContext`, no mocks, no fs), so they +//! stay cheap to run and cheap to change. + +use std::sync::Arc; + +use d_engine_proto::common::LogId; +use d_engine_proto::server::replication::AppendEntriesResponse; +use d_engine_proto::server::replication::SuccessResult; +use d_engine_proto::server::replication::append_entries_response; +use tonic::Status; + +use super::RaftRole; +use super::candidate_state::CandidateState; +use super::follower_state::FollowerState; +use super::learner_state::LearnerState; +use crate::MaybeCloneOneshot; +use crate::MaybeCloneOneshotReceiver; +use crate::RaftNodeConfig; +use crate::RaftOneshot; +use crate::raft_role::role_state::PendingAck; +use crate::raft_role::role_state::RaftRoleState; +use crate::test_utils::mock::MockTypeConfig; + +type Rx = MaybeCloneOneshotReceiver>; + +fn config() -> Arc { + Arc::new( + RaftNodeConfig::new() + .expect("RaftNodeConfig::new") + .validate() + .expect("RaftNodeConfig::validate"), + ) +} + +fn follower(term: u64) -> RaftRole { + let mut s = FollowerState::new(1, config(), None, None); + s.shared_state_mut().update_current_term(term); + RaftRole::Follower(Box::new(s)) +} + +fn learner(term: u64) -> RaftRole { + let mut s = LearnerState::new(1, config()); + s.shared_state_mut().update_current_term(term); + RaftRole::Learner(Box::new(s)) +} + +fn candidate() -> RaftRole { + RaftRole::Candidate(Box::new(CandidateState::new(1, config()))) +} + +/// Put a withheld success ACK for `index` straight into `role`'s queue, bypassing +/// the AppendEntries workflow. Returns the receiver a caller would be blocked on. +fn withhold( + role: &mut RaftRole, + index: u64, + claimed_term: u64, + term_when_withheld: u64, +) -> Rx { + let (tx, rx) = MaybeCloneOneshot::new(); + role.state_mut() + .pending_append_acks_mut() + .expect("role keeps a pending-ack queue") + .insert( + (term_when_withheld, index), + PendingAck { + claimed_term, + withheld_at: std::time::Instant::now(), + senders: vec![tx], + }, + ); + rx +} + +fn queue_len(role: &mut RaftRole) -> usize { + role.state_mut().pending_append_acks_mut().map_or(0, |q| q.len()) +} + +// -- resolve_pending_acks ----------------------------------------------------- + +/// A withheld ACK is released as a success once its claimed index is durable and +/// the node is still on the term it withheld under. +#[test] +fn test_resolve_releases_success_when_durable() { + let mut role = follower(5); + let mut rx = withhold(&mut role, 8, 5, 5); + + role.state_mut().resolve_pending_acks(8); + + let resp = rx.try_recv().expect("released").unwrap(); + assert!(matches!( + resp.result, + Some(append_entries_response::Result::Success(SuccessResult { + last_match: Some(LogId { index: 8, term: 5 }), + })), + )); + assert_eq!(queue_len(&mut role), 0); +} + +/// A withheld ACK stays queued while its claimed index is still beyond durable. +#[test] +fn test_resolve_keeps_waiting_until_durable() { + let mut role = follower(5); + let mut rx = withhold(&mut role, 8, 5, 5); + + role.state_mut().resolve_pending_acks(7); + + assert!(rx.try_recv().is_err()); + assert_eq!(queue_len(&mut role), 1); +} + +/// A withheld ACK is rejected with a conflict if the node moved to a newer term +/// since withholding: a higher-term leader may have overwritten the log at that +/// index, so the durability claim can no longer be trusted (#446). +#[test] +fn test_resolve_rejects_stale_term_ack() { + let mut role = follower(6); // node is now on term 6 + let mut rx = withhold(&mut role, 8, 5, 5); // ACK was withheld under term 5 + + role.state_mut().resolve_pending_acks(8); + + let resp = rx.try_recv().expect("resolved").unwrap(); + assert!( + !resp.is_success(), + "stale-term ACK must resolve to a conflict" + ); + assert_eq!(queue_len(&mut role), 0); +} + +/// Every sender queued on one index is answered β€” a leader retry or a heartbeat +/// can leave more than one waiter on the same index. +#[test] +fn test_resolve_answers_every_sender_on_an_index() { + let mut role = follower(5); + let (tx1, mut rx1) = MaybeCloneOneshot::new(); + let (tx2, mut rx2) = MaybeCloneOneshot::new(); + role.state_mut().pending_append_acks_mut().unwrap().insert( + (5, 8), + PendingAck { + claimed_term: 5, + withheld_at: std::time::Instant::now(), + senders: vec![tx1, tx2], + }, + ); + + role.state_mut().resolve_pending_acks(8); + + assert!(rx1.try_recv().unwrap().unwrap().is_success()); + assert!(rx2.try_recv().unwrap().unwrap().is_success()); +} + +/// `resolve_pending_acks` on a role that keeps no queue (Candidate) is a no-op. +#[test] +fn test_resolve_is_noop_without_a_queue() { + let mut role = candidate(); + role.state_mut().resolve_pending_acks(10); // must not panic +} + +// -- take / restore across a role transition --------------------------------- + +/// A withheld ACK survives a Learner -> Follower promotion: `take_pending_acks` +/// moves the queue out of the old role and `restore_pending_acks` installs it in +/// the new one. Dropping it here would strand the leader on a response that never +/// arrives β€” the bug #446 fixed. +#[test] +fn test_pending_ack_survives_learner_promotion() { + let mut old = learner(5); + let mut rx = withhold(&mut old, 8, 5, 5); + + let carried = old.take_pending_acks(); + assert_eq!(carried.len(), 1); + assert_eq!( + queue_len(&mut old), + 0, + "take must move the queue, not copy it" + ); + + let mut new = follower(5); + new.restore_pending_acks(carried); + + new.state_mut().resolve_pending_acks(8); + assert!(rx.try_recv().expect("released after promotion").unwrap().is_success()); +} + +/// The symmetric Follower -> Learner demotion also carries the queue. +#[test] +fn test_pending_ack_survives_follower_demotion() { + let mut old = follower(5); + let mut rx = withhold(&mut old, 8, 5, 5); + + let carried = old.take_pending_acks(); + let mut new = learner(5); + new.restore_pending_acks(carried); + + new.state_mut().resolve_pending_acks(8); + assert!(rx.try_recv().expect("released after demotion").unwrap().is_success()); +} + +/// Restoring the queue into a role that cannot hold one (Candidate) fails every +/// withheld ACK with a conflict, so the leader retries rather than timing out. +#[test] +fn test_restore_into_candidate_fails_pending_acks() { + let mut old = follower(5); + let mut rx = withhold(&mut old, 8, 5, 5); + let carried = old.take_pending_acks(); + + let mut candidate = candidate(); + candidate.restore_pending_acks(carried); + + let resp = rx.try_recv().expect("failed, not dropped").unwrap(); + assert!(!resp.is_success()); +} + +/// `take_pending_acks` on a role with no queue yields an empty map, never panics. +#[test] +fn test_take_from_queueless_role_is_empty() { + assert!(candidate().take_pending_acks().is_empty()); +} diff --git a/d-engine-core/src/raft_role/role_state.rs b/d-engine-core/src/raft_role/role_state.rs index 89c6181e..7fcc1d3e 100644 --- a/d-engine-core/src/raft_role/role_state.rs +++ b/d-engine-core/src/raft_role/role_state.rs @@ -31,10 +31,13 @@ use d_engine_proto::common::LogId; use d_engine_proto::server::election::VotedFor; use d_engine_proto::server::replication::AppendEntriesRequest; use d_engine_proto::server::replication::AppendEntriesResponse; +use d_engine_proto::server::replication::SuccessResult; +use d_engine_proto::server::replication::append_entries_response; use d_engine_proto::server::storage::SnapshotAck; use d_engine_proto::server::storage::SnapshotChunk; use d_engine_proto::server::storage::SnapshotMetadata; use d_engine_proto::server::storage::SnapshotResponse; +use std::collections::BTreeMap; use std::sync::atomic::{AtomicBool, Ordering}; use tokio::sync::mpsc; use tokio::time::Instant; @@ -56,6 +59,80 @@ pub(crate) enum PeerReplicationState { Snapshot, } +/// A success `AppendEntriesResponse` that has been computed but not yet sent, +/// because this node's own `durable_index` had not reached the index the response +/// claims. Held until fsync catches up, so an ACK never asserts durability the +/// node cannot yet guarantee (RPO=0, #446). +/// +/// `senders` accumulates when more than one request from the same leader term +/// claims the same index (a leader retry, or a heartbeat landing on the tail). +/// +/// The response body is not stored: it is rebuilt on release, after re-checking +/// the term it was withheld under. A response frozen under a term the node has +/// since left must never be sent. +pub(crate) struct PendingAck { + pub(crate) claimed_term: u64, + /// When this ACK was first withheld β€” diffed against release time to measure + /// how long the follower sat waiting for its own fsync before it could reply. + pub(crate) withheld_at: std::time::Instant, + pub(crate) senders: + Vec>>, +} + +/// Withheld ACKs keyed by `(term withheld under, claimed index)`. +/// +/// The term is part of the key so an entry left by a previous leader can never +/// absorb a request from a newer one: they land in different slots, and the +/// stale one is failed on the next release. +pub(crate) type PendingAcks = BTreeMap<(u64, u64), PendingAck>; + +/// Send the terminal response for one withheld ACK β€” a rebuilt success if +/// `confirm`, a conflict otherwise β€” to every accumulated sender. +fn resolve_pending_ack( + node_id: u32, + index: u64, + ack: PendingAck, + confirm: bool, + current_term: u64, +) { + metrics::histogram!( + "core.raft.follower.ack_withhold_ms", + "outcome" => if confirm { "confirmed" } else { "rejected" } + ) + .record(ack.withheld_at.elapsed().as_secs_f64() * 1_000.0); + let response = if confirm { + AppendEntriesResponse::success( + node_id, + current_term, + Some(LogId { + index, + term: ack.claimed_term, + }), + ) + } else { + AppendEntriesResponse::conflict(node_id, current_term, None, None) + }; + for sender in ack.senders { + if let Err(e) = sender.send(Ok(response)) { + error!("withheld AppendEntries ACK (index {index}): send failed: {e:?}"); + } + } +} + +/// Fail every withheld ACK with a conflict response. Used when the queue passes to +/// a role that cannot hold it (Candidate or Leader): the node no longer recognises +/// the leader those ACKs were owed to, so that leader's replication worker should +/// retry now rather than wait out an RPC timeout. +pub(crate) fn reject_pending_acks( + acks: PendingAcks, + node_id: u32, + current_term: u64, +) { + for ((_, index), ack) in acks { + resolve_pending_ack(node_id, index, ack, false, current_term); + } +} + #[async_trait] pub(crate) trait RaftRoleState: Send + Sync + 'static { type T: TypeConfig; @@ -397,16 +474,60 @@ pub(crate) trait RaftRoleState: Send + Sync + 'static { Ok(()) } - /// Handle LogFlushed(durable) event: entries up to `durable` are now crash-safe. - /// Leader: recalculates commit_index (uses durable_index in quorum calculation). - /// Default: no-op for Candidate/Follower/Learner (ACK already sent on memory write). + /// Release withheld AppendEntries ACKs now that the log is durable through + /// `durable`. For each withheld ACK: + /// + /// - withheld under a term this node has since left β†’ fail it with a conflict. + /// A higher-term leader may have overwritten the log at that index; within a + /// single term a follower's entries are never replaced, so the term check + /// alone is a sufficient content guard and no log lookup is needed. + /// - claimed index now `<= durable` β†’ rebuild and send the success. + /// - otherwise β†’ keep waiting. + /// + /// No-op for Candidate/Leader (no queue). Runs on every fsync completion, so it + /// stays limited to integer comparisons β€” no log lookup. (#446) + fn resolve_pending_acks( + &mut self, + durable: u64, + ) { + let node_id = self.node_id(); + let current_term = self.current_term(); + let Some(pending) = self.pending_append_acks_mut() else { + return; + }; + if pending.is_empty() { + return; + } + let resolved: Vec<((u64, u64), bool)> = pending + .iter() + .filter_map(|(&(term, index), _)| { + if term != current_term { + Some(((term, index), false)) // stale term -> conflict + } else if index <= durable { + Some(((term, index), true)) // durable -> success + } else { + None // keep waiting + } + }) + .collect(); + for (key, confirm) in resolved { + if let Some(ack) = pending.remove(&key) { + resolve_pending_ack(node_id, key.1, ack, confirm, current_term); + } + } + } + + /// A batch of log entries reached `durable` on disk (fsync complete). + /// + /// Follower/Learner: release any withheld AppendEntries ACKs this now covers. + /// Leader: overridden to recalculate `commit_index`. Candidate: no-op. async fn handle_log_flushed( &mut self, - _durable: u64, + durable: u64, _ctx: &RaftContext, _internal_event_tx: &mpsc::UnboundedSender, ) { - // Candidate: no-op + self.resolve_pending_acks(durable); } /// Handle AppendEntries result from a per-follower ReplicationWorker. @@ -534,10 +655,18 @@ pub(crate) trait RaftRoleState: Send + Sync + 'static { // My term might be updated, has to fetch it again let my_term = self.current_term(); + // `state_snapshot` was captured at the top of `handle_inbound_event`, before + // `commit_hard_state` above may have advanced our term. Patch it here so the + // AppendEntriesResponse reports the real, just-updated term β€” not the stale + // snapshot. + let state_snapshot = StateSnapshot { + current_term: my_term, + ..state_snapshot.clone() + }; // Handle replication request match ctx .replication_handler() - .handle_append_entries(append_entries_request, state_snapshot, ctx.raft_log()) + .handle_append_entries(append_entries_request, &state_snapshot, ctx.raft_log()) .await { Ok(AppendResponseWithUpdates { @@ -560,13 +689,53 @@ pub(crate) trait RaftRoleState: Send + Sync + 'static { } debug!("AppendEntriesResponse: {:?}", response); - // MemFirst: ACK immediately after memory write. IO thread fsyncs async. - // Safety: quorum uses last_entry_id (in-memory); crash safety is guaranteed by - // majority replication, not per-follower durability. + // RPO=0 (#446): a success response asserts the claimed entry is + // fsync-durable on this node. If this node's own `durable_index` + // has not reached that index, withhold the response until it does + // (released by `resolve_pending_acks`). Conflict and higher-term + // responses assert nothing about durability and are sent at once. + let claim = match &response.result { + Some(append_entries_response::Result::Success(SuccessResult { + last_match: Some(log_id), + })) => Some((log_id.index, log_id.term)), + _ => None, + }; - for sender in senders { - if let Err(e) = sender.send(Ok(response)) { - error!("Failed to send: {:?}", e); + match claim { + Some((index, claimed_term)) if ctx.storage.raft_log.durable_index() < index => { + let term_when_withheld = self.current_term(); + match self.pending_append_acks_mut() { + Some(pending) => { + pending + .entry((term_when_withheld, index)) + .or_insert_with(|| PendingAck { + claimed_term, + withheld_at: std::time::Instant::now(), + senders: Vec::new(), + }) + .senders + .extend(senders); + } + None => { + // Only Follower and Learner produce a success + // response here, and both carry the queue. Reaching + // this arm means a role invariant broke β€” send the + // ACK now rather than strand the leader. + error!( + "withheld a success ACK on a role with no pending-ACK queue" + ); + for sender in senders { + let _ = sender.send(Ok(response)); + } + } + } + } + _ => { + for sender in senders { + if let Err(e) = sender.send(Ok(response)) { + error!("failed to send AppendEntries response: {e:?}"); + } + } } } } @@ -926,19 +1095,11 @@ pub(crate) trait RaftRoleState: Send + Sync + 'static { .into()) } - fn peer_replication_state( - &self, - _node_id: u32, - ) -> PeerReplicationState { - // Default: unknown peer, be conservative. Also the default for non-leader roles. - PeerReplicationState::Probe - } - - fn set_peer_replication_state( - &mut self, - _node_id: u32, - _state: PeerReplicationState, - ) { + /// The withheld-ACK queue, for the roles that keep one (Follower, Learner). + /// `None` for Candidate and Leader. Carried across a Follower<->Learner + /// transition by `RaftRole::take_pending_acks` / `restore_pending_acks` (#446). + fn pending_append_acks_mut(&mut self) -> Option<&mut PendingAcks> { + None } } @@ -989,7 +1150,12 @@ pub(super) async fn schedule_and_execute_purge( // can still catch up via AppendEntries instead of InstallSnapshot. let retained = ctx.node_config().raft.snapshot.retained_log_entries; let purge_upto_index = last_included.index.saturating_sub(retained); - info!("purge_upto_index={purge_upto_index}"); + + info!( + "node_id={} purge_upto_index={purge_upto_index}", + ctx.node_id + ); + // retained >= last_included.index β†’ nothing to purge; skip log lookup entirely. if purge_upto_index == 0 { return Ok(()); @@ -1018,7 +1184,7 @@ pub(super) async fn schedule_and_execute_purge( } /// Cleanup after an InstallSnapshotChunk install (or a confirmed no-op at/beyond -/// `target`). Shares `pending_purge_upto` with pathβ‘ 's watermark β€” failure isn't cleared, +/// `target`). Shares `pending_purge_upto` with `schedule_and_execute_purge`'s watermark β€” failure isn't cleared, /// retried by whichever purge trigger runs next. No `retained_log_entries` subtraction: /// entries below `target` are redundant with a snapshot we already have, not held back for /// a lagging peer. diff --git a/d-engine-core/src/raft_test/raft_comprehensive_tests.rs b/d-engine-core/src/raft_test/raft_comprehensive_tests.rs index 6f169270..ff1962dd 100644 --- a/d-engine-core/src/raft_test/raft_comprehensive_tests.rs +++ b/d-engine-core/src/raft_test/raft_comprehensive_tests.rs @@ -83,12 +83,12 @@ fn prepare_succeed_majority_confirmation() -> ( // last_entry_id() / durable_index() return the post-write value. This is // required for the single-voter inline-flush path in verify_internal_quorum // to advance past the noop entry and fire the commit signal. - replication_handler - .expect_prepare_batch_requests() - .returning(move |payloads, _, _, _, _| { + replication_handler.expect_prepare_batch_requests().returning( + move |payloads, _, _, _, _, _| { li_prepare.fetch_add(payloads.len() as u64, Ordering::Relaxed); Ok(crate::PrepareResult::default()) - }); + }, + ); (raft_log, replication_handler) } @@ -681,7 +681,7 @@ async fn test_leader_verification_fails_downgrades() { let mut replication_handler = crate::MockReplicationCore::new(); replication_handler .expect_prepare_batch_requests() - .returning(|_, _, _, _, _| Err(crate::Error::Fatal("Verification failed".to_string()))); + .returning(|_, _, _, _, _, _| Err(crate::Error::Fatal("Verification failed".to_string()))); let mut raft_log = crate::MockRaftLog::new(); raft_log.expect_last_entry_id().returning(|| 11); @@ -936,9 +936,11 @@ async fn test_leader_ready_notification_suppressed_when_noop_fails() { // prepare_batch_requests returning Err simulates noop failure: // verify_internal_quorum returns Err β†’ BecomeFollower queued β†’ notify_leader_change(Some) never sent. let mut replication_handler = crate::MockReplicationCore::new(); - replication_handler.expect_prepare_batch_requests().returning(|_, _, _, _, _| { - Err(crate::Error::Fatal("Noop verification failed".to_string())) - }); + replication_handler + .expect_prepare_batch_requests() + .returning(|_, _, _, _, _, _| { + Err(crate::Error::Fatal("Noop verification failed".to_string())) + }); raft.ctx.storage.raft_log = Arc::new(raft_log); raft.ctx.handlers.replication_handler = replication_handler; @@ -1437,9 +1439,10 @@ async fn test_snapshot_push_completed_uses_snapshot_boundary_not_leader_tip() { let current_term = raft.current_term(); // Establish an active snapshot transfer β€” the handler seeds next_index only for a // peer that is actually mid-snapshot. - raft.role - .state_mut() - .set_peer_replication_state(peer_id, crate::role_state::PeerReplicationState::Snapshot); + let crate::RaftRole::Leader(leader) = &mut raft.role else { + panic!("expected Leader role after BecomeLeader"); + }; + leader.set_peer_replication_state(peer_id, crate::role_state::PeerReplicationState::Snapshot); raft.handle_internal_event(InternalEvent::SnapshotPushCompleted { peer_id, success: true, @@ -1660,17 +1663,24 @@ async fn test_peer_stream_error_does_not_touch_peer_in_snapshot_state() { raft.handle_internal_event(InternalEvent::BecomeLeader).await.unwrap(); let peer_id = 42; - raft.role - .state_mut() - .set_peer_replication_state(peer_id, crate::role_state::PeerReplicationState::Snapshot); + { + let crate::RaftRole::Leader(leader) = &mut raft.role else { + panic!("expected Leader role after BecomeLeader"); + }; + leader + .set_peer_replication_state(peer_id, crate::role_state::PeerReplicationState::Snapshot); + } let next_index_before = raft.role.state().next_index(peer_id); raft.handle_internal_event(InternalEvent::PeerStreamError { peer_id }) .await .unwrap(); + let crate::RaftRole::Leader(leader) = &raft.role else { + panic!("expected Leader role after BecomeLeader"); + }; assert_eq!( - raft.role.state().peer_replication_state(peer_id), + leader.peer_replication_state(peer_id), crate::role_state::PeerReplicationState::Snapshot, "a bidi stream error must not downgrade a peer that is mid-snapshot-transfer" ); @@ -1696,16 +1706,25 @@ async fn test_peer_stream_error_downgrades_non_snapshot_peer_to_probe() { raft.handle_internal_event(InternalEvent::BecomeLeader).await.unwrap(); let peer_id = 42; - raft.role - .state_mut() - .set_peer_replication_state(peer_id, crate::role_state::PeerReplicationState::Replicate); + { + let crate::RaftRole::Leader(leader) = &mut raft.role else { + panic!("expected Leader role after BecomeLeader"); + }; + leader.set_peer_replication_state( + peer_id, + crate::role_state::PeerReplicationState::Replicate, + ); + } raft.handle_internal_event(InternalEvent::PeerStreamError { peer_id }) .await .unwrap(); + let crate::RaftRole::Leader(leader) = &raft.role else { + panic!("expected Leader role after BecomeLeader"); + }; assert_eq!( - raft.role.state().peer_replication_state(peer_id), + leader.peer_replication_state(peer_id), crate::role_state::PeerReplicationState::Probe, "a bidi stream error for a non-snapshotting peer must still downgrade it to Probe" ); @@ -1936,7 +1955,7 @@ async fn test_leadership_verification_failure_downgrades() { let mut replication_handler = crate::MockReplicationCore::new(); replication_handler .expect_prepare_batch_requests() - .returning(|_, _, _, _, _| Err(crate::Error::Fatal("Majority timeout".to_string()))); + .returning(|_, _, _, _, _, _| Err(crate::Error::Fatal("Majority timeout".to_string()))); let mut raft_log = crate::MockRaftLog::new(); raft_log.expect_last_entry_id().returning(|| 11); @@ -2039,7 +2058,7 @@ async fn test_network_partition_minority_loses_leadership() { let mut replication_handler = crate::MockReplicationCore::new(); replication_handler .expect_prepare_batch_requests() - .returning(|_, _, _, _, _| Err(crate::Error::Fatal("Partition".to_string()))); + .returning(|_, _, _, _, _, _| Err(crate::Error::Fatal("Partition".to_string()))); let mut raft_log = crate::MockRaftLog::new(); raft_log.expect_last_entry_id().returning(|| 11); @@ -2675,3 +2694,48 @@ async fn test_graceful_shutdown_persists_hardstate() { "Expected clean exit, but got {result:?}" ); } + +/// Test: a node that loses leadership and is elected again must not carry any per-peer +/// replication state from its previous term. Both the trust state and the in-flight +/// bookkeeping belong to one leadership; reusing them would let stale in-flight slots gate +/// the new term's first sends. +/// +/// # Scenario +/// - Node becomes leader; peer 42 is promoted to `Replicate` (a real ACK was seen). +/// - Node steps down to follower, then wins again. +/// - Expected: peer 42 is back to the default `Probe` (nothing carried over). +#[tokio::test] +async fn test_re_elected_leader_starts_with_fresh_peer_state() { + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut raft = MockBuilder::new(graceful_rx).build_raft(); + let (raft_log, replication_core) = prepare_succeed_majority_confirmation(); + raft.ctx.storage.raft_log = Arc::new(raft_log); + raft.ctx.handlers.replication_handler = replication_core; + + raft.handle_internal_event(InternalEvent::BecomeCandidate).await.unwrap(); + raft.handle_internal_event(InternalEvent::BecomeLeader).await.unwrap(); + + let peer_id = 42; + { + let crate::RaftRole::Leader(leader) = &mut raft.role else { + panic!("expected Leader role after BecomeLeader"); + }; + leader.set_peer_replication_state( + peer_id, + crate::role_state::PeerReplicationState::Replicate, + ); + } + + raft.handle_internal_event(InternalEvent::BecomeFollower(None)).await.unwrap(); + raft.handle_internal_event(InternalEvent::BecomeCandidate).await.unwrap(); + raft.handle_internal_event(InternalEvent::BecomeLeader).await.unwrap(); + + let crate::RaftRole::Leader(leader) = &raft.role else { + panic!("expected Leader role after the second BecomeLeader"); + }; + assert_eq!( + leader.peer_replication_state(peer_id), + crate::role_state::PeerReplicationState::Probe, + "a new leadership must start every peer from the default Probe, not the previous term's state" + ); +} diff --git a/d-engine-core/src/replication/mod.rs b/d-engine-core/src/replication/mod.rs index dc19d0f3..c3c77e25 100644 --- a/d-engine-core/src/replication/mod.rs +++ b/d-engine-core/src/replication/mod.rs @@ -123,6 +123,13 @@ where /// The caller (leader Raft loop) then fires each request to a per-follower /// `ReplicationWorker` task and returns immediately β€” no blocking `.await`. /// + /// `peer_gating_decisions` marks which peers have a full in-flight window. + /// The caller computes this once, before calling here, and reuses the same + /// map again at dispatch time. Passed straight through to + /// `retrieve_to_be_synced_logs_for_peers`, which skips entry prep for gated + /// peers. One shared map, not two separate calculations β€” a peer gated here + /// but not at dispatch (or vice versa) would block that peer's heartbeat. + /// /// # Returns /// - `Ok(PrepareResult)` β€” append requests per in-range peer + snapshot targets. /// Empty `append_requests` when there are no replication targets or all need snapshots. @@ -134,6 +141,7 @@ where leader_state_snapshot: LeaderStateSnapshot, cluster_metadata: &crate::raft_role::ClusterMetadata, ctx: &crate::RaftContext, + peer_gating_decisions: &HashMap, ) -> Result; /// Handles successful AppendEntries responses @@ -163,10 +171,10 @@ where /// Determines follower commit index advancement /// /// Applies Leader's commit index according to: - /// - min(leader_commit, last_local_log_index) + /// - min(leader_commit, last_verified_log_index) fn if_update_commit_index_as_follower( my_commit_index: u64, - last_raft_log_id: u64, + last_verified_log_index: u64, leader_commit_index: u64, ) -> Option; @@ -181,6 +189,14 @@ where /// /// `first_index` is the leader's retained-log boundary (`raft_log.first_entry_id()`), /// passed in so the purge check happens before the range fetch, not after. + /// + /// `peer_gating_decisions[peer_id] == true` means that peer's in-flight + /// window is full: skip it here, insert nothing into the result map. The + /// caller's dispatch phase sees no entries for this peer and sends an + /// empty-entries heartbeat instead β€” this is how the window-full peer + /// still gets a liveness heartbeat without new data. This map is computed + /// once by the top-level caller, not here, so the same gating decision + /// also governs dispatch β€” see `prepare_batch_requests` doc above. fn retrieve_to_be_synced_logs_for_peers( &self, new_entries: &[Entry], @@ -189,6 +205,7 @@ where peer_next_indices: &HashMap, raft_log: &Arc>, first_index: u64, + peer_gating_decisions: &HashMap, ) -> HashMap; /// Handles an incoming AppendEntries RPC request (called by ALL ROLES) diff --git a/d-engine-core/src/replication/replication_handler.rs b/d-engine-core/src/replication/replication_handler.rs index a22fb6d0..2bf7ea09 100644 --- a/d-engine-core/src/replication/replication_handler.rs +++ b/d-engine-core/src/replication/replication_handler.rs @@ -1,10 +1,3 @@ -use std::cmp; -use std::collections::HashMap; -use std::collections::HashSet; -use std::fmt::Debug; -use std::marker::PhantomData; -use std::sync::Arc; - use async_trait::async_trait; use bytes::BytesMut; use d_engine_proto::client::WriteCommand; @@ -17,6 +10,12 @@ use d_engine_proto::server::replication::AppendEntriesResponse; use d_engine_proto::server::replication::ConflictResult; use d_engine_proto::server::replication::SuccessResult; use prost::Message; +use std::cmp; +use std::collections::HashMap; +use std::collections::HashSet; +use std::fmt::Debug; +use std::marker::PhantomData; +use std::sync::Arc; use tracing::debug; use tracing::error; use tracing::trace; @@ -68,6 +67,7 @@ where leader_state_snapshot: LeaderStateSnapshot, cluster_metadata: &crate::raft_role::ClusterMetadata, ctx: &crate::RaftContext, + peer_gating_decisions: &HashMap, ) -> Result { let replication_targets = &cluster_metadata.replication_targets; @@ -101,6 +101,7 @@ where &replication_data.peer_next_indices, raft_log, min_log_index, + peer_gating_decisions, ); let mut append_requests = Vec::with_capacity(replication_targets.len()); @@ -255,6 +256,7 @@ where peer_next_indices: &HashMap, raft_log: &Arc>, first_index: u64, + peer_gating_decisions: &HashMap, ) -> HashMap { let _timer = ScopedTimer::new("retrieve_to_be_synced_logs_for_peers"); @@ -270,6 +272,18 @@ where continue; } + // Phase 2 window gate: same peer_gating_decisions Phase 5 dispatch uses + // (computed once in execute_and_process_raft_rpc before Phase 1). If gated, + // skip entry preparation β€” Phase 5 falls back to an empty-entries heartbeat + // automatically, so liveness is never blocked by window pressure. + if peer_gating_decisions.get(&id).copied().unwrap_or(false) { + trace!( + "peer {} in-flight window full, skip entry preparation (will send heartbeat)", + id + ); + continue; + } + debug!("peer: {} next: {}", id, peer_next_id); let mut entries = Vec::new(); @@ -347,9 +361,14 @@ where self.my_id, request ); let current_term = state_snapshot.current_term; - let mut last_log_id_option = raft_log.last_log_id(); - //if there is no new entries need to insert, we just return the last local log index + // Only `prev_log_index` is verified here; the follower's own tail may differ. + let mut last_log_id_option = Some(LogId { + term: request.prev_log_term, + index: request.prev_log_index, + }); + + // With no new entries, the verified point is `prev_log_index` itself. let mut commit_index_update = None; let response = self.check_append_entries_request_is_legal(current_term, &request, raft_log); @@ -381,7 +400,7 @@ where if let Some(new_commit_index) = Self::if_update_commit_index_as_follower( state_snapshot.commit_index, - raft_log.last_entry_id(), + last_log_id_option.map(|id| id.index).unwrap_or(request.prev_log_index), request.leader_commit_index, ) { debug!("new commit index received: {:?}", new_commit_index); @@ -399,11 +418,10 @@ where }) } - ///If leaderCommit > commitIndex, set commitIndex = min(leaderCommit, index - /// of last new entry) + ///If leaderCommit > commitIndex, set commitIndex = min(leaderCommit, last_verified_log_index) fn if_update_commit_index_as_follower( my_commit_index: u64, - last_raft_log_id: u64, + last_verified_log_index: u64, leader_commit_index: u64, ) -> Option { debug!( @@ -414,7 +432,7 @@ where ); if leader_commit_index > my_commit_index { - return Some(cmp::min(leader_commit_index, last_raft_log_id)); + return Some(cmp::min(leader_commit_index, last_verified_log_index)); } None } diff --git a/d-engine-core/src/replication/replication_handler_test/basic_scenarios_test.rs b/d-engine-core/src/replication/replication_handler_test/basic_scenarios_test.rs index 39be7be0..b4e96c1f 100644 --- a/d-engine-core/src/replication/replication_handler_test/basic_scenarios_test.rs +++ b/d-engine-core/src/replication/replication_handler_test/basic_scenarios_test.rs @@ -68,6 +68,7 @@ async fn test_single_voter_builds_no_replication_requests() { total_voters: 1, }, &context, + &HashMap::new(), ) .await .unwrap(); @@ -127,6 +128,7 @@ async fn test_two_node_cluster_builds_one_replication_request() { total_voters: 2, }, &context, + &HashMap::new(), ) .await .unwrap(); @@ -200,6 +202,7 @@ async fn test_three_node_cluster_builds_two_replication_requests() { total_voters: 3, }, &context, + &HashMap::new(), ) .await .unwrap(); @@ -286,6 +289,7 @@ async fn test_five_node_cluster_builds_four_replication_requests() { total_voters: 5, }, &context, + &HashMap::new(), ) .await .unwrap(); @@ -301,3 +305,88 @@ async fn test_five_node_cluster_builds_four_replication_requests() { assert!(peer_ids.contains(&peer4_id)); assert!(peer_ids.contains(&peer5_id)); } + +/// A window-gated peer must still receive a request, just without entries (liveness), and its +/// position fields must match what an ungated peer at the same position gets. +/// +/// # Scenario +/// - Peers 2 and 3 both at `next_index = 1`, one old entry at index 1. +/// - `peer_gating_decisions = {2: true}`. +/// - Expected: two requests; peer 2's has empty entries, peer 3's has one entry; both carry the +/// same `prev_log_index`. +#[tokio::test] +async fn test_gated_peer_still_gets_empty_request_for_liveness() { + use crate::mock_insert_log_entries; + use crate::mock_log_entries_exist; + + let (_graceful_tx, graceful_rx) = watch::channel(()); + let mut context = mock_raft_context( + "/tmp/test_gated_peer_still_gets_empty_request_for_liveness", + graceful_rx, + None, + ); + let handler = ReplicationHandler::::new(1); + + let old_entries = mock_insert_log_entries(vec![1], 1, 1); + let mut raft_log = MockRaftLog::new(); + mock_log_entries_exist(&mut raft_log, old_entries); + raft_log.expect_first_entry_id().returning(|| 1); + raft_log.expect_entry_term().returning(|_| None); + context.storage.raft_log = Arc::new(raft_log); + + let peer = |id: u32| NodeMeta { + id, + address: format!("http://127.0.0.1:{}", 55000 + id), + role: NodeRole::Follower.into(), + status: NodeStatus::Active.into(), + }; + + let result = handler + .prepare_batch_requests( + vec![], + StateSnapshot { + current_term: 1, + voted_for: None, + commit_index: 0, + role: Leader.into(), + }, + LeaderStateSnapshot { + next_index: HashMap::from([(2, 1), (3, 1)]), + match_index: HashMap::new(), + noop_log_id: None, + }, + &ClusterMetadata { + single_voter: false, + replication_targets: vec![peer(2), peer(3)], + total_voters: 3, + }, + &context, + &HashMap::from([(2_u32, true)]), + ) + .await + .unwrap(); + + assert_eq!(result.append_requests.len(), 2, "both peers get a request"); + let req_of = |id: u32| { + &result + .append_requests + .iter() + .find(|(p, _, _)| *p == id) + .unwrap_or_else(|| panic!("no request for peer {id}")) + .1 + }; + assert!( + req_of(2).entries.is_empty(), + "the gated peer's request is an empty keepalive" + ); + assert_eq!( + req_of(3).entries.len(), + 1, + "the ungated peer still gets its entry" + ); + assert_eq!( + req_of(2).prev_log_index, + req_of(3).prev_log_index, + "the keepalive is a consistent request at the same position" + ); +} diff --git a/d-engine-core/src/replication/replication_handler_test/heartbeat_match_index_test.rs b/d-engine-core/src/replication/replication_handler_test/heartbeat_match_index_test.rs new file mode 100644 index 00000000..f90c815c --- /dev/null +++ b/d-engine-core/src/replication/replication_handler_test/heartbeat_match_index_test.rs @@ -0,0 +1,209 @@ +//! What a follower reports as `last_match`, and why it must never exceed what the request verified. +//! +//! A leader turns `last_match` into that follower's `match_index`, and counts it toward the +//! commit quorum. The follower has only verified its log against the leader's up to +//! `prev_log_index + entries.len()` of the request it just handled. Anything past that point +//! may be a tail left by an earlier leader that differs from the leader's log, so reporting +//! it would let unreplicated entries be counted as replicated. +//! +//! Scenario used throughout: the follower holds entries 1..=50 from term 1. The new leader +//! (term 2) has the same entries 1..=10, then its own, different entry 11. + +use std::sync::Arc; + +use d_engine_proto::common::Entry; +use d_engine_proto::server::replication::{AppendEntriesRequest, append_entries_response}; + +use crate::MockRaftLog; +use crate::MockTypeConfig; +use crate::RaftLog; +use crate::ReplicationCore; +use crate::ReplicationHandler; +use crate::StateSnapshot; +use crate::test_utils::{RaftLogCoreTestContext, mock_entries}; + +const OLD_TERM: u64 = 1; +const NEW_TERM: u64 = 2; + +fn follower_state() -> StateSnapshot { + StateSnapshot { + role: d_engine_proto::common::NodeRole::Follower as i32, + current_term: NEW_TERM, + voted_for: None, + commit_index: 0, + } +} + +/// A follower whose log is 1..=50, all from the old term (entries 11..=50 were never committed). +async fn follower_with_stale_tail(name: &str) -> RaftLogCoreTestContext { + let ctx = RaftLogCoreTestContext::new(name); + ctx.append_entries(1, 50, OLD_TERM).await; + ctx +} + +fn request( + prev_log_index: u64, + entries: Vec, +) -> AppendEntriesRequest { + AppendEntriesRequest { + term: NEW_TERM, + leader_id: 1, + prev_log_index, + prev_log_term: OLD_TERM, + entries, + leader_commit_index: 0, + } +} + +/// The handler is generic over a mock log type; this mock forwards every call the follower +/// path makes to the real in-memory log, so the behavior under test is the real one. +fn log_backed_by(follower: &RaftLogCoreTestContext) -> Arc { + let mut mock = MockRaftLog::new(); + + let real = follower.raft_log.clone(); + mock.expect_last_log_id().returning(move || real.last_log_id()); + let real = follower.raft_log.clone(); + mock.expect_last_entry_id().returning(move || real.last_entry_id()); + let real = follower.raft_log.clone(); + mock.expect_entry_term().returning(move |index| real.entry_term(index)); + let real = follower.raft_log.clone(); + mock.expect_first_index_for_term() + .returning(move |term| real.first_index_for_term(term)); + let real = follower.raft_log.clone(); + mock.expect_filter_out_conflicts_and_append().returning( + move |prev_index, prev_term, entries| { + tokio::task::block_in_place(|| { + tokio::runtime::Handle::current() + .block_on(real.filter_out_conflicts_and_append(prev_index, prev_term, entries)) + }) + }, + ); + Arc::new(mock) +} + +/// Handles `request` on the follower and returns the index it reports, or `None` for a rejection. +async fn reported_last_match( + follower: &RaftLogCoreTestContext, + request: AppendEntriesRequest, +) -> Option { + let handler = ReplicationHandler::::new(2); + let response = handler + .handle_append_entries(request, &follower_state(), &log_backed_by(follower)) + .await + .expect("handle_append_entries") + .response; + + match response.result { + Some(append_entries_response::Result::Success(success)) => { + Some(success.last_match.expect("success carries last_match").index) + } + _ => None, + } +} + +/// An empty request verified the logs only up to `prev_log_index`; the reply must not go past it. +#[tokio::test(flavor = "multi_thread")] +async fn test_empty_request_reply_does_not_claim_more_than_prev_log_index() { + let follower = follower_with_stale_tail("empty_request_reply").await; + + let reported = reported_last_match(&follower, request(10, vec![])).await; + + assert_eq!( + reported, + Some(10), + "the follower's own last entry is 50, but the request only verified index 10" + ); +} + +/// A request that re-sends entries the follower already has must be answered with the end of +/// the entries it carried, not with the follower's longer log. +#[tokio::test(flavor = "multi_thread")] +async fn test_resent_entries_reply_ends_at_the_last_entry_of_the_request() { + let follower = follower_with_stale_tail("resent_entries_reply").await; + let resent = mock_entries(11, 5, OLD_TERM); + + let reported = reported_last_match(&follower, request(10, resent)).await; + + assert_eq!(reported, Some(15), "the request carried entries 11..=15"); +} + +/// The leader's first request after election carries its own entry at index 11. The follower +/// must drop its stale tail, and the reply must end at index 11. +#[tokio::test(flavor = "multi_thread")] +async fn test_conflicting_entry_replaces_the_stale_tail_and_reply_ends_at_it() { + let follower = follower_with_stale_tail("conflicting_entry_reply").await; + let leaders_entry = mock_entries(11, 1, NEW_TERM); + + let reported = reported_last_match(&follower, request(10, leaders_entry)).await; + + assert_eq!(reported, Some(11)); + assert_eq!( + follower.raft_log.last_entry_id(), + 11, + "entries 12..=50 belonged to the earlier leader and must be gone" + ); +} + +/// An empty request whose `prev_log_index` the follower does not hold must be rejected. +#[tokio::test(flavor = "multi_thread")] +async fn test_empty_request_beyond_the_followers_log_is_rejected() { + let follower = follower_with_stale_tail("empty_request_beyond_log").await; + + let reported = reported_last_match(&follower, request(60, vec![])).await; + + assert_eq!( + reported, None, + "the follower has no entry 60 to match against" + ); +} + +/// An empty request whose `prev_log_term` differs from the follower's entry must be rejected. +#[tokio::test(flavor = "multi_thread")] +async fn test_empty_request_with_mismatching_prev_term_is_rejected() { + let follower = follower_with_stale_tail("empty_request_term_mismatch").await; + let mut mismatching = request(10, vec![]); + mismatching.prev_log_term = NEW_TERM; + + let reported = reported_last_match(&follower, mismatching).await; + + assert_eq!( + reported, None, + "the follower's entry 10 is from term 1, not term 2" + ); +} + +/// Sanity check for the scenario setup itself: the follower really holds entries 1..=50. +#[tokio::test(flavor = "multi_thread")] +async fn test_scenario_setup_follower_holds_the_stale_tail() { + let follower = follower_with_stale_tail("scenario_setup").await; + + assert_eq!(follower.raft_log.last_entry_id(), 50); + assert_eq!(follower.raft_log.entry_term(50), Some(OLD_TERM)); + let _ = Arc::strong_count(&follower.raft_log); +} + +/// A heartbeat (empty request) carries `leader_commit_index`, and the follower must cap its +/// commit advance at the position the request actually verified β€” `prev_log_index` β€” not at its +/// own (possibly stale) tail. Otherwise a follower holding entries 11..=50 left by an earlier +/// leader would apply them as "committed" when the leader only ever verified up to index 10. +#[tokio::test(flavor = "multi_thread")] +async fn test_empty_request_commit_index_does_not_exceed_prev_log_index() { + let follower = follower_with_stale_tail("empty_request_commit_cap").await; + + // The leader has committed up to 25, but still believes this follower matches only up to 10. + let mut heartbeat = request(10, vec![]); + heartbeat.leader_commit_index = 25; + + let handler = ReplicationHandler::::new(2); + let commit_index_update = handler + .handle_append_entries(heartbeat, &follower_state(), &log_backed_by(&follower)) + .await + .expect("handle_append_entries") + .commit_index_update; + + assert_eq!( + commit_index_update, + Some(10), + "commit must stop at the verified prev_log_index (10), not jump to the follower's own stale tail (50)" + ); +} diff --git a/d-engine-core/src/replication/replication_handler_test/log_retrieval_test.rs b/d-engine-core/src/replication/replication_handler_test/log_retrieval_test.rs index ed7c2fc0..38060ebe 100644 --- a/d-engine-core/src/replication/replication_handler_test/log_retrieval_test.rs +++ b/d-engine-core/src/replication/replication_handler_test/log_retrieval_test.rs @@ -69,6 +69,7 @@ async fn test_retrieve_only_new_entries_when_peer_caught_up() { &peer_next_indices, &context.raft_log, 1, + &HashMap::new(), ); // Assert: Only new entries returned (peer already has old logs) @@ -125,6 +126,7 @@ async fn test_retrieve_old_and_new_entries_when_peer_behind() { &peer_next_indices, &context.raft_log, 1, + &HashMap::new(), ); // Assert: Both old and new entries returned @@ -177,6 +179,7 @@ async fn test_retrieve_only_old_entries_when_no_new_entries() { &peer_next_indices, &context.raft_log, 1, + &HashMap::new(), ); // Assert: Only old entry returned (no new entries) @@ -230,6 +233,7 @@ async fn test_retrieve_limited_old_entries_with_max_limit() { &peer_next_indices, &context.raft_log, 1, + &HashMap::new(), ); // Assert: Only first 2 old entries + new entry (limited by max) @@ -285,6 +289,7 @@ async fn test_retrieve_only_new_entries_when_max_limit_zero() { &peer_next_indices, &context.raft_log, 1, + &HashMap::new(), ); // Assert: Only new entry (no old logs due to max=0) @@ -340,6 +345,7 @@ async fn test_leader_id_excluded_from_replication_targets() { &peer_next_indices, &context.raft_log, 1, + &HashMap::new(), ); // Assert: Peer3 receives entries @@ -395,6 +401,7 @@ async fn test_retrieve_corrupt_gap_when_range_read_short() { &peer_next_indices, &context.raft_log, 11, // first_index + &HashMap::new(), ); assert_eq!( @@ -459,6 +466,7 @@ async fn test_retrieve_corrupt_gap_when_range_read_gapped() { &peer_next_indices, &context.raft_log, 11, + &HashMap::new(), ); assert_eq!( @@ -473,3 +481,43 @@ async fn test_retrieve_corrupt_gap_when_range_read_gapped() { "gapped range read must be classified as CorruptGap" ); } + +/// A peer whose in-flight window is gated must get no prepared entries, while an ungated peer +/// with the same position still does. The gate is checked before the log is touched. +/// +/// # Scenario +/// - Peers 2 and 3 are both behind (`next_index = 1`), one old entry exists at index 1. +/// - `peer_gating_decisions = {2: true, 3: false}`. +/// - Expected: peer 3 is `Ready(old entry)`; peer 2 has no entry in the result. +#[tokio::test] +async fn test_gated_peer_gets_no_entries_while_ungated_peer_does() { + let mut context = setup_mock_replication_test_context(1); + let my_id = 1; + + let old_entries = mock_insert_log_entries(vec![1], 1, 1); + let raft_log_mut = Arc::get_mut(&mut context.raft_log).unwrap(); + mock_log_entries_exist(raft_log_mut, old_entries.clone()); + + let peer_next_indices = HashMap::from([(2_u32, 1_u64), (3_u32, 1_u64)]); + let handler = ReplicationHandler::::new(my_id); + + let result = handler.retrieve_to_be_synced_logs_for_peers( + &[], + 1, + 100, + &peer_next_indices, + &context.raft_log, + 1, + &HashMap::from([(2_u32, true), (3_u32, false)]), + ); + + assert!( + !result.contains_key(&2), + "a window-full peer must not have entries prepared" + ); + assert_eq!( + result.get(&3), + Some(&crate::PeerEntriesResult::Ready(old_entries)), + "an ungated peer at the same position must still get its entries" + ); +} diff --git a/d-engine-core/src/replication/replication_handler_test/mod.rs b/d-engine-core/src/replication/replication_handler_test/mod.rs index 6fc2262e..18eb92ff 100644 --- a/d-engine-core/src/replication/replication_handler_test/mod.rs +++ b/d-engine-core/src/replication/replication_handler_test/mod.rs @@ -41,3 +41,6 @@ mod quorum_calculation_test; #[cfg(test)] mod snapshot_trigger_test; + +#[cfg(test)] +mod heartbeat_match_index_test; diff --git a/d-engine-core/src/replication/replication_handler_test/quorum_calculation_test.rs b/d-engine-core/src/replication/replication_handler_test/quorum_calculation_test.rs index 332170b1..93e22afd 100644 --- a/d-engine-core/src/replication/replication_handler_test/quorum_calculation_test.rs +++ b/d-engine-core/src/replication/replication_handler_test/quorum_calculation_test.rs @@ -74,6 +74,7 @@ async fn test_two_node_cluster_builds_one_peer_request() { total_voters: 2, }, &context, + &HashMap::new(), ) .await .unwrap(); diff --git a/d-engine-core/src/replication/replication_handler_test/snapshot_trigger_test.rs b/d-engine-core/src/replication/replication_handler_test/snapshot_trigger_test.rs index 985058c6..38c327d8 100644 --- a/d-engine-core/src/replication/replication_handler_test/snapshot_trigger_test.rs +++ b/d-engine-core/src/replication/replication_handler_test/snapshot_trigger_test.rs @@ -122,6 +122,7 @@ async fn test_prepare_batch_requests_routes_lagging_peer_to_snapshot() { total_voters: 2, }, &ctx, + &HashMap::new(), ) .await .unwrap(); @@ -191,6 +192,7 @@ async fn test_prepare_batch_requests_caught_up_peer_gets_append_entries() { total_voters: 2, }, &ctx, + &HashMap::new(), ) .await .unwrap(); @@ -261,6 +263,7 @@ async fn test_prepare_batch_requests_splits_snapshot_and_append_peers() { total_voters: 3, }, &ctx, + &HashMap::new(), ) .await .unwrap(); @@ -349,6 +352,7 @@ async fn test_prepare_batch_requests_routes_fresh_peer_to_append_when_log_never_ total_voters: 2, }, &ctx, + &HashMap::new(), ) .await .unwrap(); @@ -420,6 +424,7 @@ async fn test_prepare_batch_requests_routes_never_replicated_peer_to_snapshot_ex total_voters: 2, }, &ctx, + &HashMap::new(), ) .await .unwrap(); diff --git a/d-engine-core/src/state_machine_handler/default_state_machine_handler_test.rs b/d-engine-core/src/state_machine_handler/default_state_machine_handler_test.rs index e5958c32..67915722 100644 --- a/d-engine-core/src/state_machine_handler/default_state_machine_handler_test.rs +++ b/d-engine-core/src/state_machine_handler/default_state_machine_handler_test.rs @@ -2,6 +2,7 @@ use super::DefaultStateMachineHandler; use super::DefaultStateMachineWriter; use super::StateMachineHandler; use super::StateMachineWriterOps; +#[cfg(feature = "watch")] use super::broadcast_watch_events; use super::new_reader_writer_pair; use crate::Error; @@ -14,9 +15,7 @@ use crate::test_utils::snapshot_config; use bytes::Bytes; use d_engine_proto::client::WriteCommand; use d_engine_proto::client::write_command::batch_op::Op; -use d_engine_proto::client::write_command::{ - Batch, BatchOp as ProtoBatchOp, Insert, Operation, batch_op, -}; +use d_engine_proto::client::write_command::{Batch, BatchOp as ProtoBatchOp, Insert, Operation}; use d_engine_proto::common::Entry; use d_engine_proto::common::EntryPayload; use d_engine_proto::common::LogId; @@ -313,7 +312,7 @@ mod apply_chunk_test { let cmd = WriteCommand { operation: Some(Operation::Batch(Batch { ops: vec![ProtoBatchOp { - op: Some(batch_op::Op::Insert(Insert { + op: Some(Op::Insert(Insert { key: Bytes::from_static(b"k1"), value: Bytes::from_static(b"v1"), ttl_secs: 0, @@ -374,7 +373,7 @@ mod apply_chunk_test { let cmd = WriteCommand { operation: Some(Operation::Batch(Batch { ops: vec![ProtoBatchOp { - op: Some(batch_op::Op::Insert(Insert { + op: Some(Op::Insert(Insert { key: Bytes::from_static(b"k1"), value: Bytes::from_static(b"v1"), ttl_secs: 0, diff --git a/d-engine-core/src/state_machine_handler/snapshot_policy/log_size.rs b/d-engine-core/src/state_machine_handler/snapshot_policy/log_size.rs index 96a466f7..da59406e 100644 --- a/d-engine-core/src/state_machine_handler/snapshot_policy/log_size.rs +++ b/d-engine-core/src/state_machine_handler/snapshot_policy/log_size.rs @@ -4,7 +4,7 @@ use std::sync::atomic::AtomicBool; use std::sync::atomic::AtomicU64; use std::sync::atomic::Ordering; - +use tracing::error; use tracing::trace; use tracing::warn; @@ -39,12 +39,28 @@ impl SnapshotPolicy for LogSizePolicy { let lag = self.calculate_lag(ctx); let threshold = self.threshold.load(Ordering::Relaxed); - if threshold > 0 && lag >= threshold.saturating_mul(10) { - warn!( - lag, - threshold, - "Log lag exceeds 10x snapshot threshold β€” snapshots may not be keeping up" - ); + metrics::gauge!("core.raft.snapshot.log_lag").set(lag as f64); + + // The in-memory Raft log grows until a snapshot purges it. If snapshot + // creation can't keep up with the write rate this climbs unbounded and + // eventually OOMs the node β€” make it loud well before that. + if threshold > 0 { + if lag >= threshold.saturating_mul(50) { + error!( + lag, + threshold, + "Raft log lag is 50x the snapshot threshold β€” snapshot creation is \ + NOT keeping up with writes; the in-memory log is growing unbounded \ + and will OOM this node. Check snapshot/apply throughput." + ); + } else if lag >= threshold.saturating_mul(10) { + warn!( + lag, + threshold, + "Raft log lag exceeds 10x the snapshot threshold β€” snapshots may \ + not be keeping up" + ); + } } let should_trigger = lag >= threshold; diff --git a/d-engine-core/src/state_machine_handler/worker_test.rs b/d-engine-core/src/state_machine_handler/worker_test.rs index 6c082abe..018e1806 100644 --- a/d-engine-core/src/state_machine_handler/worker_test.rs +++ b/d-engine-core/src/state_machine_handler/worker_test.rs @@ -1337,12 +1337,21 @@ async fn test_local_snapshot_ready_reports_operation_failed_when_superseded_clea let result = response_rx.await.unwrap(); - // Restore permissions before any assertion can panic and skip this β€” otherwise - // the tempdir is left behind, unremovable by the test harness's own cleanup. - let mut perms = std::fs::metadata(&dir_path).unwrap().permissions(); - perms.set_mode(0o700); - std::fs::set_permissions(&dir_path, perms).unwrap(); - std::fs::remove_dir_all(&dir_path).unwrap(); + // The tempdir might already be gone: `OwnedSnapshotDir::drop`'s detached cleanup + // thread (command.rs) races this teardown and, under load, can win β€” that's a + // benign outcome (goal is just "no leftover dir"), not a test failure. + if let Ok(meta) = std::fs::metadata(&dir_path) { + let mut perms = meta.permissions(); + perms.set_mode(0o700); + std::fs::set_permissions(&dir_path, perms).unwrap(); + if let Err(e) = std::fs::remove_dir_all(&dir_path) { + assert_eq!( + e.kind(), + std::io::ErrorKind::NotFound, + "unexpected teardown error: {e}" + ); + } + } assert!( matches!( diff --git a/d-engine-core/src/storage/buffered_raft_log_test/mod.rs b/d-engine-core/src/storage/buffered_raft_log_test/mod.rs deleted file mode 100644 index 8489dd83..00000000 --- a/d-engine-core/src/storage/buffered_raft_log_test/mod.rs +++ /dev/null @@ -1,39 +0,0 @@ -//! BufferedRaftLog unit tests -//! -//! This module contains comprehensive unit tests for `BufferedRaftLog` organized by -//! functional domains for better maintainability. -//! -//! ## Test Organization -//! -//! - `basic_operations_test`: CRUD operations, range queries, conflict resolution -//! - `flush_strategy_test`: DiskFirst/MemFirst/Batched persistence strategies (Mock-based) -//! - `id_allocation_test`: ID pre-allocation logic and concurrency -//! - `term_index_test`: Term boundary tracking and calculation -//! - `concurrent_operations_test`: Thread-safety and race condition tests -//! - `crash_recovery_test`: Mock-based crash recovery simulation -//! - `performance_test`: Performance characteristics (not absolute throughput) -//! - `edge_cases_test`: Boundary conditions and unusual scenarios -//! - `helper_functions_test`: Internal utility function tests -//! -//! ## Test Strategy -//! -//! These tests use `MockStorageEngine` to verify algorithm correctness without real disk I/O. -//! Integration tests with `FileStorageEngine` are in `d-engine-server/tests/integration/`. - -// mod basic_operations_test; -// mod concurrent_fsync_test; -// mod concurrent_operations_test; -// mod drain_fsync_test; -// mod durable_index_test; -// mod edge_cases_test; -// mod flush_strategy_test; -// mod id_allocation_test; -// mod performance_test; -// mod pipeline_overlap_test; -// mod quorum_durability_test; -// mod raft_properties_test; -// mod remove_range_test; -// mod shutdown_test; -// mod term_index_test; -// mod term_segments_test; -// mod worker_test; diff --git a/d-engine-core/src/storage/buffered_raft_log_test/performance_test.rs b/d-engine-core/src/storage/buffered_raft_log_test/performance_test.rs deleted file mode 100644 index a9530bf9..00000000 --- a/d-engine-core/src/storage/buffered_raft_log_test/performance_test.rs +++ /dev/null @@ -1,244 +0,0 @@ -//! Performance tests for BufferedRaftLog with controllable delays -//! -//! These tests verify BufferedRaftLog performance behavior during concurrent -//! operations like flush, using MockStorageEngine with controllable delays. - -use std::sync::Arc; -use std::time::Duration; - -use bytes::Bytes; -use tokio::sync::Barrier; -use tokio::time::Instant; - -use crate::{ - BufferedRaftLog, FlushPolicy, MockLogStore, MockMetaStore, MockStorageEngine, MockTypeConfig, - PersistenceConfig, PersistenceStrategy, RaftLog, -}; -use d_engine_proto::common::{Entry, EntryPayload}; - -// Test helper: Creates storage with controllable delay -fn create_delayed_storage(delay_ms: u64) -> Arc { - let mut log_store = MockLogStore::new(); - log_store.expect_last_index().returning(|| 0); - log_store.expect_load_purge_boundary().returning(|| Ok(None)); - log_store.expect_truncate().returning(|_| Ok(())); - log_store.expect_reset().returning(|| Ok(())); - log_store.expect_is_write_durable().returning(|| true); - log_store.expect_flush().returning(|| Ok(())); - - // Add controllable delay to persist_entries - log_store.expect_persist_entries().returning(move |_| { - let delay = Duration::from_millis(delay_ms); - std::thread::sleep(delay); - Ok(()) - }); - - Arc::new(MockStorageEngine::from(log_store, MockMetaStore::new())) -} - -// Tests reset performance during active flush -#[tokio::test] -async fn test_reset_performance_during_active_flush() { - // persist_entries mock sleeps for FLUSH_DELAY_MS. - // reset() waits for the IO thread to finish its current in-flight operation before processing - // Reset β€” this is correct behavior. The test verifies reset completes within a bounded time - // (3x the flush delay) and does not block indefinitely. - const FLUSH_DELAY_MS: u64 = 200; - let max_reset_duration_ms = FLUSH_DELAY_MS * 3; // 600ms: accounts for IO thread overhead - - let test_cases = vec![ - ( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1000, - }, - ), - ( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - ), - ]; - - for (strategy, flush_policy) in test_cases { - let storage = create_delayed_storage(FLUSH_DELAY_MS); - let config = PersistenceConfig { - strategy: strategy.clone(), - flush_policy: flush_policy.clone(), - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }; - - let (log, receiver) = BufferedRaftLog::::new(1, config, storage); - let log = log.start(receiver, None); - let barrier = Arc::new(Barrier::new(2)); - - // Start long-running append+flush in background (slow due to persist_entries delay) - let flush_log = log.clone(); - let flush_barrier = barrier.clone(); - tokio::spawn(async move { - flush_barrier.wait().await; // Sync point - let entries: Vec = (1..=10) - .map(|i| Entry { - index: i, - term: 1, - payload: None, - }) - .collect(); - let _ = flush_log.append_entries(entries).await; - let _ = flush_log.flush().await; - }); - - // Wait for flush to start - barrier.wait().await; - - // Measure reset performance during active flush - let start = Instant::now(); - log.reset().await.unwrap(); - let duration = start.elapsed(); - - assert!( - duration.as_millis() < max_reset_duration_ms as u128, - "Reset took {}ms during active flush ({:?}/{:?})", - duration.as_millis(), - strategy, - flush_policy - ); - } -} - -// Tests filter_out_conflicts performance with active flush -#[tokio::test] -async fn test_filter_conflicts_performance_during_flush() { - let is_ci = std::env::var("CI").is_ok(); - // Relax time limit in CI environment - let test_cases = if is_ci { - vec![(10, 500), (100, 500), (1000, 500)] - } else { - vec![(10, 50), (100, 50), (1000, 50)] - }; - - const FLUSH_DELAY_MS: u64 = 300; - - for (idle_flush_interval_ms, max_duration_ms) in test_cases { - let storage = create_delayed_storage(FLUSH_DELAY_MS); - let config = PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }; - - let (log, receiver) = BufferedRaftLog::::new(1, config, storage); - let log = log.start(receiver, None); - let barrier = Arc::new(Barrier::new(2)); - - // Populate with test data - let mut entries = vec![]; - for i in 1..=1000 { - entries.push(Entry { - index: i, - term: 1, - payload: Some(EntryPayload::command(Bytes::from(vec![0; 256]))), - }); - } - log.append_entries(entries).await.unwrap(); - - // Start long flush in background (slow due to persist_entries delay) - let flush_log = log.clone(); - let flush_barrier = barrier.clone(); - tokio::spawn(async move { - flush_barrier.wait().await; - let _ = flush_log.flush().await; - }); - - // Wait for flush to start - barrier.wait().await; - - // Measure performance during active flush - let start = Instant::now(); - log.filter_out_conflicts_and_append( - 500, - 1, - vec![Entry { - index: 501, - term: 1, - payload: Some(EntryPayload::command(Bytes::from(vec![1; 256]))), - }], - ) - .await - .unwrap(); - - let duration = start.elapsed(); - assert!( - duration.as_millis() < max_duration_ms as u128, - "Operation took {}ms with {}ms interval during flush", - duration.as_millis(), - idle_flush_interval_ms - ); - } -} - -// Tests fresh cluster performance consistency -#[tokio::test] -async fn test_fresh_cluster_performance_consistency() { - let is_ci = std::env::var("CI").is_ok(); - // Relax time limit in CI environment - let max_duration_ms = if is_ci { 50 } else { 5 }; - - let test_cases = vec![ - ( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1000, - }, - ), - ( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - ), - ]; - - for (strategy, flush_policy) in test_cases { - let mut log_store = MockLogStore::new(); - log_store.expect_is_write_durable().returning(|| true); - log_store.expect_flush().return_once(|| Ok(())); - log_store.expect_last_index().returning(|| 0); - log_store.expect_load_purge_boundary().returning(|| Ok(None)); - log_store.expect_truncate().returning(|_| Ok(())); - log_store.expect_persist_entries().returning(|_| Ok(())); - log_store.expect_reset().returning(|| Ok(())); - - let config = PersistenceConfig { - strategy: strategy.clone(), - flush_policy: flush_policy.clone(), - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }; - - let (log, receiver) = BufferedRaftLog::::new( - 1, - config, - Arc::new(MockStorageEngine::from(log_store, MockMetaStore::new())), - ); - let log = log.start(receiver, None); - - // Measure reset performance in fresh cluster - let start = Instant::now(); - log.reset().await.unwrap(); - let duration = start.elapsed(); - - assert!( - duration.as_millis() < max_duration_ms as u128, - "Fresh cluster reset took {}ms ({:?}/{:?})", - duration.as_millis(), - strategy, - flush_policy - ); - } -} diff --git a/d-engine-core/src/storage/buffered_raft_log_test/quorum_durability_test.rs b/d-engine-core/src/storage/buffered_raft_log_test/quorum_durability_test.rs deleted file mode 100644 index 0429eebc..00000000 --- a/d-engine-core/src/storage/buffered_raft_log_test/quorum_durability_test.rs +++ /dev/null @@ -1,152 +0,0 @@ -//! Quorum Durability Tests -//! -//! MemFirst design: leader contributes `last_entry_id` (in-memory) to quorum. -//! IO thread persistence is async and NOT on the commit critical path. -//! -//! Follower ACK path: followers ACK immediately after memory write (no `wait_durable`). -//! IO thread fsyncs asynchronously; crash safety is guaranteed by quorum, not per-follower durability. - -use crate::storage::raft_log::RaftLog; -use crate::test_utils::BufferedRaftLogTestContext; -use crate::{FlushPolicy, PersistenceStrategy}; -use std::time::Duration; - -/// Flush policy with a far-future safety timer β€” IO thread only fsyncs on WriteNotify. -/// In current_thread test runtime, durable_index stays at 0 immediately after append_entries -/// because the IO thread task has no chance to run until the test yields. -fn no_auto_flush_policy() -> FlushPolicy { - FlushPolicy::Batch { - idle_flush_interval_ms: 999_999, - } -} - -// ── Leader quorum uses last_entry_id (in-memory), not durable_index ── - -/// MemFirst: leader's quorum contribution is last_entry_id (in-memory), not durable_index. -/// -/// Even when durable_index=0 (IO thread has not flushed), quorum must be satisfied -/// as soon as last_entry_id + follower ACKs form a majority. IO persistence is async -/// and must NOT block commit. -/// -/// This test FAILS if calculate_majority_matched_index uses durable_index (the bug -/// introduced by fix #329 which incorrectly put IO thread latency on the commit -/// critical path, causing +617Β΅s avg latency regression in 3-node embedded bench). -#[tokio::test] -async fn test_memfirst_quorum_uses_last_entry_id_not_durable_index() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - no_auto_flush_policy(), // durable_index stays 0 β€” IO thread won't run - "test_memfirst_quorum_last_entry_id", - ); - - // Entry written to SkipMap (in memory). IO thread has not flushed yet. - ctx.append_entries(1, 1, 1).await; - - assert_eq!(ctx.raft_log.last_entry_id(), 1); - assert_eq!(ctx.raft_log.durable_index(), 0); // IO thread hasn't run - - let result = ctx.raft_log.calculate_majority_matched_index( - 1, - 0, - vec![1], // one follower acked index=1; together with leader = majority of 3 - ); - - // MemFirst: leader contributes last_entry_id=1. - // quorum = [leader=1, follower=1] β†’ majority of {leader, f1, f2} satisfied β†’ Some(1). - assert_eq!( - result, - Some(1), - "MemFirst: quorum must use last_entry_id, not durable_index β€” IO must not block commit" - ); -} - -/// Once the leader flushes, quorum calculation should succeed. -/// -/// This test verifies the positive case: after flush, durable_index=1, -/// quorum should proceed normally. -#[tokio::test] -async fn test_quorum_succeeds_after_leader_flush() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 999_999, // only threshold trigger, no timer - }, - "test_quorum_after_flush", - ); - - // Append entry 1 β€” threshold=1 so flush fires immediately - ctx.append_entries(1, 1, 1).await; - - // Wait for flush to complete - tokio::time::sleep(Duration::from_millis(50)).await; - - assert_eq!(ctx.raft_log.last_entry_id(), 1); - assert_eq!( - ctx.raft_log.durable_index(), - 1, - "durable_index must be 1 after threshold flush" - ); - - let new_commit = ctx.raft_log.calculate_majority_matched_index( - 1, - 0, - vec![1], // follower acked - ); - - // Correct: durable_index=1 = last_entry_id=1, quorum should pass - assert_eq!( - new_commit, - Some(1), - "quorum must succeed after leader flush" - ); -} - -// ── Bug 2: gap between last_entry_id and durable_index ── - -/// Demonstrates that after append_entries with MemFirst + no-auto-flush, -/// last_entry_id and durable_index diverge. -/// -/// This is the root condition enabling the bug: both values exist, -/// but quorum calculation only uses the unsafe one. -#[tokio::test] -async fn test_last_entry_id_diverges_from_durable_index_with_mem_first() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - no_auto_flush_policy(), - "test_diverge_mem_first", - ); - - ctx.append_entries(1, 5, 1).await; // entries 1..=5, no flush - - assert_eq!(ctx.raft_log.last_entry_id(), 5, "memory index should be 5"); - assert_eq!( - ctx.raft_log.durable_index(), - 0, - "durable_index must remain 0: no flush has run" - ); - // This gap (5 vs 0) is exactly what the quorum bug exploits. -} - -/// After explicit flush, durable_index must equal last_entry_id. -#[tokio::test] -async fn test_durable_index_equals_last_entry_id_after_flush() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_no_diverge_after_flush", - ); - - ctx.append_entries(1, 5, 1).await; - ctx.raft_log.flush().await.unwrap(); - - let last = ctx.raft_log.last_entry_id(); - let durable = ctx.raft_log.durable_index(); - - assert_eq!(last, 5); - assert_eq!( - durable, last, - "durable_index must equal last_entry_id after flush" - ); -} diff --git a/d-engine-core/src/storage/buffered_raft_log_test/shutdown_test.rs b/d-engine-core/src/storage/buffered_raft_log_test/shutdown_test.rs deleted file mode 100644 index c14d7c4a..00000000 --- a/d-engine-core/src/storage/buffered_raft_log_test/shutdown_test.rs +++ /dev/null @@ -1,337 +0,0 @@ -use std::sync::Arc; -use std::time::Duration; - -use bytes::Bytes; - -use crate::storage::raft_log::RaftLog; -use crate::test_utils::BufferedRaftLogTestContext; -use crate::{ - BufferedRaftLog, FlushPolicy, MockLogStore, MockMetaStore, MockStorageEngine, MockTypeConfig, - PersistenceConfig, PersistenceStrategy, -}; -use d_engine_proto::common::{Entry, EntryPayload}; - -fn entry( - index: u64, - term: u64, -) -> Entry { - Entry { - index, - term, - payload: None, - } -} - -/// Verifies that the IO thread exits cleanly when the outer tokio runtime is dropped before -/// `close()` is called β€” the root scenario that caused SIGABRT in CI. -/// -/// Root cause: the previous implementation used `Handle::current()` to borrow the outer -/// runtime's timer wheel. When the outer runtime dropped, `interval.tick().await` accessed a -/// destroyed timer β†’ panic while holding a pthread mutex β†’ SIGABRT (uncatchable). -/// -/// After the fix the IO thread owns its own `new_current_thread` runtime; dropping the outer -/// runtime has no effect on the IO thread's timers. -#[test] -fn test_io_thread_survives_runtime_drop() { - // Build an outer runtime β€” simulates the tokio test runtime that is dropped at test end. - let rt = tokio::runtime::Builder::new_current_thread() - .enable_all() - .build() - .expect("failed to build outer runtime"); - - let storage = Arc::new(MockStorageEngine::with_id( - "test_io_thread_runtime_drop".to_string(), - )); - - let raft_log = rt.block_on(async { - let (log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 50, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - storage, - ); - let log = log.start(receiver, None); - for i in 1..=10 { - log.append_entries(vec![Entry { - index: i, - term: 1, - payload: None, - }]) - .await - .unwrap(); - } - log - }); - - // Drop the outer runtime BEFORE dropping raft_log. - // Old code: SIGABRT (process abort). - // Fixed code: IO thread's own runtime is unaffected; it exits cleanly on Drop. - drop(rt); - - // Allow the IO thread time to process the Shutdown signal sent by Drop. - std::thread::sleep(Duration::from_millis(200)); - - // If we reach here the process did not abort β€” test passes. - drop(raft_log); -} - -#[tokio::test] -async fn test_shutdown_closes_channel_properly() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 100, - }, - "test_shutdown_channel", - ); - - // Add some entries - for i in 1..=10 { - ctx.raft_log - .append_entries(vec![Entry { - index: i, - term: 1, - payload: None, - }]) - .await - .unwrap(); - } - - // Drop raft_log to trigger shutdown - drop(ctx.raft_log); - - // Verify shutdown completes without hanging - tokio::time::sleep(Duration::from_millis(50)).await; -} - -#[tokio::test] -async fn test_shutdown_awaits_worker_completion() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 5000, - }, - "test_shutdown_await_workers", - ); - - // Append entries to trigger worker activity - for i in 1..=50 { - ctx.raft_log - .append_entries(vec![Entry { - index: i, - term: 1, - payload: Some(EntryPayload::command(Bytes::from(vec![0u8; 100]))), - }]) - .await - .unwrap(); - } - - // Force flush to ensure workers have work - ctx.raft_log.flush().await.unwrap(); - - // Give workers time to start processing - tokio::time::sleep(Duration::from_millis(50)).await; - - // Trigger shutdown via Drop - let shutdown_start = std::time::Instant::now(); - drop(ctx.raft_log); - let shutdown_duration = shutdown_start.elapsed(); - - // Verify shutdown completed in reasonable time - assert!( - shutdown_duration < Duration::from_millis(500), - "Shutdown took too long: {shutdown_duration:?}", - ); -} - -#[tokio::test] -async fn test_shutdown_handles_slow_workers() { - let storage = Arc::new(MockStorageEngine::with_id( - "test_shutdown_slow_workers".to_string(), - )); - - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 100, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - storage, - ); - - let raft_log = raft_log.start(receiver, None); - - // Append entries - for i in 1..=10 { - raft_log - .append_entries(vec![Entry { - index: i, - term: 1, - payload: Some(EntryPayload::command(Bytes::from(vec![0u8; 50]))), - }]) - .await - .unwrap(); - } - - // Trigger flush - raft_log.flush().await.unwrap(); - - // Give workers time to process - tokio::time::sleep(Duration::from_millis(50)).await; - - // Shutdown should wait for workers - let shutdown_start = std::time::Instant::now(); - drop(raft_log); - let shutdown_duration = shutdown_start.elapsed(); - - assert!( - shutdown_duration < Duration::from_millis(1000), - "Shutdown with slow workers took too long" - ); -} - -#[tokio::test] -async fn test_shutdown_with_multiple_flushes() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 100, - }, - "test_shutdown_multiple_flushes", - ); - - // Create multiple flush operations - for batch in 0..5 { - for i in 1..=10 { - let index = batch * 10 + i; - ctx.raft_log - .append_entries(vec![Entry { - index, - term: 1, - payload: Some(EntryPayload::command(Bytes::from(vec![0u8; 100]))), - }]) - .await - .unwrap(); - } - ctx.raft_log.flush().await.unwrap(); - } - - // Give workers time to process - tokio::time::sleep(Duration::from_millis(200)).await; - - // Shutdown should handle all pending work - let shutdown_start = std::time::Instant::now(); - drop(ctx.raft_log); - let shutdown_duration = shutdown_start.elapsed(); - - assert!( - shutdown_duration < Duration::from_millis(500), - "Shutdown with multiple flushes took too long" - ); -} - -/// Verifies that a fatal IOTask::ReplaceRange failure: -/// 1. Propagates the error synchronously to the caller via the done channel. -/// 2. Shuts down the IO thread. -/// 3. Poisons the log β€” writes after the failure are rejected outright, not -/// silently accepted into memory with no IO thread left to persist them. -/// -/// Root issue: without `return` after done.send(Err), batch_processor continues -/// running on a storage whose on-disk state is now inconsistent with memory. -/// With the fix, it exits immediately so no further IO is attempted. -/// -/// ## Update (2026-07-19) -/// Point 3 and the poisoning check are new. The old version of this test -/// asserted the opposite of point 3 β€” that a write after the fatal error -/// "succeeds (in-memory only)". That was accurate but was itself a bug this -/// session closed: a write silently accepted with no live IO thread to ever -/// persist it. The trailing `flush()` check is removed β€” with writes now -/// rejected, `max_index` never gets ahead of `durable_index`, so `flush()` -/// short-circuits to `Ok(())` and no longer exercises the dead-channel path. -#[tokio::test] -async fn test_replace_range_failure_propagates_error_and_shuts_down_io_thread() { - let mut log_store = MockLogStore::new(); - - // replace_range always fails β€” simulates an unrecoverable disk error. - log_store - .expect_replace_range() - .returning(|_, _| Err(crate::Error::Fatal("simulated disk failure".into()))); - - log_store.expect_last_index().returning(|| 0); - log_store.expect_persist_entries().returning(|_| Ok(())); - log_store.expect_entry().returning(|_| Ok(None)); - log_store.expect_get_entries().returning(|_| Ok(vec![])); - log_store.expect_purge().returning(|_| Ok(())); - log_store.expect_load_purge_boundary().returning(|| Ok(None)); - log_store.expect_reset().returning(|| Ok(())); - log_store.expect_truncate().returning(|_| Ok(())); - log_store.expect_is_write_durable().returning(|| true); - log_store.expect_flush().returning(|| Ok(())); - log_store.expect_flush_async().returning(|| Ok(())); - - let mut meta_store = MockMetaStore::new(); - meta_store.expect_save_hard_state().returning(|_| Ok(())); - meta_store.expect_load_hard_state().returning(|| Ok(None)); - meta_store.expect_flush().returning(|| Ok(())); - meta_store.expect_flush_async().returning(|| Ok(())); - - let storage = Arc::new(MockStorageEngine::from(log_store, meta_store)); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, // no auto-flush - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - storage, - ); - let raft_log = raft_log.start(receiver, None); - std::thread::sleep(Duration::from_millis(10)); - - // Append [1..4] term=1 and let the IO thread persist them (durable_index β†’ 4). - raft_log - .append_entries(vec![entry(1, 1), entry(2, 1), entry(3, 1), entry(4, 1)]) - .await - .unwrap(); - std::thread::sleep(Duration::from_millis(30)); - - // Trigger IOTask::ReplaceRange: term conflict at index 3 (term 1 β†’ 2). - let result = raft_log - .filter_out_conflicts_and_append(2, 1, vec![entry(3, 2), entry(4, 2)]) - .await; - - // Error must be propagated back to the caller via the done channel. - assert!( - result.is_err(), - "expected ReplaceRange failure to be propagated to caller" - ); - - // Allow IO thread time to fully exit after the fatal error. - tokio::time::sleep(Duration::from_millis(50)).await; - - assert!( - raft_log.is_poisoned(), - "a replace_range() failure must poison the log" - ); - - // Writes after the failure must be rejected outright, not silently - // accepted into memory with no live IO thread to ever persist them. - let append_result = raft_log.append_entries(vec![entry(5, 2)]).await; - assert!( - append_result.is_err(), - "writes must be rejected once poisoned" - ); -} diff --git a/d-engine-core/src/storage/buffered_raft_log_test/worker_test.rs b/d-engine-core/src/storage/buffered_raft_log_test/worker_test.rs deleted file mode 100644 index c2181802..00000000 --- a/d-engine-core/src/storage/buffered_raft_log_test/worker_test.rs +++ /dev/null @@ -1,30 +0,0 @@ -use std::time::Duration; - -use crate::storage::raft_log::RaftLog; -use crate::test_utils::BufferedRaftLogTestContext; -use crate::{FlushPolicy, PersistenceStrategy}; - -/// Verifies that the flush worker continues operating normally after processing a large number -/// of flush tasks β€” the worker does not exit or become unresponsive under sustained load. -#[tokio::test] -async fn test_flush_worker_sustains_throughput_under_load() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 50, - }, - "test_flush_worker_sustains_throughput", - ); - - for i in 1..=100 { - ctx.append_entries(i, 1, 1).await; - } - - tokio::time::sleep(Duration::from_millis(200)).await; - - // Verify the worker is still alive by confirming continued processing - ctx.append_entries(101, 50, 1).await; - tokio::time::sleep(Duration::from_millis(100)).await; - - assert_eq!(ctx.raft_log.last_entry_id(), 150); -} diff --git a/d-engine-core/src/storage/fsync_coordinator.rs b/d-engine-core/src/storage/fsync_coordinator.rs deleted file mode 100644 index 47b74313..00000000 --- a/d-engine-core/src/storage/fsync_coordinator.rs +++ /dev/null @@ -1,178 +0,0 @@ -use crate::BufferedRaftLog; -use crate::Error; -use crate::LogStore; -use crate::Result; -use crate::TypeConfig; -use std::sync::Arc; -use std::sync::Mutex; -use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; -use tokio::sync::oneshot; -use tracing::error; - -/// Tracks whether a fsync task is currently running on the blocking pool. -/// Ensures at most one physical `flush_wal` call is in flight at any time, -/// restoring natural batching: entries that arrive while a fsync is running -/// accumulate in `pending_max`/`pending_replies`, and are picked up by the -/// SAME task once it finishes its current round β€” rather than spawning a -/// new competing task per `write_notify` wakeup. -pub(super) struct FsyncCoordinator { - inflight: AtomicBool, - pending_max: AtomicU64, - pending_replies: Mutex>>>, - generation: AtomicU64, // Bumped on every reset; fences out stale in-flight fsync results. -} - -impl FsyncCoordinator { - pub(super) fn new() -> Self { - Self { - inflight: AtomicBool::new(false), - pending_max: AtomicU64::new(0), - pending_replies: Mutex::new(Vec::new()), - generation: AtomicU64::new(0), - } - } - - /// Called from the IO thread on every wakeup. Records new work and, if no - /// fsync task is currently running, kicks one off. Never spawns a second - /// concurrent task β€” additional calls while one is in flight just update - /// the pending state for it to pick up next round. - pub(super) fn submit( - self: &Arc, - this: &Arc>, - max_index: u64, - replies: Vec>>, - ) { - if max_index > 0 { - self.pending_max.fetch_max(max_index, Ordering::AcqRel); - } - if !replies.is_empty() { - self.pending_replies.lock().unwrap().extend(replies); - } - - if self - .inflight - .compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire) - .is_err() - { - return; // Already running β€” it will pick up what we just recorded. - } - - metrics::gauge!("core.raft.fsync.inflight").set(1.0); - - let coord = Arc::clone(self); - let this = Arc::clone(this); - tokio::task::spawn_blocking(move || coord.run_until_caught_up(&this)); - } - - /// Runs on the blocking pool. Keeps fsyncing and re-checking for newly - /// accumulated work until there's nothing left, then clears `inflight`. - pub(super) fn run_until_caught_up( - &self, - this: &Arc>, - ) { - loop { - let gen_at_start = self.generation.load(Ordering::Acquire); - - let max_index = self.pending_max.swap(0, Ordering::AcqRel); - let replies = std::mem::take(&mut *self.pending_replies.lock().unwrap()); - - if this.is_poisoned() { - for reply in replies { - let _ = reply.send(Err(Error::Fatal("raft log storage is poisoned".into()))); - } - self.inflight.store(false, Ordering::Release); - metrics::gauge!("core.raft.fsync.inflight").set(0.0); - return; - } - - if max_index == 0 && replies.is_empty() { - self.inflight.store(false, Ordering::Release); - metrics::gauge!("core.raft.fsync.inflight").set(0.0); - // Re-check: something may have slipped in between the swap - // above and clearing `inflight`. If so, re-arm. - if (self.pending_max.load(Ordering::Acquire) > 0 - || !self.pending_replies.lock().unwrap().is_empty()) - && self - .inflight - .compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire) - .is_ok() - { - metrics::gauge!("core.raft.fsync.inflight").set(1.0); - continue; - } - return; - } - - if max_index > 0 { - let batch_size = - max_index.saturating_sub(this.durable_index.load(Ordering::Acquire)); - metrics::histogram!("core.raft.fsync.batch_entries").record(batch_size as f64); - } - - let result = if this.log_store.is_write_durable() { - Ok(()) - } else { - let t0 = std::time::Instant::now(); - let r = this.log_store.flush(); - let elapsed = t0.elapsed(); - metrics::histogram!("core.raft.fsync.duration_ms") - .record(elapsed.as_secs_f64() * 1_000.0); - metrics::counter!("core.raft.fsync.busy_nanos_total") - .increment(elapsed.as_nanos() as u64); - r - }; - - // Fence check: if a reset happened while this batch was in flight, - // its result is for data that no longer exists β€” discard. - if self.generation.load(Ordering::Acquire) != gen_at_start { - for reply in replies { - let _ = reply.send(Err(crate::Error::Fatal( - "stale fsync generation, superseded by reset".into(), - ))); - } - continue; // do NOT call advance_durable_and_notify - } - - match &result { - Ok(()) => this.advance_durable_and_notify(max_index), - Err(e) => { - // One fsync failure = fatal, no threshold, no retry-and-hope. - // Durability state is now unknown, this node - // must stop promising any further persistence. - this.mark_poisoned_and_notify(format!("fsync failed: {e:?}")); // mirrors advance_durable_and_notify's pattern - error!( - "WAL fsync failed at index {}: {:?} β€” node entering fatal state", - max_index, e - ); - } - } - - for reply in replies { - let _ = reply.send(match &result { - Ok(()) => Ok(()), - Err(e) => Err(Error::Fatal(format!("WAL fsync failed: {:?}", e))), - }); - } - } - } - - /// Called from reset_internal() before clearing in-memory state. - /// Bumps generation to fence the in-flight physical flush (if any), - /// AND drains anything already queued but not yet picked up by a - /// flush round β€” that queued data was submitted before reset and - /// must not be silently adopted by the next round. - pub(super) fn fence_reset(&self) { - self.generation.fetch_add(1, Ordering::AcqRel); - self.pending_max.store(0, Ordering::Release); - let stale = std::mem::take(&mut *self.pending_replies.lock().unwrap()); - for reply in stale { - let _ = reply.send(Err(Error::Fatal( - "stale fsync generation, superseded by reset".into(), - ))); - } - } -} - -#[cfg(test)] -#[path = "fsync_coordinator_test.rs"] -mod tests; diff --git a/d-engine-core/src/storage/fsync_coordinator_test.rs b/d-engine-core/src/storage/fsync_coordinator_test.rs deleted file mode 100644 index 76cf0bf7..00000000 --- a/d-engine-core/src/storage/fsync_coordinator_test.rs +++ /dev/null @@ -1,503 +0,0 @@ -//! Direct, isolated unit tests for `FsyncCoordinator`. -//! -//! This module is a child of `fsync_coordinator` (see the `#[path = ...] -//! mod tests;` declaration at the bottom of `fsync_coordinator.rs`), so it can -//! construct `FsyncCoordinator` directly and inspect its private fields -//! (`inflight`/`pending_max`/`pending_replies`/`generation`) without going -//! through `BufferedRaftLog::append_entries`/`write_notify`/`batch_processor` -//! at all β€” no IO thread, no `raft_log.start(...)`, most tests need no -//! `tokio` runtime either (`run_until_caught_up` is a plain sync fn). -//! -//! Scope: protocol/logic correctness of `FsyncCoordinator`'s own state -//! machine. Not performance β€” see `benches/` for throughput regression -//! guards. - -use super::*; -use crate::FlushPolicy; -use crate::MockLogStore; -use crate::MockMetaStore; -use crate::MockStorageEngine; -use crate::MockTypeConfig; -use crate::PersistenceConfig; -use crate::PersistenceStrategy; -use crate::Result; -use std::sync::Arc; - -/// Build a `BufferedRaftLog` for direct `FsyncCoordinator` method calls β€” -/// never `.start()`-ed, no IO thread, no channel plumbing. Only `log_store`/ -/// `durable_index`/`advance_durable_and_notify` are ever touched by the -/// methods under test here. -fn minimal_raft_log(storage: MockStorageEngine) -> Arc> { - let (raft_log, _receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - Arc::new(storage), - ); - Arc::new(raft_log) -} - -// ── Initial state ────────────────────────────────────────────────────────── - -/// `FsyncCoordinator::new()` starts with no work pending and no fence armed. -/// -/// Expected: -/// - `inflight == false`, `pending_max == 0`, `pending_replies` empty, -/// `generation == 0`. -#[test] -fn test_new_initializes_empty_state() { - let coord = FsyncCoordinator::new(); - - assert!( - !coord.inflight.load(Ordering::Acquire), - "inflight must start false" - ); - assert_eq!( - coord.pending_max.load(Ordering::Acquire), - 0, - "pending_max must start at 0" - ); - assert!( - coord.pending_replies.lock().unwrap().is_empty(), - "pending_replies must start empty" - ); - assert_eq!( - coord.generation.load(Ordering::Acquire), - 0, - "generation must start at 0" - ); -} - -// ── fence_reset() ──────────────────────────────────────────────────────────── - -/// `fence_reset()` zeroes `pending_max` β€” any batch size recorded before -/// reset must not leak into a post-reset round. -/// -/// Expected: -/// - After manually setting `pending_max` to a non-zero value and calling -/// `fence_reset()`, `pending_max` reads back as `0`. -#[test] -fn test_fence_reset_zeroes_pending_max() { - let coord = FsyncCoordinator::new(); - // Directly seed pending_max β€” no need to go through submit() (which would - // spawn a real background task and race with fence_reset() below). - coord.pending_max.store(10, Ordering::Release); - - coord.fence_reset(); - - assert_eq!( - coord.pending_max.load(Ordering::Acquire), - 0, - "fence_reset() must zero pending_max" - ); -} - -/// `fence_reset()` drains `pending_replies` and answers each with `Err` β€” -/// callers queued before reset must not be silently dropped nor receive a -/// stale `Ok(())`. -/// -/// Expected: -/// - Manually push several oneshot senders into `pending_replies`, call -/// `fence_reset()`, assert every corresponding receiver resolves to -/// `Err(..)`. -/// - `pending_replies` is empty after the call. -#[test] -fn test_fence_reset_drains_pending_replies_with_err() { - let coord = FsyncCoordinator::new(); - - let (tx1, mut rx1) = oneshot::channel(); - let (tx2, mut rx2) = oneshot::channel(); - coord.pending_replies.lock().unwrap().extend([tx1, tx2]); - - coord.fence_reset(); - - assert!( - coord.pending_replies.lock().unwrap().is_empty(), - "pending_replies must be empty after fence_reset()" - ); - assert!( - rx1.try_recv() - .expect("tx1 must have been answered, not silently dropped") - .is_err(), - "queued reply must receive Err, not a silent drop or stale Ok" - ); - assert!( - rx2.try_recv() - .expect("tx2 must have been answered, not silently dropped") - .is_err(), - "queued reply must receive Err, not a silent drop or stale Ok" - ); -} - -/// `fence_reset()` increments `generation` β€” this is the fence itself; any -/// round whose `gen_at_start` predates this call must be recognized as stale. -/// -/// Expected: -/// - `generation` after the call is exactly one more than before. -/// - Calling `fence_reset()` twice in a row increments it twice (no -/// accidental no-op / debounce). -#[test] -fn test_fence_reset_increments_generation() { - let coord = FsyncCoordinator::new(); - - coord.fence_reset(); - assert_eq!( - coord.generation.load(Ordering::Acquire), - 1, - "first fence_reset() must bump generation to 1" - ); - coord.fence_reset(); - assert_eq!( - coord.generation.load(Ordering::Acquire), - 2, - "second fence_reset() must bump generation to 2" - ); -} - -/// `fence_reset()` is safe to call with nothing pending (no in-flight round, -/// no queued replies) β€” e.g. `reset()` called on a freshly created log that -/// never wrote anything. -/// -/// Expected: -/// - No panic. `generation` still increments. `pending_max` stays `0`. -#[test] -fn test_fence_reset_is_safe_with_nothing_pending() { - let coord = FsyncCoordinator::new(); - - // No panic expected from this call β€” nothing pending to drain/zero. - coord.fence_reset(); - - assert_eq!( - coord.generation.load(Ordering::Acquire), - 1, - "generation must still increment even with nothing pending" - ); - assert_eq!( - coord.pending_max.load(Ordering::Acquire), - 0, - "pending_max must stay 0" - ); - assert!( - coord.pending_replies.lock().unwrap().is_empty(), - "pending_replies must stay empty" - ); -} - -// ── submit() ───────────────────────────────────────────────────────────── - -/// `submit()` records `max_index` via `fetch_max`, not last-write-wins β€” a -/// smaller, later `max_index` must not regress the recorded high-water mark. -/// -/// Expected: -/// - `submit(.., 100, vec![])` then `submit(.., 50, vec![])` leaves -/// `pending_max == 100`, not `50`. -#[test] -fn test_submit_pending_max_uses_fetch_max_not_last_write() { - let coord = Arc::new(FsyncCoordinator::new()); - let raft_log = minimal_raft_log(MockStorageEngine::new()); - - // Pretend a round is already in flight so submit() just records state - // instead of actually spawning a background task β€” keeps this test - // deterministic, no real concurrency needed to verify fetch_max order. - coord.inflight.store(true, Ordering::Release); - - coord.submit(&raft_log, 100, vec![]); - coord.submit(&raft_log, 50, vec![]); - - assert_eq!( - coord.pending_max.load(Ordering::Acquire), - 100, - "pending_max must stay at the high-water mark (100), not regress to \ - a later, smaller submit() value (50)" - ); -} - -/// The first `submit()` call (no round in flight) wins the CAS and flips -/// `inflight` to `true` synchronously β€” the caller doesn't need to wait for -/// the spawned task to observe this. -/// -/// Expected: -/// - Immediately after the first `submit()` call returns, `inflight` reads -/// `true`. -#[tokio::test] -async fn test_submit_first_call_sets_inflight_true() { - // Gated so the spawned task stays stuck in flush() β€” guarantees it can't - // race ahead and clear `inflight` again before our assertion below runs. - let (storage, _flush_gate) = - MockStorageEngine::not_durable_gated_flush("submit_first_call_sets_inflight_true".into()); - let coord = Arc::new(FsyncCoordinator::new()); - let raft_log = minimal_raft_log(storage); - - coord.submit(&raft_log, 1, vec![]); - - assert!( - coord.inflight.load(Ordering::Acquire), - "inflight must be true immediately after the first submit() call wins the CAS" - ); - // _flush_gate is dropped here without ever being sent β€” the spawned - // task's gate.recv() returns Err (sender dropped) and its flush() call - // proceeds to return Ok(()), so nothing hangs at test teardown. -} - -/// A `submit()` call while a round is already in flight must not spawn a -/// second competing task β€” it only records `pending_max`/`pending_replies` -/// for the running task to pick up. -/// -/// Expected: -/// - With `inflight` manually pre-set to `true`, calling `submit()` updates -/// `pending_max`/`pending_replies` as normal. -/// - The underlying mock's `flush()` call count does not increase (no new -/// physical fsync was triggered by this call). -#[test] -fn test_submit_second_call_does_not_spawn_second_task_while_inflight() { - let (storage, flush_call_count) = MockStorageEngine::not_durable( - "submit_second_call_does_not_spawn_second_task_while_inflight".into(), - ); - let coord = Arc::new(FsyncCoordinator::new()); - let raft_log = minimal_raft_log(storage); - - // Manually pre-set inflight β€” simulates "a round is already running", so - // this submit() call must lose the CAS and only record state, matching - // the doc comment's scenario directly without needing a real first round - // to actually be dispatched. - coord.inflight.store(true, Ordering::Release); - - let (tx, mut rx) = oneshot::channel::>(); - coord.submit(&raft_log, 1, vec![tx]); - - assert_eq!( - coord.pending_max.load(Ordering::Acquire), - 1, - "submit() must still record pending_max even though it lost the CAS" - ); - assert_eq!( - coord.pending_replies.lock().unwrap().len(), - 1, - "submit() must still queue the reply even though it lost the CAS" - ); - assert!( - rx.try_recv().is_err(), - "the queued reply must still be pending β€” nobody has processed it yet" - ); - assert_eq!( - flush_call_count.load(Ordering::Acquire), - 0, - "no physical flush() call should have been triggered β€” losing the CAS \ - must not spawn a competing task" - ); -} - -// ── run_until_caught_up() ──────────────────────────────────────────────── - -/// With nothing pending, `run_until_caught_up` clears `inflight` and returns -/// immediately β€” no physical `flush()` call. -/// -/// Expected: -/// - `inflight` was `true` (simulating a just-won CAS with no work), -/// `pending_max == 0`, `pending_replies` empty. -/// - After the call: `inflight == false`. The mock's `flush()` is never -/// called. -#[test] -fn test_run_until_caught_up_returns_immediately_when_nothing_pending() { - let (storage, flush_call_count) = MockStorageEngine::not_durable( - "run_until_caught_up_returns_immediately_when_nothing_pending".into(), - ); - let coord = FsyncCoordinator::new(); - let raft_log = minimal_raft_log(storage); - - // Simulate having just won the CAS in submit() β€” pending_max/pending_replies - // stay at their default empty state, matching the "nothing pending" scenario. - coord.inflight.store(true, Ordering::Release); - - coord.run_until_caught_up(&raft_log); - - assert!( - !coord.inflight.load(Ordering::Acquire), - "inflight must be cleared when there's nothing to do" - ); - assert_eq!( - flush_call_count.load(Ordering::Acquire), - 0, - "no physical flush() call should happen when nothing is pending" - ); -} - -/// A successful physical flush advances `durable_index` to `max_index`. -/// -/// Expected: -/// - After manually seeding `pending_max`/`inflight` and calling -/// `run_until_caught_up` against a mock whose `flush()` returns `Ok(())`, -/// `raft_log.durable_index()` equals the seeded `max_index`. -#[test] -fn test_run_until_caught_up_advances_durable_index_on_success() { - let (storage, _flush_call_count) = MockStorageEngine::not_durable( - "run_until_caught_up_advances_durable_index_on_success".into(), - ); - let coord = FsyncCoordinator::new(); - let raft_log = minimal_raft_log(storage); - - coord.inflight.store(true, Ordering::Release); - coord.pending_max.store(5, Ordering::Release); - - coord.run_until_caught_up(&raft_log); - - assert_eq!( - raft_log.durable_index.load(Ordering::Acquire), - 5, - "a successful flush must advance durable_index to the round's max_index" - ); -} - -/// A failed physical flush must NOT advance `durable_index` β€” the data was -/// never confirmed durable. -/// -/// Expected: -/// - With a mock `flush()` returning `Err(..)`, `durable_index()` stays at -/// its pre-call value after `run_until_caught_up` returns. -#[test] -fn test_run_until_caught_up_does_not_advance_durable_index_on_flush_failure() { - let storage = MockStorageEngine::not_durable_always_failing_flush( - "run_until_caught_up_does_not_advance_durable_index_on_flush_failure".into(), - ); - let coord = FsyncCoordinator::new(); - let raft_log = minimal_raft_log(storage); - let pre = raft_log.durable_index.load(Ordering::Acquire); - - coord.inflight.store(true, Ordering::Release); - coord.pending_max.store(5, Ordering::Release); - - coord.run_until_caught_up(&raft_log); - - assert_eq!( - raft_log.durable_index.load(Ordering::Acquire), - pre, - "a failed flush must not advance durable_index" - ); -} - -/// A failed physical flush sends `Err` (not a hang, not `Ok`) to every -/// queued reply. -/// -/// Expected: -/// - Every oneshot receiver corresponding to the queued replies resolves -/// to `Err(..)`. -#[test] -fn test_run_until_caught_up_sends_err_to_replies_on_flush_failure() { - let storage = MockStorageEngine::not_durable_always_failing_flush( - "run_until_caught_up_sends_err_to_replies_on_flush_failure".into(), - ); - let coord = FsyncCoordinator::new(); - let raft_log = minimal_raft_log(storage); - - let (tx, mut rx) = oneshot::channel::>(); - coord.inflight.store(true, Ordering::Release); - coord.pending_max.store(5, Ordering::Release); - coord.pending_replies.lock().unwrap().push(tx); - - coord.run_until_caught_up(&raft_log); - - assert!( - rx.try_recv() - .expect("reply must have been answered, not silently dropped") - .is_err(), - "a failed flush must send Err to queued replies, not hang or Ok" - ); -} - -/// The reset fence: if `generation` no longer matches what it was when this -/// round started, the physical flush's result must be discarded β€” not -/// applied to `durable_index`, not reported as `Ok` to callers. -/// -/// Uses a mock `flush()` that bumps the coordinator's `generation` as a side -/// effect of being called β€” a deterministic, thread-free way to simulate -/// "a reset happened while this batch was physically flushing", instead of -/// racing real threads against a gate. -/// -/// Expected: -/// - `durable_index()` does not advance to the round's `max_index`. -/// - The round's queued replies resolve to `Err(..)`, not `Ok(())`. -#[test] -fn test_run_until_caught_up_discards_stale_generation_result_without_advancing() { - let coord = Arc::new(FsyncCoordinator::new()); - let coord_in_mock = Arc::clone(&coord); - - let mut mock_log_store = MockLogStore::new(); - let mock_meta_store = MockMetaStore::new(); - mock_log_store.expect_last_index().returning(|| 0); - mock_log_store.expect_load_purge_boundary().returning(|| Ok(None)); - mock_log_store.expect_is_write_durable().returning(|| false); - mock_log_store.expect_flush().returning(move || { - // Simulates "a reset happened while this batch was physically - // flushing" β€” deterministic, no real concurrency/gate needed. - coord_in_mock.generation.fetch_add(1, Ordering::AcqRel); - Ok(()) - }); - let storage = MockStorageEngine::from(mock_log_store, mock_meta_store); - let raft_log = minimal_raft_log(storage); - - let (tx, mut rx) = oneshot::channel::>(); - coord.inflight.store(true, Ordering::Release); - coord.pending_max.store(5, Ordering::Release); - coord.pending_replies.lock().unwrap().push(tx); - - coord.run_until_caught_up(&raft_log); - - assert_eq!( - raft_log.durable_index.load(Ordering::Acquire), - 0, - "a stale-generation result must not advance durable_index" - ); - assert!( - rx.try_recv() - .expect("reply must have been answered, not silently dropped") - .is_err(), - "a stale-generation result must send Err, not a resurrected Ok(())" - ); -} - -/// Multiple `submit()` calls made while a round is in flight are coalesced -/// into a single subsequent physical `flush()` call by the same task β€” not -/// one physical flush per `submit()` call. -/// -/// Expected: -/// - Two `submit()` calls queued behind one manually-simulated in-flight -/// round, followed by one `run_until_caught_up` pass, result in exactly -/// one additional physical `flush()` call serving both. -#[test] -fn test_run_until_caught_up_coalesces_queued_submits_into_one_flush() { - let (storage, flush_call_count) = MockStorageEngine::not_durable( - "run_until_caught_up_coalesces_queued_submits_into_one_flush".into(), - ); - let coord = FsyncCoordinator::new(); - let raft_log = minimal_raft_log(storage); - - // Simulate two submit() calls that both lost the CAS while a round was - // in flight β€” both just accumulated into the same pending state. - let (tx1, mut rx1) = oneshot::channel::>(); - let (tx2, mut rx2) = oneshot::channel::>(); - coord.inflight.store(true, Ordering::Release); - coord.pending_max.store(10, Ordering::Release); - coord.pending_replies.lock().unwrap().extend([tx1, tx2]); - - coord.run_until_caught_up(&raft_log); - - assert_eq!( - flush_call_count.load(Ordering::Acquire), - 1, - "two queued submissions must be served by exactly one physical flush() call" - ); - assert!( - rx1.try_recv().expect("tx1 must have been answered").is_ok(), - "both queued replies must resolve to Ok" - ); - assert!( - rx2.try_recv().expect("tx2 must have been answered").is_ok(), - "both queued replies must resolve to Ok" - ); -} diff --git a/d-engine-core/src/storage/fsync_worker.rs b/d-engine-core/src/storage/fsync_worker.rs new file mode 100644 index 00000000..07b87dec --- /dev/null +++ b/d-engine-core/src/storage/fsync_worker.rs @@ -0,0 +1,240 @@ +use crate::Error; +use crate::InternalEvent; +use crate::LogStore; +use crate::Result; +use d_engine_proto::common::LogId; +use parking_lot::Mutex; +use std::sync::Arc; +use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; +use tokio::sync::mpsc; +use tokio::sync::oneshot; +use tokio::time::Instant; +use tracing::error; + +/// Schedules physical fsync calls β€” batches concurrent requests into one +/// flush() at a time. The final content check is `RaftLogCore:: +/// try_advance_durable_index`; this only orders the pending mark term-first. +pub(super) struct FsyncWorker { + node_id: u32, + inflight: AtomicBool, + /// Highest `(term, index)` awaiting fsync. Term-first: a newer term's mark + /// wins over an older term's higher index, so a stale pre-truncation submit + /// cannot swallow the valid post-truncation one. + pending_max: Mutex, + pending_replies: Mutex>>>, + + /// Bumped on truncation/reset. A round whose start predates the bump errs + /// its queued flush() replies instead of reporting a superseded result. + generation: AtomicU64, + + log_store: Arc, + poisoned: Arc, + internal_event_tx: Option>, + + // How far the *previous* completed fsync round reached + last_synced_index: AtomicU64, +} + +impl FsyncWorker { + pub(super) fn new( + node_id: u32, + log_store: Arc, + poisoned: Arc, + internal_event_tx: Option>, + ) -> Self { + Self { + node_id, + inflight: AtomicBool::new(false), + pending_max: Mutex::new(LogId::default()), + pending_replies: Mutex::new(Vec::new()), + generation: AtomicU64::new(0), + log_store, + poisoned, + internal_event_tx, + last_synced_index: AtomicU64::new(0), + } + } + + /// Called from the IO thread on every wakeup. Records new work and, if no + /// fsync task is currently running, kicks one off. Never spawns a second + /// concurrent task β€” additional calls while one is in flight just update + /// the pending state for it to pick up next round. + pub(super) fn submit( + self: &Arc, + mark: LogId, + replies: Vec>>, + ) { + if mark.index > 0 { + let mut p = self.pending_max.lock(); + if (mark.term, mark.index) > (p.term, p.index) { + *p = mark; + } + } + if !replies.is_empty() { + self.pending_replies.lock().extend(replies); + } + + if self + .inflight + .compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire) + .is_err() + { + metrics::counter!("core.raft.fsync.coalesced_submit").increment(1); + return; // Already running β€” it will pick up what we just recorded. + } + metrics::counter!("core.raft.fsync.fresh_round").increment(1); + + metrics::gauge!("core.raft.fsync.inflight").set(1.0); + + let worker = Arc::clone(self); + tokio::task::spawn_blocking(move || worker.run_until_caught_up()); + } + + /// Runs on the blocking pool. Keeps fsyncing and re-checking for newly + /// accumulated work until there's nothing left, then clears `inflight`. + pub(super) fn run_until_caught_up(&self) { + loop { + let gen_at_start = self.generation.load(Ordering::Acquire); + + let mark = std::mem::take(&mut *self.pending_max.lock()); + let replies = std::mem::take(&mut *self.pending_replies.lock()); + + if self.is_poisoned() { + for reply in replies { + let _ = reply.send(Err(Error::Fatal("raft log storage is poisoned".into()))); + } + self.inflight.store(false, Ordering::Release); + metrics::gauge!("core.raft.fsync.inflight").set(0.0); + return; + } + + if mark.index == 0 && replies.is_empty() { + self.inflight.store(false, Ordering::Release); + metrics::gauge!("core.raft.fsync.inflight").set(0.0); + // Re-check: something may have slipped in between the swap + // above and clearing `inflight`. If so, re-arm. + if (self.pending_max.lock().index > 0 || !self.pending_replies.lock().is_empty()) + && self + .inflight + .compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire) + .is_ok() + { + metrics::gauge!("core.raft.fsync.inflight").set(1.0); + continue; + } + return; + } + + if mark.index > 0 { + let prev = self.last_synced_index.load(Ordering::Acquire); + let batch_size = mark.index.saturating_sub(prev); + metrics::histogram!("core.raft.fsync.batch_entries").record(batch_size as f64); + } + + let result = if self.log_store.is_write_durable() { + Ok(()) + } else { + let t0 = std::time::Instant::now(); + let r = self.log_store.flush(); + let elapsed = t0.elapsed(); + metrics::histogram!("core.raft.fsync.duration_ms") + .record(elapsed.as_secs_f64() * 1_000.0); + metrics::counter!("core.raft.fsync.busy_nanos_total") + .increment(elapsed.as_nanos() as u64); + r + }; + + // Before the generation fence: a truncation can't make a failed fsync harmless. + if let Err(e) = &result { + self.mark_poisoned_and_notify(format!("fsync failed: {e:?}")); + error!( + ?self.node_id, + "WAL fsync failed at index {}: {:?} β€” node entering fatal state", + mark.index, e + ); + for reply in replies { + let _ = reply.send(Err(Error::Fatal(format!("WAL fsync failed: {:?}", e)))); + } + continue; // next round sees `poisoned` and drains the rest + } + + // Skip replying if this round is already known stale. + if self.generation.load(Ordering::Acquire) != gen_at_start { + for reply in replies { + let _ = reply.send(Err(crate::Error::Fatal( + "stale fsync generation, superseded by reset".into(), + ))); + } + continue; // do not publish FsyncCompleted + } + + self.last_synced_index.store(mark.index, Ordering::Release); + self.notify_fsync_completed(mark); + + for reply in replies { + let _ = reply.send(Ok(())); + } + } + } + + /// Called from reset_internal() before clearing in-memory state. + /// Bumps generation to fence the in-flight physical flush (if any), + /// AND drains anything already queued but not yet picked up by a + /// flush round β€” that queued data was submitted before reset and + /// must not be silently adopted by the next round. + pub(super) fn fence_reset(&self) { + *self.pending_max.lock() = LogId::default(); + let stale = std::mem::take(&mut *self.pending_replies.lock()); + for reply in stale { + let _ = reply.send(Err(Error::Fatal( + "stale fsync generation, superseded by reset".into(), + ))); + } + self.bump_generation(); + } + + pub(super) fn bump_generation(&self) { + self.generation.fetch_add(1, Ordering::AcqRel); + } + + fn is_poisoned(&self) -> bool { + self.poisoned.load(Ordering::Acquire) + } + + fn notify_fsync_completed( + &self, + mark: LogId, + ) { + if let Some(tx) = &self.internal_event_tx { + let _ = tx.send(InternalEvent::FsyncCompleted { + mark, + sent_at: Instant::now(), + }); + } + } + + fn mark_poisoned_and_notify( + &self, + error: String, + ) { + self.poisoned.store(true, Ordering::Release); + if let Some(tx) = &self.internal_event_tx + && tx + .send(InternalEvent::FatalError { + source: "RaftLog".into(), + error, + }) + .is_err() + { + error!( + ?self.node_id, + "FatalError delivery failed (channel closed) β€” node may be poisoned \ + but still running; durability is no longer guaranteed" + ); + } + } +} + +#[cfg(test)] +#[path = "fsync_worker_test.rs"] +mod fsync_worker_test; diff --git a/d-engine-core/src/storage/fsync_worker_test.rs b/d-engine-core/src/storage/fsync_worker_test.rs new file mode 100644 index 00000000..442538e4 --- /dev/null +++ b/d-engine-core/src/storage/fsync_worker_test.rs @@ -0,0 +1,538 @@ +//! Direct, isolated unit tests for `FsyncWorker`. +//! +//! This module is a child of `fsync_worker` (see the `#[path = ...] mod +//! fsync_worker_test;` at the bottom of `fsync_worker.rs`), so it can construct +//! `FsyncWorker` directly and inspect its private fields (`inflight` / +//! `pending_max` / `pending_replies` / `generation`) without going through +//! `RaftLogCore`'s append/flush pipeline at all. `run_until_caught_up` is a +//! plain sync fn, so most tests need no tokio runtime. +//! +//! Scope: protocol/logic correctness of `FsyncWorker`'s own state machine. +//! Not performance β€” see `benches/` for throughput regression guards. + +use super::*; +use crate::InternalEvent; +use crate::MockLogStore; +use crate::MockStorageEngine; +use crate::Result; +use crate::StorageEngine; +use d_engine_proto::common::LogId; +use std::sync::Arc; +use std::sync::atomic::{AtomicBool, Ordering}; +use tokio::sync::{mpsc, oneshot}; + +/// Build a bare `FsyncWorker` over a mock log store β€” no tokio runtime, no +/// `RaftLogCore`. `internal_event_tx` is `None`, so the worker never sends +/// `FsyncCompleted`; tests that need to observe completion wire their own. +fn minimal_worker(log_store: Arc) -> Arc> { + Arc::new(FsyncWorker::new( + 1, + log_store, + Arc::new(AtomicBool::new(false)), + None, + )) +} + +// ── Initial state ────────────────────────────────────────────────────────── + +/// `FsyncWorker::new()` starts with no work pending and no fence armed. +/// +/// Expected: +/// - `inflight == false`, `pending_max == 0`, `pending_replies` empty, +/// `generation == 0`. +#[test] +fn test_new_initializes_empty_state() { + let worker = minimal_worker(Arc::new(MockLogStore::new())); + + assert!( + !worker.inflight.load(Ordering::Acquire), + "inflight must start false" + ); + assert_eq!( + worker.pending_max.lock().index, + 0, + "pending_max must start at 0" + ); + assert!( + worker.pending_replies.lock().is_empty(), + "pending_replies must start empty" + ); + assert_eq!( + worker.generation.load(Ordering::Acquire), + 0, + "generation must start at 0" + ); +} + +// ── fence_reset() ──────────────────────────────────────────────────────────── + +/// `fence_reset()` zeroes `pending_max` β€” any batch size recorded before +/// reset must not leak into a post-reset round. +#[test] +fn test_fence_reset_zeroes_pending_max() { + let worker = minimal_worker(Arc::new(MockLogStore::new())); + *worker.pending_max.lock() = LogId { term: 1, index: 10 }; + + worker.fence_reset(); + + assert_eq!( + worker.pending_max.lock().index, + 0, + "fence_reset() must zero pending_max" + ); +} + +/// `fence_reset()` drains `pending_replies` and answers each with `Err` β€” +/// callers queued before reset must not be silently dropped nor receive a +/// stale `Ok(())`. +#[test] +fn test_fence_reset_drains_pending_replies_with_err() { + let worker = minimal_worker(Arc::new(MockLogStore::new())); + + let (tx1, mut rx1) = oneshot::channel(); + let (tx2, mut rx2) = oneshot::channel(); + worker.pending_replies.lock().extend([tx1, tx2]); + + worker.fence_reset(); + + assert!( + worker.pending_replies.lock().is_empty(), + "pending_replies must be empty after fence_reset()" + ); + assert!( + rx1.try_recv().expect("tx1 must have been answered").is_err(), + "queued reply must receive Err, not a silent drop or stale Ok" + ); + assert!( + rx2.try_recv().expect("tx2 must have been answered").is_err(), + "queued reply must receive Err, not a silent drop or stale Ok" + ); +} + +/// `fence_reset()` increments `generation` β€” the fence itself. +#[test] +fn test_fence_reset_increments_generation() { + let worker = minimal_worker(Arc::new(MockLogStore::new())); + + worker.fence_reset(); + assert_eq!( + worker.generation.load(Ordering::Acquire), + 1, + "first fence_reset() must bump to 1" + ); + worker.fence_reset(); + assert_eq!( + worker.generation.load(Ordering::Acquire), + 2, + "second fence_reset() must bump to 2" + ); +} + +/// `fence_reset()` is safe to call with nothing pending. +#[test] +fn test_fence_reset_is_safe_with_nothing_pending() { + let worker = minimal_worker(Arc::new(MockLogStore::new())); + + worker.fence_reset(); + + assert_eq!( + worker.generation.load(Ordering::Acquire), + 1, + "generation must still increment" + ); + assert_eq!( + worker.pending_max.lock().index, + 0, + "pending_max must stay 0" + ); + assert!( + worker.pending_replies.lock().is_empty(), + "pending_replies must stay empty" + ); +} + +// ── submit() ───────────────────────────────────────────────────────────── + +/// `submit()` records the high-water mark β€” a smaller, later `mark` must not +/// regress it. +#[test] +fn test_submit_pending_max_uses_fetch_max_not_last_write() { + let worker = minimal_worker(Arc::new(MockLogStore::new())); + + // Pretend a round is in flight so submit() only records state. + worker.inflight.store(true, Ordering::Release); + + worker.submit( + LogId { + term: 1, + index: 100, + }, + vec![], + ); + worker.submit(LogId { term: 1, index: 50 }, vec![]); + + assert_eq!( + worker.pending_max.lock().index, + 100, + "pending_max must stay at the high-water mark (100), not regress to 50" + ); +} + +/// The first `submit()` call wins the CAS and flips `inflight` to `true` +/// synchronously. +#[tokio::test] +async fn test_submit_first_call_sets_inflight_true() { + let (storage, _flush_gate) = + MockStorageEngine::not_durable_gated_flush("submit_first_call_sets_inflight_true".into()); + let worker = minimal_worker(storage.log_store()); + + worker.submit(LogId { term: 1, index: 1 }, vec![]); + + assert!( + worker.inflight.load(Ordering::Acquire), + "inflight must be true immediately after the first submit() wins the CAS" + ); +} + +/// A `submit()` call while a round is in flight must not spawn a second task. +#[test] +fn test_submit_second_call_does_not_spawn_second_task_while_inflight() { + let (storage, flush_call_count) = MockStorageEngine::not_durable( + "submit_second_call_does_not_spawn_second_task_while_inflight".into(), + ); + let worker = minimal_worker(storage.log_store()); + + worker.inflight.store(true, Ordering::Release); + + let (tx, mut rx) = oneshot::channel::>(); + worker.submit(LogId { term: 1, index: 1 }, vec![tx]); + + assert_eq!( + worker.pending_max.lock().index, + 1, + "submit() must record pending_max" + ); + assert_eq!( + worker.pending_replies.lock().len(), + 1, + "submit() must queue the reply" + ); + assert!(rx.try_recv().is_err(), "queued reply must still be pending"); + assert_eq!( + flush_call_count.load(Ordering::Acquire), + 0, + "losing the CAS must not trigger a physical flush" + ); +} + +// ── run_until_caught_up() ──────────────────────────────────────────────── + +/// With nothing pending, `run_until_caught_up` clears `inflight` and returns +/// without flushing. +#[test] +fn test_run_until_caught_up_returns_immediately_when_nothing_pending() { + let (storage, flush_call_count) = MockStorageEngine::not_durable( + "run_until_caught_up_returns_immediately_when_nothing_pending".into(), + ); + let worker = minimal_worker(storage.log_store()); + worker.inflight.store(true, Ordering::Release); + + worker.run_until_caught_up(); + + assert!( + !worker.inflight.load(Ordering::Acquire), + "inflight must be cleared" + ); + assert_eq!( + flush_call_count.load(Ordering::Acquire), + 0, + "no flush when nothing pending" + ); +} + +/// A successful physical flush sends `FsyncCompleted` for the round's mark. +#[test] +fn test_run_until_caught_up_sends_fsync_completed_on_success() { + let (storage, _flush_call_count) = MockStorageEngine::not_durable( + "run_until_caught_up_sends_fsync_completed_on_success".into(), + ); + let (tx, mut rx) = mpsc::unbounded_channel(); + let worker = Arc::new(FsyncWorker::new( + 1, + storage.log_store(), + Arc::new(AtomicBool::new(false)), + Some(tx), + )); + + worker.inflight.store(true, Ordering::Release); + *worker.pending_max.lock() = LogId { term: 1, index: 5 }; + + worker.run_until_caught_up(); + + match rx.try_recv().expect("must emit FsyncCompleted") { + InternalEvent::FsyncCompleted { mark, sent_at: _ } => { + assert_eq!( + mark, + LogId { term: 1, index: 5 }, + "must report the round's mark" + ); + } + other => panic!("expected FsyncCompleted, got {other:?}"), + } +} + +/// A failed physical flush does NOT send `FsyncCompleted` β€” it poisons and +/// notifies a `FatalError` instead. +#[test] +fn test_run_until_caught_up_does_not_send_fsync_completed_on_flush_failure() { + let storage = MockStorageEngine::not_durable_always_failing_flush( + "run_until_caught_up_does_not_send_fsync_completed_on_flush_failure".into(), + ); + let (tx, mut rx) = mpsc::unbounded_channel(); + let worker = Arc::new(FsyncWorker::new( + 1, + storage.log_store(), + Arc::new(AtomicBool::new(false)), + Some(tx), + )); + + worker.inflight.store(true, Ordering::Release); + *worker.pending_max.lock() = LogId { term: 1, index: 5 }; + + worker.run_until_caught_up(); + + while let Ok(event) = rx.try_recv() { + assert!( + !matches!(event, InternalEvent::FsyncCompleted { .. }), + "a failed flush must not emit FsyncCompleted, got {event:?}" + ); + } +} + +/// A failed physical flush sends `Err` (not a hang, not `Ok`) to queued replies. +#[test] +fn test_run_until_caught_up_sends_err_to_replies_on_flush_failure() { + let storage = MockStorageEngine::not_durable_always_failing_flush( + "run_until_caught_up_sends_err_to_replies_on_flush_failure".into(), + ); + let worker = minimal_worker(storage.log_store()); + + let (tx, mut rx) = oneshot::channel::>(); + worker.inflight.store(true, Ordering::Release); + *worker.pending_max.lock() = LogId { term: 1, index: 5 }; + worker.pending_replies.lock().push(tx); + + worker.run_until_caught_up(); + + assert!( + rx.try_recv().expect("reply must have been answered").is_err(), + "a failed flush must send Err to queued replies" + ); +} + +/// The reset fence: a round whose generation no longer matches at completion +/// must be discarded β€” no `FsyncCompleted`, replies get `Err`. +#[test] +fn test_run_until_caught_up_discards_stale_generation_result_without_advancing() { + // `Arc::new_cyclic` lets the mock's flush() closure bump the worker's own + // `generation` field β€” the worker owns the mock, so the closure needs a + // `Weak` upgraded at flush time. + let worker: Arc> = + Arc::new_cyclic(|weak: &std::sync::Weak>| { + let weak_in_mock = weak.clone(); + let mut mock_log_store = MockLogStore::new(); + mock_log_store.expect_is_write_durable().returning(|| false); + mock_log_store.expect_flush().returning(move || { + if let Some(w) = weak_in_mock.upgrade() { + w.generation.fetch_add(1, Ordering::AcqRel); + } + Ok(()) + }); + FsyncWorker::new( + 1, + Arc::new(mock_log_store), + Arc::new(AtomicBool::new(false)), + None, + ) + }); + + let (tx, mut rx) = oneshot::channel::>(); + worker.inflight.store(true, Ordering::Release); + *worker.pending_max.lock() = LogId { term: 1, index: 5 }; + worker.pending_replies.lock().push(tx); + + worker.run_until_caught_up(); + + assert!( + rx.try_recv().expect("reply must have been answered").is_err(), + "a stale-generation result must send Err, not a resurrected Ok" + ); +} + +/// A failed physical flush must poison the node even when a truncation / reset +/// bumps `generation` while the flush is in flight. +/// +/// Scenario: a follower's fdatasync for entries 1..=10 fails (EIO) at the same +/// moment a new leader's conflict truncation bumps `generation`. The stale-generation +/// fence only decides whether a *successful* result may be published; it must never +/// swallow a *failure*. Otherwise the node keeps running, a later fsync may report +/// success over pages the kernel already gave up on, and `durable_index` could +/// advance past data that is not on disk. +/// +/// Expected: +/// - `poisoned == true` +/// - a `FatalError` event is sent +/// - no `FsyncCompleted` is sent +#[test] +fn test_run_until_caught_up_poisons_on_flush_failure_even_when_generation_changed() { + let poisoned = Arc::new(AtomicBool::new(false)); + let (tx, mut rx) = mpsc::unbounded_channel(); + + let worker: Arc> = Arc::new_cyclic({ + let poisoned = poisoned.clone(); + move |weak: &std::sync::Weak>| { + let weak_in_mock = weak.clone(); + let mut mock_log_store = MockLogStore::new(); + mock_log_store.expect_is_write_durable().returning(|| false); + mock_log_store.expect_flush().returning(move || { + // Truncation lands while the fdatasync is in flight, then it fails. + if let Some(w) = weak_in_mock.upgrade() { + w.generation.fetch_add(1, Ordering::AcqRel); + } + Err(crate::Error::Fatal("simulated fdatasync EIO".into())) + }); + FsyncWorker::new(1, Arc::new(mock_log_store), poisoned, Some(tx)) + } + }); + + worker.inflight.store(true, Ordering::Release); + *worker.pending_max.lock() = LogId { term: 1, index: 10 }; + + worker.run_until_caught_up(); + + assert!( + poisoned.load(Ordering::Acquire), + "a failed fsync must poison the node even if generation changed mid-flight" + ); + + let mut saw_fatal = false; + while let Ok(event) = rx.try_recv() { + match event { + InternalEvent::FatalError { .. } => saw_fatal = true, + InternalEvent::FsyncCompleted { .. } => { + panic!("a failed flush must not emit FsyncCompleted") + } + _ => {} + } + } + assert!(saw_fatal, "a failed fsync must notify FatalError"); +} + +/// The other half of the fence: a matching generation accepts the result β€” +/// `FsyncCompleted` emitted, replies resolve `Ok`. +#[test] +fn test_run_until_caught_up_accepts_result_when_generation_unchanged() { + let (storage, _flush_call_count) = MockStorageEngine::not_durable( + "run_until_caught_up_accepts_result_when_generation_unchanged".into(), + ); + let (tx, mut rx) = mpsc::unbounded_channel(); + let worker = Arc::new(FsyncWorker::new( + 1, + storage.log_store(), + Arc::new(AtomicBool::new(false)), + Some(tx), + )); + + // Two unrelated fences happened earlier β€” generation is 2, not 0. + worker.fence_reset(); + worker.fence_reset(); + assert_eq!(worker.generation.load(Ordering::Acquire), 2); + + let (reply_tx, mut reply_rx) = oneshot::channel::>(); + worker.inflight.store(true, Ordering::Release); + *worker.pending_max.lock() = LogId { term: 1, index: 5 }; + worker.pending_replies.lock().push(reply_tx); + + worker.run_until_caught_up(); + + match rx.try_recv().expect("must emit FsyncCompleted") { + InternalEvent::FsyncCompleted { mark, sent_at: _ } => { + assert_eq!(mark, LogId { term: 1, index: 5 }); + } + other => panic!("expected FsyncCompleted, got {other:?}"), + } + assert!( + reply_rx.try_recv().expect("reply must have been answered").is_ok(), + "a matching generation must resolve replies as Ok" + ); +} + +/// Multiple queued `submit()`s are coalesced into one physical `flush()` call. +#[test] +fn test_run_until_caught_up_coalesces_queued_submits_into_one_flush() { + let (storage, flush_call_count) = MockStorageEngine::not_durable( + "run_until_caught_up_coalesces_queued_submits_into_one_flush".into(), + ); + let worker = minimal_worker(storage.log_store()); + + let (tx1, mut rx1) = oneshot::channel::>(); + let (tx2, mut rx2) = oneshot::channel::>(); + worker.inflight.store(true, Ordering::Release); + *worker.pending_max.lock() = LogId { term: 1, index: 10 }; + worker.pending_replies.lock().extend([tx1, tx2]); + + worker.run_until_caught_up(); + + assert_eq!( + flush_call_count.load(Ordering::Acquire), + 1, + "two submits must share one flush()" + ); + assert!( + rx1.try_recv().expect("tx1 answered").is_ok(), + "tx1 must resolve Ok" + ); + assert!( + rx2.try_recv().expect("tx2 answered").is_ok(), + "tx2 must resolve Ok" + ); +} + +/// Term-first ordering: a newer term's mark wins over an older term's higher +/// index, so a stale pre-truncation batch can't swallow the valid tail. +#[test] +fn test_submit_term_first_keeps_valid_mark_over_stale_higher_index() { + let (storage, _flush_call_count) = MockStorageEngine::not_durable( + "submit_term_first_keeps_valid_mark_over_stale_higher_index".into(), + ); + let (tx, mut rx) = mpsc::unbounded_channel(); + let worker = Arc::new(FsyncWorker::new( + 1, + storage.log_store(), + Arc::new(AtomicBool::new(false)), + Some(tx), + )); + + worker.inflight.store(true, Ordering::Release); + worker.submit(LogId { term: 1, index: 10 }, vec![]); // stale pre-truncation + worker.submit(LogId { term: 2, index: 2 }, vec![]); // valid post-truncation + + assert_eq!( + *worker.pending_max.lock(), + LogId { term: 2, index: 2 }, + "term-first: the newer-term mark must win over the stale higher index" + ); + + worker.run_until_caught_up(); + + match rx.try_recv().expect("must emit FsyncCompleted") { + InternalEvent::FsyncCompleted { mark, sent_at: _ } => { + assert_eq!( + mark, + LogId { term: 2, index: 2 }, + "must report the valid tail" + ); + } + other => panic!("expected FsyncCompleted, got {other:?}"), + } +} diff --git a/d-engine-core/src/storage/mod.rs b/d-engine-core/src/storage/mod.rs index 41886906..dd3bf7e2 100644 --- a/d-engine-core/src/storage/mod.rs +++ b/d-engine-core/src/storage/mod.rs @@ -1,20 +1,19 @@ -mod buffered_raft_log; -pub(super) mod fsync_coordinator; +pub(super) mod fsync_worker; mod lease; mod raft_log; +mod raft_log_core; mod snapshot_path_manager; mod state_machine; mod storage_engine; +pub use raft_log_core::*; -pub use buffered_raft_log::*; pub use lease::*; #[doc(hidden)] pub use raft_log::*; pub(crate) use snapshot_path_manager::*; pub use state_machine::*; pub use storage_engine::*; -#[cfg(test)] -mod buffered_raft_log_test; + #[cfg(test)] mod snapshot_path_manager_test; #[cfg(any(test, feature = "__test_support"))] diff --git a/d-engine-core/src/storage/raft_log.rs b/d-engine-core/src/storage/raft_log.rs index b21778bc..90e4c4f3 100644 --- a/d-engine-core/src/storage/raft_log.rs +++ b/d-engine-core/src/storage/raft_log.rs @@ -77,6 +77,17 @@ pub trait RaftLog: Send + Sync + 'static { /// - DiskFirst: equals `last_entry_id()` (every append blocks until durable). fn durable_index(&self) -> u64; + /// Content-validated durable-watermark advance. `index`/`term` describe + /// what a completed fsync claims is now safe β€” rejected (`None`) if + /// `entry_term(index) != Some(term)`, meaning the log content at that + /// index has changed (truncated + replaced) since fsync started on it. + /// `Some(new_value)` only when it actually advanced β€” callers use this + /// to decide whether to fire `handle_log_flushed`. + fn try_advance_durable_index( + &self, + mark: LogId, + ) -> Option; + /// Returns the LogId (term + index) of the last entry. /// /// # Returns @@ -209,14 +220,14 @@ pub trait RaftLog: Send + Sync + 'static { /// - Persist entries to durable storage BEFORE updating in-memory state /// - Call fsync/flush before returning Ok(()) /// - Ensures entries survive crashes immediately - /// - Example: BufferedRaftLog with PersistenceStrategy::DiskFirst + /// - Example: a store that fsyncs before returning /// /// 2. **Memory-First (Performance-optimized, Acceptable for Followers)**: /// - Update in-memory state first /// - Enqueue entries for asynchronous durability /// - MUST guarantee eventual durability via background flush /// - MUST call flush() before acknowledging commits - /// - Example: BufferedRaftLog with PersistenceStrategy::MemFirst + /// - Example: a store that fsyncs asynchronously /// - WARNING: Leader MUST wait_durable() before responding to AppendEntries RPCs /// /// # Safety Invariants @@ -228,10 +239,10 @@ pub trait RaftLog: Send + Sync + 'static { /// - MUST update term indexes (first/last_index_for_term) atomically /// /// # Raft Protocol Integration - /// - Leaders using MemFirst MUST call wait_durable(index) before: + /// - Leaders using async fsync MUST call wait_durable(index) before: /// * Responding success to AppendEntries RPC /// * Advancing commit index - /// - Followers can use MemFirst safely because leader durability guarantees safety + /// - Followers can use async fsync safely because leader durability guarantees safety /// /// # Failure Semantics /// - On error, implementer MAY roll back partial writes @@ -252,7 +263,7 @@ pub trait RaftLog: Send + Sync + 'static { /// /// # Usage Pattern /// ```rust,ignore - /// // Leader with MemFirst strategy + /// // Leader with async fsync /// raft_log.append_entries(new_entries).await?; /// raft_log.wait_durable(max_index).await?; // MUST wait before RPC response /// respond_to_client(Ok(())); @@ -261,7 +272,7 @@ pub trait RaftLog: Send + Sync + 'static { /// # Safety Invariants /// - MUST NOT return until flush() for this index completes successfully /// - If implementation doesn't support async durability, return Ok(()) immediately - /// - Critical for MemFirst strategy correctness + /// - Critical for async-fsync correctness async fn wait_durable( &self, index: u64, @@ -460,12 +471,6 @@ pub trait RaftLog: Send + Sync + 'static { hard_state: &crate::HardState, ) -> Result<()>; - /// Returns `true` if a storage-layer failure has permanently poisoned this - /// log β€” no further writes/commands will be attempted, and callers above - /// the storage layer (e.g. the Raft protocol loop) must stop dispatching - /// new work to this node. - fn is_poisoned(&self) -> bool; - /// Gracefully closes the log, ensuring all pending IO completes and any /// background IO threads have exited before returning. /// diff --git a/d-engine-core/src/storage/buffered_raft_log.rs b/d-engine-core/src/storage/raft_log_core.rs similarity index 52% rename from d-engine-core/src/storage/buffered_raft_log.rs rename to d-engine-core/src/storage/raft_log_core.rs index 713286e9..5bae1266 100644 --- a/d-engine-core/src/storage/buffered_raft_log.rs +++ b/d-engine-core/src/storage/raft_log_core.rs @@ -1,238 +1,29 @@ -//! High-performance buffered Raft log β€” notify-then-spawn-fsync architecture. -//! -//! ## Durability guarantee (Level 3 β€” concurrent pipeline, #422) -//! -//! **Level 3 (current)**: `db.write()` β†’ OS page cache β†’ concurrent `fdatasync` via `spawn_blocking` -//! - Power-loss safe: `durable_index` advances only after physical fsync completes -//! - Process crash safe: RocksDB WAL replay on restart -//! - Performance: IO thread never blocks on fsync β€” batches pipeline at OS/RocksDB layer -//! -//! **Level 2 (pre-#407)**: `db.write()` β†’ OS page cache only, no fdatasync -//! - Was the default; superseded by Level 3 for correctness -//! -//! ## Write path -//! -//! `append_entries` inserts entries into in-memory SkipMap, calls `write_notify.notify_one()`. -//! Multiple concurrent writers coalesce into a single IO thread wakeup. -//! -//! ## IO thread (notify-then-spawn-fsync) -//! -//! On wakeup from `write_notify`: -//! 1. **Read** β€” scan SkipMap range `(durable_index, max_index]` -//! 2. **Persist** β€” write range to OS page cache via `persist_entries` -//! 3. **Spawn fsync** β€” dispatch fdatasync to `spawn_blocking` pool via `spawn_fsync`, return immediately -//! 4. **Loop** β€” back to `select!` for next wakeup; prior fsync runs concurrently in pool -//! -//! `durable_index` is advanced inside the blocking task via `advance_durable_and_notify` -//! (`fetch_max`, AcqRel). Multiple concurrent tasks completing out of order are safe: -//! a late-arriving lower index is a no-op. -//! -//! ## Fsync triggers -//! -//! All four triggers below funnel through `run_batch_turn` (persist pending -//! entries, drain any queued commands, then dispatch) into the single -//! `FsyncCoordinator::submit()` entry point β€” there is no separate inline path. -//! -//! 1. **Notify-driven** (normal): `write_notify` β†’ `run_batch_turn` (no reply) -//! 2. **Explicit** (flush API): `flush()` β†’ `IOTask::Flush(tx)` β†’ `run_batch_turn` with reply sender -//! 3. **Idle timer** (safety net): `idle_flush_interval_ms` elapsed β†’ `persist_pending_range` + `submit()` -//! 4. **Shutdown**: `IOTask::Shutdown` β†’ `run_batch_turn`, then `close()` waits (bounded by -//! `shutdown_timeout_ms`) for the IO thread's runtime to drain any in-flight fsync task -//! -//! ## Durability contract -//! -//! `durable_index` advances only after physical fdatasync in the blocking task. -//! Concurrent fsyncs coalesce at the storage layer: if batch B's fsync covers A's WAL -//! position, A's `flush_wal` returns fast with no extra disk IO β€” storage-layer group commit. - -use super::fsync_coordinator::FsyncCoordinator; -use crate::Error; -use crate::FlushPolicy; -use crate::HardState; -use crate::LogStore; -use crate::MetaStore; -use crate::NetworkError; -use crate::PersistenceConfig; -use crate::RaftLog; -use crate::Result; -use crate::StorageEngine; -use crate::TypeConfig; -use crate::alias::SOF; -use crate::scoped_timer::ScopedTimer; -use async_trait::async_trait; +//! Direct, single-threaded implementation of `RaftLog`: `persist_entries`/ +//! `replace_range`/`purge`/`reset` execute inline on the caller's own task β€” +//! no dedicated OS thread, no command channel. Only the physical fsync is +//! handed off, to `FsyncWorker`'s own execution context (see fsync_worker.rs). + +use crate::{ + Error, HardState, InternalEvent, LogStore, MetaStore, NetworkError, RaftLog, Result, + StorageEngine, TypeConfig, alias::SOF, fsync_worker::FsyncWorker, scoped_timer::ScopedTimer, +}; use crossbeam_skiplist::SkipMap; -use d_engine_proto::common::Entry; -use d_engine_proto::common::LogId; +use d_engine_proto::common::{Entry, LogId}; use parking_lot::RwLock; -use std::collections::HashMap; -use std::ops::RangeInclusive; -use std::sync::Arc; -use std::sync::atomic::AtomicBool; -use std::sync::atomic::AtomicU64; -use std::sync::atomic::AtomicUsize; -use std::sync::atomic::Ordering; -use std::time::Duration; -use tokio::sync::Notify; -use tokio::sync::mpsc; -use tokio::sync::oneshot; -use tracing::debug; -use tracing::error; -use tracing::warn; - -/// Maximum number of historical term segments (one per leader election). -/// 1024 is far more than any realistic cluster lifetime. -const MAX_TERM_SEGMENTS: usize = 1024; - -/// Compact term boundary index for O(1) `entry_term()` lookups. -/// -/// Hot path (99%+ of queries): two `Acquire` loads, no lock, no CAS. -/// Cold path (election recovery): atomic array reverse-scan, no lock, no unsafe. -/// -/// When the array is full, new segments are silently dropped and `entry_term()` -/// falls back to the SkipMap (O(log n)) β€” correct but slower. -/// -/// Memory: 2 AtomicU64 + 1 AtomicUsize + 2Γ—1024 AtomicU64 = ~16KB, all inline. -pub(crate) struct TermSegments { - /// Term of the most-recently appended segment. - pub(crate) last_term: AtomicU64, - /// First log index belonging to `last_term`. - pub(crate) last_term_start: AtomicU64, - /// Number of valid historical segments stored in `seg_starts`/`seg_terms`. - seg_count: AtomicUsize, - /// First index of each historical segment, in append order. - seg_starts: [AtomicU64; MAX_TERM_SEGMENTS], - /// Term of each historical segment, parallel to `seg_starts`. - seg_terms: [AtomicU64; MAX_TERM_SEGMENTS], -} - -impl TermSegments { - pub(crate) fn new() -> Self { - Self { - last_term: AtomicU64::new(0), - last_term_start: AtomicU64::new(0), - seg_count: AtomicUsize::new(0), - seg_starts: std::array::from_fn(|_| AtomicU64::new(0)), - seg_terms: std::array::from_fn(|_| AtomicU64::new(0)), - } - } - - /// Return the term for `index`, or `None` if the log is empty or index is out of range. - /// - /// Hot path: two `Acquire` loads, no lock, no CAS β€” O(1). - /// Cold path: reverse-scan atomic array, k = election count (typically < 10) β€” O(k). - pub(crate) fn get( - &self, - index: u64, - ) -> Option { - let last_start = self.last_term_start.load(Ordering::Acquire); - let last_term = self.last_term.load(Ordering::Acquire); - if last_term == 0 { - return None; - } - if index >= last_start { - return Some(last_term); - } - // Cold path: reverse-scan historical segments. - let count = self.seg_count.load(Ordering::Acquire); - (0..count).rev().find_map(|i| { - let start = self.seg_starts[i].load(Ordering::Acquire); - if start <= index { - Some(self.seg_terms[i].load(Ordering::Acquire)) - } else { - None - } - }) - } - - /// Update after appending entries at the tail. Entries must be in ascending index order. - /// - /// Common case (same term): one `Acquire` load, no write β€” O(1). - /// Term change: two atomic stores to next slot, no lock β€” O(1). - pub(crate) fn on_append( - &self, - entries: &[Entry], - ) { - for entry in entries { - let lt = self.last_term.load(Ordering::Acquire); - if entry.term == lt { - // Same term: pull segment start back if needed (truncation + re-insert). - let ls = self.last_term_start.load(Ordering::Acquire); - if entry.index < ls { - self.last_term_start.store(entry.index, Ordering::Release); - } - continue; - } - if lt == 0 { - // First entries ever: initialise hot atomics. - self.last_term_start.store(entry.index, Ordering::Release); - self.last_term.store(entry.term, Ordering::Release); - continue; - } - // New term boundary: archive current segment into the next slot. - // When the array is full, skip the write β€” entry_term() falls back - // to the SkipMap cold path for any overflow segments. - let ls = self.last_term_start.load(Ordering::Acquire); - let i = self.seg_count.fetch_add(1, Ordering::AcqRel); - if i < MAX_TERM_SEGMENTS { - self.seg_starts[i].store(ls, Ordering::Release); - self.seg_terms[i].store(lt, Ordering::Release); - } - // Always update hot atomics so the current term remains O(1). - self.last_term_start.store(entry.index, Ordering::Release); - self.last_term.store(entry.term, Ordering::Release); - } - } - - /// Reset to empty. Called on log reset (snapshot install / full rewind). - pub(crate) fn clear(&self) { - self.seg_count.store(0, Ordering::Release); - self.last_term.store(0, Ordering::Release); - self.last_term_start.store(0, Ordering::Release); - } -} - -/// IO tasks for the dedicated raft-io thread. -/// -/// All blocking storage operations route through this channel so they never run -/// on tokio worker threads or the inbound event loop. -#[derive(Debug)] -pub enum IOTask { - /// Atomically truncate from `truncate_from` then persist `new_entries`. - /// Conflict-resolution path: truncate + write are a single atomic IO unit. - /// `done` is signalled after the IO thread finishes the replace so callers - /// can flush() knowing the truncation is durable. - ReplaceRange { - truncate_from: u64, - new_entries: Vec, - done: oneshot::Sender>, +use std::{ + collections::HashMap, + ops::RangeInclusive, + sync::{ + Arc, + atomic::{AtomicBool, AtomicU64, AtomicUsize, Ordering}, }, - /// Purge log entries up to `cutoff` from storage. - Purge { - cutoff: LogId, - done: oneshot::Sender<()>, - }, - /// Reset log storage (snapshot install). Routes through IO thread so reset - /// never runs on the inbound event loop. - Reset { done: oneshot::Sender> }, - /// Persist any pending entries then fsync. The IO thread sends the fsync - /// result (Ok or Err) back to the caller via the oneshot channel. - /// Replaces the former `FlushNow` + `WaitDurable` two-message dance. - Flush(oneshot::Sender>), - /// Shutdown the IO thread - Shutdown, -} + time::Duration, +}; +use tokio::sync::{mpsc, oneshot}; +use tonic::async_trait; +use tracing::{debug, error, warn}; -/// High-performance buffered Raft log with event-driven architecture -/// -/// This implementation provides in-memory first access with configurable -/// persistence strategies while ensuring thread safety and avoiding deadlocks. -/// -/// Key design principles: -/// - Lock-free reads for 99% of operations -/// - Event-driven asynchronous processing -/// - Deadlock prevention through proper error handling -/// - Memory-efficient batch operations -pub struct BufferedRaftLog +pub struct RaftLogCore where T: TypeConfig, { @@ -241,12 +32,9 @@ where pub(crate) log_store: Arc< as StorageEngine>::LogStore>, pub(crate) meta_store: Arc< as StorageEngine>::MetaStore>, - /// Safety-net timer interval (ms). Normal-path latency is determined by fsync time. - idle_flush_interval_ms: u64, - - /// Max time to wait, on shutdown, for an in-flight fsync task to finish before - /// giving up. Bounds `close()` against a stuck/slow disk β€” the task itself is - /// not cancelled, it keeps running in the background regardless. + /// Max time `close()` waits for a final flush before giving up. The + /// in-flight fsync task itself is never cancelled β€” it keeps running in + /// the background regardless; this only bounds how long the caller waits. shutdown_timeout_ms: u64, // --- In-memory state --- @@ -254,18 +42,26 @@ where // This is the single source of truth β€” every other field below is either // a boundary marker or a lookup accelerator derived from what's here. pub(crate) entries: RwLock>, + // How far the log has actually been made crash-safe (fsynced to disk). // Raft must not tell a client or a peer a write is safe ahead of this point, // regardless of what's already visible in `entries`. pub(crate) durable_index: AtomicU64, + + // How far entries have been written to the log store (page cache, not + // yet fsynced) β€” the frontier `persist_entries` scans forward from. + // Advance with `fetch_max`, never a plain store: a slower concurrent + // call must not regress a value a faster call already advanced past. + pub(crate) persisted_index: AtomicU64, + // The next index to be allocated pub(crate) next_id: AtomicU64, // --- In-memory index --- // O(1) answer to "is this index currently held in memory" β€” lets callers // (e.g. entry_term()) reject an out-of-range index without touching `entries`. - min_index: AtomicU64, // Smallest log index (0 if empty) - max_index: AtomicU64, // Largest log index (0 if empty) + min_index: AtomicU64, // Smallest log index (0 if empty) + memory_max_index: AtomicU64, // Largest log index held in memory (0 if empty) β€” may be ahead of what's persisted/durable // The term of the last entry ever purged (compacted away after a snapshot). // Raft's AppendEntries consistency check needs the term at prev_log_index @@ -273,7 +69,7 @@ where // this, a follower can't tell "purged, but we agree" apart from "conflict". // // Must be published in the same critical section as the entries removal - // and the min_index/max_index advance it corresponds to β€” a reader must + // and the min_index/memory_max_index advance it corresponds to β€” a reader must // never be able to observe the entry gone but this boundary not yet set. last_purged_index: AtomicU64, last_purged_term: AtomicU64, @@ -289,29 +85,19 @@ where // of the log, asked on essentially every AppendEntries in normal replication. term_segments: TermSegments, - // --- Flush coordination --- - /// Coalesced write notification. `append_entries` calls `notify_one()` after - /// inserting into the SkipMap. Multiple concurrent writers coalesce into a - /// single IO thread wakeup, eliminating per-write kernel cond_signal overhead. - pub(crate) write_notify: Arc, - pub(crate) command_sender: mpsc::UnboundedSender, - // --- P0: LogFlushed event notification --- // Sends InternalEvent::LogFlushed(durable) to Raft loop after each fsync. // None in tests; Some(internal_event_tx) in production. - log_flush_tx: Option>, + internal_event_tx: Option>, - // IO thread handle β€” set by start(), used by close() to join before returning. - io_thread_handle: std::sync::Mutex>>, - - fsync_coordinator: Arc, + fsync_worker: Arc as StorageEngine>::LogStore>>, // set once on first fsync failure, never cleared for this instance's lifetime - pub(super) poisoned: AtomicBool, + pub(super) poisoned: Arc, } #[async_trait] -impl RaftLog for BufferedRaftLog +impl RaftLog for RaftLogCore where T: TypeConfig, { @@ -329,11 +115,10 @@ where } fn last_entry_id(&self) -> u64 { - self.max_index.load(Ordering::Acquire) + self.memory_max_index.load(Ordering::Acquire) } fn durable_index(&self) -> u64 { - // fsync is async; only entries confirmed by batch_processor are crash-safe. self.durable_index.load(Ordering::Acquire) } @@ -346,6 +131,9 @@ where } } + // #446: this is what election-eligibility comparisons (is_target_log_more_recent) + // read. It must keep reflecting the in-memory tail, never durable_index β€” a node + // with an un-fsynced entry must still be able to reject a less-up-to-date candidate. fn last_log_id(&self) -> Option { let last_index = self.last_entry_id(); if last_index > 0 { @@ -378,7 +166,7 @@ where entry_id: u64, ) -> Option { // Bounds check: skip TermSegments entirely for out-of-range queries. - let max = self.max_index.load(Ordering::Acquire); + let max = self.memory_max_index.load(Ordering::Acquire); let min = self.min_index.load(Ordering::Acquire); if max == 0 || entry_id < min || entry_id > max { // Cold path: check purge boundary so that AppendEntries built with @@ -438,12 +226,13 @@ where &self, range: RangeInclusive, ) -> Result> { + let entries = self.entries.read(); + // OPTIMIZED: SkipMap range scan O(k + log n); pre-allocate to avoid realloc - let capacity = (range.end().saturating_sub(*range.start()) + 1) as usize; + let capacity = + (range.end().saturating_sub(*range.start()) + 1).min(entries.len() as u64) as usize; let mut result = Vec::with_capacity(capacity); - let entries = self.entries.read(); - result.extend(entries.range(range).map(|e| e.value().clone())); Ok(result) } @@ -453,7 +242,7 @@ where entries: Vec, ) -> Result<()> { // Fast-fail optimization, not a correctness gate β€” the real durability - // boundary is persist_pending_range()/fsync_coordinator. Poisoned data + // boundary is persist_pending_range()/fsync_worker. Poisoned data // reaching memory here is harmless as long as it never gets marked durable. if self.is_poisoned() { return Err(Error::Fatal("raft log storage is poisoned".into())); @@ -465,10 +254,20 @@ where } self.insert_to_memory(&entries); - // Signal IO thread to persist. Multiple concurrent notify_one() calls - // while the IO thread is busy coalesce into one wakeup β€” no per-write - // kernel cond_signal. IO thread reads from SkipMap via max_index. - self.write_notify.notify_one(); + + // A write landed. Persist the new tail and submit it. No queue + // drain: a burst's redundant `Persist`s find `persisted_index` + // already at `memory_max_index` and no-op here (#446). + let start = self.persisted_index.load(Ordering::Acquire) + 1; + let end = self.memory_max_index.load(Ordering::Acquire); + match self.persist_pending_range(start, end, "persist").await { + Ok(Some(mark)) => { + self.persisted_index.fetch_max(mark.index, Ordering::AcqRel); + self.fsync_worker.submit(mark, Vec::new()); + } + Ok(None) => {} + Err(e) => return Err(e), // poisoned β€” same exit convention as the mutations below + } Ok(()) } @@ -488,22 +287,16 @@ where new_entries: Vec, ) -> Result> { let _timer = ScopedTimer::new("filter_out_conflicts_and_append"); - // prev_log_index == 0 means the leader wants the follower to start from scratch - // (e.g. new follower joining, or follower log fully diverged). Reset and replace. - if prev_log_index == 0 && prev_log_term == 0 { - self.reset().await?; - self.append_entries(new_entries.clone()).await?; - return Ok(new_entries.last().map(|e| LogId { - term: e.term, - index: e.index, - })); - } - // Check log consistency: use entry_term() so purge-boundary entries - // (entries removed from the SkipMap but recorded in last_purged_index/term) - // are still recognised as valid prev_log positions after snapshot install. - if self.entry_term(prev_log_index) != Some(prev_log_term) { - return Ok(self.last_log_id()); + // prev_log_index==0 has no real entry to compare against, not a reset signal + let is_virtual_log_start = prev_log_index == 0 && prev_log_term == 0; + if !is_virtual_log_start { + // Check log consistency: use entry_term() so purge-boundary entries + // (entries removed from the SkipMap but recorded in last_purged_index/term) + // are still recognised as valid prev_log positions after snapshot install. + if self.entry_term(prev_log_index) != Some(prev_log_term) { + return Ok(self.last_log_id()); + } } let last_current_index = self.last_entry_id(); @@ -573,27 +366,11 @@ where if diverge_index <= last_current_index { // Real term conflict: truncate from diverge_index, replace with tail. // Await the done channel so callers can flush() knowing the truncation - // is durable β€” durable_index may exceed max_index after truncation, + // is durable β€” durable_index may exceed memory_max_index after truncation, // which would cause flush() to short-circuit before the replace lands. self.remove_range(diverge_index..=u64::MAX); self.insert_to_memory(tail); - let (done_tx, done_rx) = oneshot::channel(); - self.command_sender - .send(IOTask::ReplaceRange { - truncate_from: diverge_index, - new_entries: tail.to_vec(), - done: done_tx, - }) - .map_err(|e| { - NetworkError::SingalSendFailed(format!( - "Failed to send ReplaceRange: {e:?}" - )) - })?; - done_rx.await.map_err(|_| { - NetworkError::SingalSendFailed( - "ReplaceRange done channel closed".into(), - ) - })??; + self.replace_range_and_submit(diverge_index, tail.to_vec()).await?; } else { // No conflict β€” pipeline overlap consumed, append only the new tail. self.append_entries(tail.to_vec()).await?; @@ -617,11 +394,10 @@ where mut peer_matched_ids: Vec, ) -> Option { let _timer = ScopedTimer::new("calculate_majority_matched_index"); - // Leader's contribution: last_entry_id (in-memory). With MemFirst (Level 2), db.write() - // returns once data reaches OS page cache β€” durable_index advances immediately. - // Followers also ACK after OS page cache write (no fsync wait). Crash safety is - // OS page cache level: process crash is recoverable, power loss is not. - peer_matched_ids.push(self.last_entry_id()); + // RPO=0 (#446): leader's own contribution must be its own durable (fsynced) + // position, not the in-memory tail β€” otherwise a majority-looking commit can + // still lose data on correlated power loss. + peer_matched_ids.push(self.durable_index()); // Sort in descending order peer_matched_ids.sort_unstable_by(|a, b| b.cmp(a)); @@ -630,6 +406,7 @@ where let majority_index = peer_matched_ids[peer_matched_ids.len() / 2]; debug!( + ?self.node_id, "Majority calculation: peers={:?}, majority_index={}", peer_matched_ids, majority_index, ); @@ -651,51 +428,73 @@ where cutoff_index: LogId, ) -> Result<()> { let _timer = ScopedTimer::new("purge_logs_up_to"); - debug!(?cutoff_index, "purge_logs_up_to"); + debug!(?self.node_id, ?cutoff_index, "purge_logs_up_to"); // Remove range + publish last_purged_* atomically in the same lock (#442). self.purge_prefix(cutoff_index); // Purged entries are backed by the snapshot; treat cutoff as durable. - // fetch_max is monotonic β€” avoids racing fsync_coordinator's concurrent - // advance on the raft-io thread β€” and this fires LogFlushed consistently - // with every other durable_index advancement in this file. - self.advance_durable_and_notify(cutoff_index.index); - - // Route purge through the IO thread so it never blocks the inbound event loop. - // Also writes the purge boundary to META_CF in the RocksDB implementation. - let (done_tx, done_rx) = oneshot::channel(); - self.command_sender - .send(IOTask::Purge { - cutoff: cutoff_index, - done: done_tx, - }) - .map_err(|e| NetworkError::SingalSendFailed(format!("Failed to send Purge: {e:?}")))?; - done_rx - .await - .map_err(|_| NetworkError::SingalSendFailed("Purge channel closed".into()))?; + // Already running on the single owner (called from role_state.rs, same + // thread as remove_range) β€” safe to apply directly, no message hop needed. + if let Some(new_durable) = self.try_advance_durable_index(cutoff_index) + && let Some(ref tx) = self.internal_event_tx + { + let _ = tx.send(crate::InternalEvent::LogFlushed { + durable_index: new_durable, + }); + } + self.purge_and_advance(cutoff_index).await + } - Ok(()) + fn try_advance_durable_index( + &self, + mark: LogId, + ) -> Option { + let prev = self.durable_index.load(Ordering::Acquire); + if mark.index <= prev { + return None; + } + if self.entry_term(mark.index) != Some(mark.term) { + return None; + } + let safe = mark.index.min( + self.memory_max_index + .load(Ordering::Acquire) + .max(self.last_purged_index.load(Ordering::Acquire)), + ); + if safe <= prev { + return None; + } + self.durable_index.fetch_max(safe, Ordering::AcqRel); + Some(safe) } async fn flush(&self) -> Result<()> { - let max_index = self.max_index.load(Ordering::Acquire); - if max_index == 0 { + let target = self.memory_max_index.load(Ordering::Acquire); + if target == 0 || self.durable_index.load(Ordering::Acquire) >= target { return Ok(()); } - if self.durable_index.load(Ordering::Acquire) >= max_index { - return Ok(()); + let persisted = self.persisted_index.load(Ordering::Acquire); + if persisted < target + && let Some(m) = self.persist_pending_range(persisted + 1, target, "flush").await? + { + self.persisted_index.fetch_max(m.index, Ordering::AcqRel); } let (tx, rx) = oneshot::channel(); - self.command_sender - .send(IOTask::Flush(tx)) - .map_err(|e| NetworkError::SingalSendFailed(format!("flush send failed: {e:?}")))?; + let term = self.entry_term(target).unwrap_or(0); + self.fsync_worker.submit( + LogId { + term, + index: target, + }, + vec![tx], + ); rx.await .map_err(|_| NetworkError::SingalSendFailed("flush channel closed".into()))? } async fn reset(&self) -> Result<()> { - let _timer = ScopedTimer::new("buffered_raft_log::reset"); + let _timer = ScopedTimer::new("raft_log_core::reset"); self.reset_internal().await } @@ -711,56 +510,45 @@ where return Err(Error::Fatal("raft log storage is poisoned".into())); } self.meta_store.save_hard_state(hard_state).inspect_err(|e| { - error!("save_hard_state failed (fatal): {e:?}"); + error!(?self.node_id, "save_hard_state failed (fatal): {e:?}"); self.mark_poisoned_and_notify(format!("save_hard_state failed: {e:?}")); }) } - fn is_poisoned(&self) -> bool { - self.is_poisoned() - } - async fn close(&self) { - // Signal the IO thread to flush remaining data and exit. - let _ = self.command_sender.send(IOTask::Shutdown); - // Take the handle β€” idempotent; second call is a no-op. - let handle = self.io_thread_handle.lock().unwrap().take(); - if let Some(handle) = handle { - // Join on a spawn_blocking thread so we don't block a tokio worker. - tokio::task::spawn_blocking(move || { - let _ = handle.join(); - }) - .await - .ok(); + let timed_out = tokio::time::timeout( + Duration::from_millis(self.shutdown_timeout_ms), + self.flush(), + ) + .await + .is_err(); + if timed_out { + error!( + ?self.node_id, + "close(): flush() did not complete within {}ms β€” returning anyway; the \ + in-flight fsync keeps running and any queued flush() replies will still \ + resolve once it does", + self.shutdown_timeout_ms + ); } + let _ = self.meta_store.flush(); } } -impl BufferedRaftLog +impl RaftLogCore where T: TypeConfig, { pub fn new( node_id: u32, - persistence_config: PersistenceConfig, storage: Arc>, - ) -> (Self, mpsc::UnboundedReceiver) { + internal_event_tx: Option>, + shutdown_timeout_ms: u64, + ) -> Arc { let log_store = storage.log_store(); let meta_store = storage.meta_store(); let disk_len = log_store.last_index(); - let FlushPolicy::Batch { - idle_flush_interval_ms, - } = persistence_config.flush_policy; - debug!( - "Creating BufferedRaftLog with node_id: {}, strategy: {:?}, idle_flush_interval_ms: {:?}, disk_len: {:?}", - node_id, persistence_config.strategy, idle_flush_interval_ms, disk_len - ); - - let shutdown_timeout_ms = persistence_config.shutdown_timeout_ms; - - //TODO: if switch to UnboundedChannel? - let (command_sender, command_receiver) = mpsc::unbounded_channel(); let entries = SkipMap::new(); // Initialize term indexes @@ -774,7 +562,10 @@ where match log_store.get_entries(1..=disk_len) { Ok(all_entries) if !all_entries.is_empty() => { loaded_count = all_entries.len(); - debug!("Successfully loaded {} entries from disk", loaded_count); + debug!( + ?node_id, + "Successfully loaded {} entries from disk", loaded_count + ); for entry in &all_entries { let index = entry.index; @@ -795,10 +586,13 @@ where term_segments.on_append(&all_entries); } Ok(_empty_entries) => { - warn!("Disk reported length {} but loaded 0 entries", disk_len); + warn!( + ?node_id, + "Disk reported length {} but loaded 0 entries", disk_len + ); } Err(e) => { - error!("Failed to load entries from storage: {:?}", e); + error!(?node_id, "Failed to load entries from storage: {:?}", e); // Handle critical error if needed } } @@ -806,12 +600,12 @@ where // Initialize atomic boundaries let min_index = entries.front().map(|e| *e.key()).unwrap_or(0); - let max_index = entries.back().map(|e| *e.key()).unwrap_or(0); + let memory_max_index = entries.back().map(|e| *e.key()).unwrap_or(0); if disk_len > 0 && loaded_count == 0 { warn!( - "Inconsistent state: disk_len={} but loaded_count=0", - disk_len + ?node_id, + "Inconsistent state: disk_len={} but loaded_count=0", disk_len ); } @@ -821,456 +615,164 @@ where Ok(Some(lid)) => (lid.index, lid.term), Ok(None) => (0, 0), Err(e) => { - warn!("Failed to load purge boundary: {:?}, defaulting to 0", e); + warn!( + ?node_id, + "Failed to load purge boundary: {:?}, defaulting to 0", e + ); (0, 0) } }; - ( - Self { - node_id, - log_store, - meta_store, - idle_flush_interval_ms, - shutdown_timeout_ms, - entries: RwLock::new(entries), - min_index: AtomicU64::new(min_index), - max_index: AtomicU64::new(max_index), - last_purged_index: AtomicU64::new(last_purged_index_val), - last_purged_term: AtomicU64::new(last_purged_term_val), - durable_index: AtomicU64::new(disk_len), - next_id: AtomicU64::new(disk_len + 1), - write_notify: Arc::new(Notify::new()), - command_sender: command_sender.clone(), - term_first_index, - term_last_index, - term_segments, - log_flush_tx: None, // set in start() - io_thread_handle: std::sync::Mutex::new(None), - fsync_coordinator: Arc::new(FsyncCoordinator::new()), - poisoned: AtomicBool::new(false), - }, - command_receiver, - ) + let poisoned = Arc::new(AtomicBool::new(false)); + let fsync_worker = Arc::new(FsyncWorker::new( + node_id, + log_store.clone(), + poisoned.clone(), + internal_event_tx.clone(), + )); + + Arc::new(Self { + node_id, + log_store, + meta_store, + shutdown_timeout_ms, + entries: RwLock::new(entries), + min_index: AtomicU64::new(min_index), + memory_max_index: AtomicU64::new(memory_max_index), + last_purged_index: AtomicU64::new(last_purged_index_val), + last_purged_term: AtomicU64::new(last_purged_term_val), + durable_index: AtomicU64::new(disk_len), + persisted_index: AtomicU64::new(disk_len), + next_id: AtomicU64::new(disk_len + 1), + term_first_index, + term_last_index, + term_segments, + internal_event_tx, + fsync_worker, + poisoned, + }) } - /// Start the command processor and return an Arc-wrapped instance. - /// - /// `batch_processor` runs on a dedicated OS thread (not a tokio worker) so that - /// synchronous RocksDB calls (`db.write`, `flush_wal`) never block the async runtime. - /// The thread owns its own `new_current_thread` runtime β€” fully independent of the - /// caller's runtime lifecycle. When the caller's runtime drops (e.g. at test teardown), - /// the IO thread's timers and channels are unaffected; it exits cleanly on `Shutdown`. - pub fn start( - mut self, - receiver: mpsc::UnboundedReceiver, - log_flush_tx: Option>, - ) -> Arc { - self.log_flush_tx = log_flush_tx; - let arc_self = Arc::new(self); - let weak_self = Arc::downgrade(&arc_self); - - let idle_flush_interval_ms = arc_self.idle_flush_interval_ms; - let shutdown_timeout_ms = arc_self.shutdown_timeout_ms; - let node_id = arc_self.node_id; - let io_handle = std::thread::Builder::new() - .name(format!("raft-io-{}", node_id)) - .spawn(move || { - let rt = tokio::runtime::Builder::new_current_thread() - .enable_all() - .build() - .expect("failed to build raft-io runtime"); - rt.block_on(Self::batch_processor( - weak_self, - receiver, - idle_flush_interval_ms, - )); - // Explicit bounded shutdown instead of an implicit Drop β€” an - // implicit Drop blocks this thread indefinitely on any still-running - // spawn_blocking task (confirmed with the not_durable_gated_flush test). - // Healthy path: a real fsync is millisecond-scale, so this virtually - // always completes well inside the window β€” same behavior as before. - // Stuck-disk path: this thread's (and therefore close()'s) wait is now - // bounded by this timeout instead of hanging forever. The stuck task - // itself is not killed β€” it keeps running in the background until - // flush() actually returns, so durable_index advancement and reply - // delivery are unaffected; nobody is just waiting for it anymore. - rt.shutdown_timeout(Duration::from_millis(shutdown_timeout_ms)); - }) - .expect("failed to spawn raft-io thread"); + /// Insert entries into the in-memory index (SkipMap + term indexes + atomics). + fn insert_to_memory( + &self, + entries: &[Entry], + ) { + { + let log = self.entries.write(); + for entry in entries { + log.insert(entry.index, entry.clone()); + } + } - *arc_self.io_thread_handle.lock().unwrap() = Some(io_handle); + self.update_term_indexes(entries); + self.term_segments.on_append(entries); - arc_self - } + let max_index = entries.iter().map(|e| e.index).max().unwrap_or(0); + let current_next = self.next_id.load(Ordering::Acquire); + if max_index >= current_next { + self.next_id.store(max_index + 1, Ordering::Release); + } - /// Notify-driven IO loop. - /// - /// Waits on `write_notify.notified()` for new entries in the SkipMap. - /// Multiple `notify_one()` calls while the IO thread is busy (persisting or - /// fsyncing) coalesce into a single wakeup, reducing kernel cond_signal overhead - /// from one-per-write to one-per-burst. - /// - /// On each wakeup: - /// 1. Read entries in `(durable_index, max_index]` from SkipMap. - /// 2. persist_entries to OS page cache (no fsync). - /// 3. Drain any pending control commands from the mpsc channel. - /// 4. fsync once β€” advance durable_index, wake WaitDurable callers. - /// - /// Safety-net timer fires after `idle_flush_interval_ms` of inactivity. - async fn batch_processor( - this: std::sync::Weak, - mut receiver: mpsc::UnboundedReceiver, - idle_flush_interval_ms: u64, - ) { - // Upgrade once at entry β€” Arc stays alive until Shutdown (the real exit signal). - let Some(this) = this.upgrade() else { return }; - - let start = tokio::time::Instant::now() + Duration::from_millis(idle_flush_interval_ms); - let mut safety_timer = - tokio::time::interval_at(start, Duration::from_millis(idle_flush_interval_ms)); - safety_timer.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); - - // Highest index in OS page cache, awaiting fsync. Reset to 0 after each fsync. - let mut pending_max: u64 = 0; - - loop { - tokio::select! { - _ = this.write_notify.notified() => { - if Self::run_batch_turn(&this, &mut receiver, &mut pending_max, Vec::new(), false).await { - break; - } - } - cmd = receiver.recv() => { - let Some(cmd) = cmd else { break }; - let should_break = match cmd { - IOTask::Shutdown => Self::run_batch_turn(&this, &mut receiver, &mut pending_max, Vec::new(), true).await, - IOTask::Flush(reply) => Self::run_batch_turn(&this, &mut receiver, &mut pending_max, vec![reply], false).await, - cmd => { - if Self::handle_non_write_cmd(cmd, &this, &mut pending_max).await { - break; - } - continue; - } - }; - if should_break { break; } + if let Some(first_entry) = entries.first() { + let mut current_min = self.min_index.load(Ordering::Relaxed); + while first_entry.index < current_min || current_min == 0 { + match self.min_index.compare_exchange_weak( + current_min, + first_entry.index, + Ordering::AcqRel, + Ordering::Acquire, + ) { + Ok(_) => break, + Err(e) => current_min = e, } - _ = safety_timer.tick() => { - let start = this.durable_index.load(Ordering::Acquire) + 1; - let end = this.max_index.load(Ordering::Acquire); - let _ = Self::persist_pending_range(&this, start, end, &mut pending_max, "safety-net").await; - - if pending_max > 0 { - this.fsync_coordinator.submit(&this, pending_max, vec![]); - pending_max = 0; - } + } + } + + if let Some(last_entry) = entries.last() { + let mut current_max = self.memory_max_index.load(Ordering::Relaxed); + while last_entry.index > current_max { + match self.memory_max_index.compare_exchange_weak( + current_max, + last_entry.index, + Ordering::AcqRel, + Ordering::Acquire, + ) { + Ok(_) => break, + Err(e) => current_max = e, } } } } - /// Writes entries in `(from, to]` that haven't reached page cache yet - /// (no fsync). Advances `pending_max` on success; propagates the error - /// as-is on failure β€” whether to notify any waiting `Flush` caller is - /// left to the caller. + // Update the term index (completely lock-free) + fn update_term_indexes( + &self, + entries: &[Entry], + ) { + for entry in entries { + let term = entry.term; + + // Update first index + self.term_first_index + .get_or_insert(term, AtomicU64::new(u64::MAX)) + .value() + .fetch_min(entry.index, Ordering::AcqRel); + + // Update last index + self.term_last_index + .get_or_insert(term, AtomicU64::new(0)) + .value() + .fetch_max(entry.index, Ordering::AcqRel); + } + } + + /// Persists entries in `(from, to]` that aren't in the OS page cache yet + /// (no fsync). `from` / `to` are only scan bounds β€” the caller passes + /// `persisted_index + 1` and a `memory_max_index` snapshot. + /// + /// The SkipMap range scan returns only entries that still exist. If a + /// concurrent term-conflict truncation removed the top of `(from, to]` + /// between the caller's `memory_max_index` read and this scan, those + /// indices are simply absent and never written. + /// + /// Returns `Some((term, index))` of the last entry written β€” its index may + /// be *below* `to` in that truncation-race case β€” or `None` when the scan + /// found nothing. Callers advance `persisted_index` and submit this mark to + /// the fsync coordinator, so neither points past a real entry. async fn persist_pending_range( - this: &Arc, + &self, from: u64, to: u64, - pending_max: &mut u64, ctx: &str, - ) -> Result<()> { - if this.is_poisoned() { + ) -> Result> { + if self.is_poisoned() { return Err(Error::Fatal("raft log storage is poisoned".to_string())); } if from > to { - return Ok(()); + return Ok(None); } - let entries = this.get_entries_range(from..=to)?; - if entries.is_empty() { - return Ok(()); - } - this.log_store - .persist_entries(entries) - .await - .inspect(|_| { - *pending_max = (*pending_max).max(to); - }) - .inspect_err(|e| { - error!("{ctx} persist_entries failed: {e:?}"); - this.mark_poisoned_and_notify(format!("{ctx}: persist_entries failed: {e:?}")); - }) - } - - async fn run_batch_turn( - this: &Arc, - receiver: &mut mpsc::UnboundedReceiver, - pending_max: &mut u64, - mut replies: Vec>>, - mut seen_shutdown: bool, - ) -> bool { - let start = this.durable_index.load(Ordering::Acquire) + 1; - let end = this.max_index.load(Ordering::Acquire); - let mut persist_failed = false; - if let Err(e) = Self::persist_pending_range(this, start, end, pending_max, "batch").await { - for reply in replies.drain(..) { - let _ = reply.send(Err(Error::Fatal(format!("persist_entries failed: {e:?}")))); - } - persist_failed = true; - } - - // `seen_shutdown` is not a gate here β€” regardless of whether the - // caller already knows shutdown is happening, any commands still - // queued must be drained and replied to. Whether to drain the queue - // and whether to eventually break the loop are separate concerns - // and must not share one flag. - while let Ok(cmd) = receiver.try_recv() { - match cmd { - IOTask::Shutdown => { - seen_shutdown = true; - } - IOTask::Flush(reply) => replies.push(reply), - cmd => { - if Self::handle_non_write_cmd(cmd, this, pending_max).await { - for reply in replies { - let _ = reply - .send(Err(Error::Fatal("fatal IO error, batch aborted".into()))); - } - return true; - } - } - } - } - - if !replies.is_empty() && !persist_failed { - let start = *pending_max + 1; - let end = this.max_index.load(Ordering::Acquire); - let _ = - Self::persist_pending_range(this, start, end, pending_max, "batch catch-up").await; - } - - this.fsync_coordinator.submit(this, *pending_max, replies); - *pending_max = 0; - if seen_shutdown { - let _ = this.meta_store.flush(); - } - seen_shutdown - } - - /// Handle IOTask variants that are NOT `Flush` or `Shutdown`. - /// - /// Callers (`batch_processor`) dispatch `Flush` and `Shutdown` directly in the outer - /// `match` before this function is ever called β€” those two arms are unreachable here. - /// - /// Returns `true` if `batch_processor` must exit immediately (fatal IO error). - async fn handle_non_write_cmd( - cmd: IOTask, - this: &Arc, - pending_max: &mut u64, - ) -> bool { - match cmd { - IOTask::Flush(_) => { - unreachable!( - "Flush must be intercepted in the drain loop before handle_non_write_cmd" - ) - } - IOTask::Shutdown => { - unreachable!("Shutdown is always filtered out before reaching handle_non_write_cmd") - } - IOTask::ReplaceRange { - truncate_from, - new_entries, - done, - } => { - // Result is relied on immediately for protocol answers - // (entry_term(), AppendEntries consistency checks) β€” must not - // proceed on an already-untrusted disk. - if this.is_poisoned() { - let _ = done.send(Err(Error::Fatal("raft log storage is poisoned".into()))); - return true; - } - - let max_idx = new_entries.last().map(|e| e.index).unwrap_or(0); - let result = this.log_store.replace_range(truncate_from, new_entries).await; - if let Err(ref e) = result { - error!("IOTask::ReplaceRange failed (fatal): {e:?}"); - this.mark_poisoned_and_notify(format!("ReplaceRange failed: {e:?}")); - let _ = done.send(result); - return true; // signal batch_processor to exit β€” disk state is corrupted - } - if max_idx > 0 { - *pending_max = (*pending_max).max(max_idx); - } - let _ = done.send(result); - false - } - IOTask::Purge { cutoff, done } => { - // Same reasoning as ReplaceRange: purge changes what future - // protocol queries see. - if this.is_poisoned() { - let _ = done.send(()); - return true; - } - - if let Err(ref e) = this.log_store.purge(cutoff).await { - error!("IOTask::Purge failed (fatal): {e:?}"); - this.mark_poisoned_and_notify(format!("Purge failed: {e:?}")); - let _ = done.send(()); - return true; // signal batch_processor to exit β€” disk state is corrupted - } - let _ = done.send(()); - false - } - IOTask::Reset { done } => { - // Deliberately NO is_poisoned() check: Reset makes no new - // durability promise β€” it's a clean wipe, not a write the - // cluster will rely on. Allowing it to run even when poisoned - // lets the node reach a known-clean state before it exits. - let result = this.log_store.reset().await; - if let Err(ref e) = result { - error!("IOTask::Reset failed (fatal): {e:?}"); - this.mark_poisoned_and_notify(format!("Reset failed: {e:?}")); - let _ = done.send(result); - return true; // signal batch_processor to exit β€” disk state is corrupted - } - *pending_max = 0; // disk wiped β€” pending page-cache watermark must be zeroed - let _ = done.send(result); - false - } - } - } - - async fn reset_internal(&self) -> Result<()> { - // Fence first: bump generation before clearing state, so any fsync task - // still in flight will observe a mismatch and discard its stale result. - self.fsync_coordinator.fence_reset(); - - self.entries.write().clear(); - self.durable_index.store(0, Ordering::Release); - self.next_id.store(1, Ordering::Release); - - // Reset boundaries - self.min_index.store(0, Ordering::Release); - self.max_index.store(0, Ordering::Release); - - // Clear term indexes to ensure consistency after reset - self.term_first_index.clear(); - self.term_last_index.clear(); - self.term_segments.clear(); - - let (done_tx, done_rx) = oneshot::channel(); - self.command_sender - .send(IOTask::Reset { done: done_tx }) - .map_err(|e| crate::Error::Fatal(format!("IOTask::Reset send failed: {e:?}")))?; - done_rx - .await - .map_err(|e| crate::Error::Fatal(format!("IOTask::Reset recv failed: {e:?}")))??; - - Ok(()) - } - - /// Insert entries into the in-memory index (SkipMap + term indexes + atomics). - fn insert_to_memory( - &self, - entries: &[Entry], - ) { - { - let log = self.entries.write(); - for entry in entries { - log.insert(entry.index, entry.clone()); - } - } - - self.update_term_indexes(entries); - self.term_segments.on_append(entries); - - let max_index = entries.iter().map(|e| e.index).max().unwrap_or(0); - let current_next = self.next_id.load(Ordering::Acquire); - if max_index >= current_next { - self.next_id.store(max_index + 1, Ordering::Release); - } - - if let Some(first_entry) = entries.first() { - let mut current_min = self.min_index.load(Ordering::Relaxed); - while first_entry.index < current_min || current_min == 0 { - match self.min_index.compare_exchange_weak( - current_min, - first_entry.index, - Ordering::AcqRel, - Ordering::Acquire, - ) { - Ok(_) => break, - Err(e) => current_min = e, - } - } - } - - if let Some(last_entry) = entries.last() { - let mut current_max = self.max_index.load(Ordering::Relaxed); - while last_entry.index > current_max { - match self.max_index.compare_exchange_weak( - current_max, - last_entry.index, - Ordering::AcqRel, - Ordering::Acquire, - ) { - Ok(_) => break, - Err(e) => current_max = e, - } - } - } - } - - /// Advance `durable_index` to `new_durable` (monotonically) and send `LogFlushed`. - pub(super) fn advance_durable_and_notify( - &self, - new_durable: u64, - ) { - let prev = self.durable_index.fetch_max(new_durable, Ordering::AcqRel); - if new_durable > prev - && let Some(ref tx) = self.log_flush_tx - { - let _ = tx.send(crate::InternalEvent::LogFlushed { - durable_index: new_durable, - }); - } - } - - /// Mark the log as permanently poisoned and notify the Raft driving loop. - /// Single choke point for both failure surfaces (persist_entries write - /// failure and fsync failure) β€” do not set `poisoned` or call `notify_fatal` - /// directly from anywhere else. - pub(super) fn mark_poisoned_and_notify( - &self, - error: String, - ) { - self.poisoned.store(true, Ordering::Release); - self.notify_fatal(error); - } + let entries = self.get_entries_range(from..=to)?; + let Some(mark) = entries.last().map(|e| LogId { + term: e.term, + index: e.index, + }) else { + return Ok(None); + }; - /// Send `InternalEvent::FatalError` to the Raft driving loop, mirroring the - /// pattern SM-worker failures already use (state_machine_handler/worker.rs). - /// Idempotent in effect: even if called multiple times (e.g. both - /// persist_entries and a later fsync fail), raft.rs::run() only needs to - /// see it once to exit. - pub(super) fn notify_fatal( - &self, - error: String, - ) { - if let Some(ref tx) = self.log_flush_tx - && tx - .send(crate::InternalEvent::FatalError { - source: "RaftLog".to_string(), - error, - }) - .is_err() - { + let t0 = std::time::Instant::now(); + self.log_store.persist_entries(entries).await.inspect_err(|e| { error!( - "FatalError delivery failed (channel closed) β€” node may be poisoned \ - but still running; durability is no longer guaranteed" + ?self.node_id, + persist_path = ctx, + from, to, "persist_entries failed: {e:?}" ); - } + self.mark_poisoned_and_notify(format!("{ctx}: persist_entries failed: {e:?}")); + })?; + metrics::histogram!("core.raft.log_store.persist_entries_duration_ms") + .record(t0.elapsed().as_secs_f64() * 1_000.0); + Ok(Some(mark)) } /// Efficient range removal with targeted term index updates @@ -1282,7 +784,10 @@ where let entries = self.entries.write(); let (new_min, new_max) = self.remove_range_locked(&entries, range); self.min_index.store(new_min, Ordering::Release); - self.max_index.store(new_max, Ordering::Release); + self.memory_max_index.store(new_max, Ordering::Release); + + self.durable_index.fetch_min(new_max, Ordering::AcqRel); + self.fsync_worker.bump_generation(); // `entries` guard drops here (end of scope) β€” write lock released. } @@ -1365,7 +870,7 @@ where /// Purge entries at/below `cutoff.index`, publishing `last_purged_index`/ /// `last_purged_term` in the SAME critical section as the entries removal - /// and the min_index/max_index advance. Only place that should ever write + /// and the min_index/memory_max_index advance. Only place that should ever write /// `last_purged_*` β€” a reader must never observe the entries gone but the /// boundary not yet recorded (#442). pub fn purge_prefix( @@ -1376,86 +881,279 @@ where let (new_min, new_max) = self.remove_range_locked(&entries, 0..=cutoff.index); self.min_index.store(new_min, Ordering::Release); - self.max_index.store(new_max, Ordering::Release); + self.memory_max_index.store(new_max, Ordering::Release); // Write term before index (Release) so readers that load index first // then term (Acquire) always observe a consistent pair. self.last_purged_term.store(cutoff.term, Ordering::Release); self.last_purged_index.store(cutoff.index, Ordering::Release); - - // `entries` guard drops here β€” everything above is now visible together - // to any reader acquiring the read lock or loading these atomics after. } - // Update the term index (completely lock-free) - fn update_term_indexes( + /// Truncate the log from `truncate_from` and replace with `new_entries`, + /// then submit the new tail for fsync. fdatasync is whole-WAL, so this also + /// covers any leading persist from the same turn. + async fn replace_range_and_submit( &self, - entries: &[Entry], - ) { - for entry in entries { - let term = entry.term; + truncate_from: u64, + new_entries: Vec, + ) -> Result<()> { + // Result is relied on immediately for protocol answers (entry_term(), + // AppendEntries consistency checks) β€” must not proceed on an + // already-untrusted disk. + if self.is_poisoned() { + return Err(Error::Fatal("raft log storage is poisoned".into())); + } - // Update first index - self.term_first_index - .get_or_insert(term, AtomicU64::new(u64::MAX)) - .value() - .fetch_min(entry.index, Ordering::AcqRel); + // Capture the new tail's term before `new_entries` is moved. + let new_tail_term = new_entries.last().map(|e| e.term).unwrap_or(0); + let new_tail = match self.log_store.replace_range(truncate_from, new_entries).await { + Ok(new_tail) => new_tail, + Err(e) => { + error!(?self.node_id, "replace_range failed (fatal): {e:?}"); + self.mark_poisoned_and_notify(format!("replace_range failed: {e:?}")); + return Err(e); + } + }; - // Update last index - self.term_last_index - .get_or_insert(term, AtomicU64::new(0)) - .value() - .fetch_max(entry.index, Ordering::AcqRel); + if new_tail >= truncate_from { + self.fsync_worker.submit( + LogId { + term: new_tail_term, + index: new_tail, + }, + vec![], + ); } + self.persisted_index.store(new_tail, Ordering::Release); + + Ok(()) } + async fn purge_and_advance( + &self, + cutoff: LogId, + ) -> Result<()> { + if self.is_poisoned() { + return Err(Error::Fatal("raft log storage is poisoned".into())); + } + if let Err(e) = self.log_store.purge(cutoff).await { + error!(?self.node_id, "purge failed (fatal): {e:?}"); + self.mark_poisoned_and_notify(format!("Purge failed: {e:?}")); + return Err(e); + } + self.persisted_index.fetch_max(cutoff.index, Ordering::AcqRel); + Ok(()) + } + + /// Returns `true` if a storage-layer failure has permanently poisoned this + /// log β€” no further writes/commands will be attempted, and callers above + /// the storage layer (e.g. the Raft protocol loop) must stop dispatching + /// new work to this node. pub(super) fn is_poisoned(&self) -> bool { self.poisoned.load(Ordering::Relaxed) } - /// Returns the number of entries in the buffer. - /// - /// # Visibility - /// This method is only available in test builds. - #[cfg(any(test, feature = "__test_support"))] + /// Mark the log as permanently poisoned and notify the Raft driving loop. + /// Single choke point for both failure surfaces (persist_entries write + /// failure and fsync failure) β€” do not set `poisoned` or call `notify_fatal` + /// directly from anywhere else. + pub(super) fn mark_poisoned_and_notify( + &self, + error: String, + ) { + self.poisoned.store(true, Ordering::Release); + self.notify_fatal(error); + } + + /// Send `InternalEvent::FatalError` to the Raft driving loop, mirroring the + /// pattern SM-worker failures already use (state_machine_handler/worker.rs). + /// Idempotent in effect: even if called multiple times (e.g. both + /// persist_entries and a later fsync fail), raft.rs::run() only needs to + /// see it once to exit. + pub(super) fn notify_fatal( + &self, + error: String, + ) { + if let Some(ref tx) = self.internal_event_tx + && tx + .send(crate::InternalEvent::FatalError { + source: "RaftLog".to_string(), + error, + }) + .is_err() + { + error!( + ?self.node_id, + "FatalError delivery failed (channel closed) β€” node may be poisoned \ + but still running; durability is no longer guaranteed" + ); + } + } + + async fn reset_internal(&self) -> Result<()> { + // Fence first: bump generation before clearing state, so any fsync task + // still in flight will observe a mismatch and discard its stale result. + self.fsync_worker.fence_reset(); + + self.entries.write().clear(); + self.durable_index.store(0, Ordering::Release); + self.next_id.store(1, Ordering::Release); + + // Reset boundaries + self.min_index.store(0, Ordering::Release); + self.memory_max_index.store(0, Ordering::Release); + + // Clear term indexes to ensure consistency after reset + self.term_first_index.clear(); + self.term_last_index.clear(); + self.term_segments.clear(); + + if let Err(e) = self.log_store.reset().await { + error!(?self.node_id, "reset failed (fatal): {e:?}"); + self.mark_poisoned_and_notify(format!("Reset failed: {e:?}")); + return Err(e); + } + self.persisted_index.store(0, Ordering::Release); + + Ok(()) + } + pub fn len(&self) -> usize { self.entries.read().len() } + pub fn is_empty(&self) -> bool { + self.entries.read().is_empty() + } + /// Returns reference to next_id atomic for test verification (test-only). - /// - /// This accessor allows tests to verify ID allocation behavior. #[cfg(test)] pub fn next_id(&self) -> &std::sync::atomic::AtomicU64 { &self.next_id } - /// Returns true if the buffer contains no entries. - /// - /// This complements the `len()` method to follow Rust API conventions. - /// Clippy requires: Any struct with a public `len` method should also - /// have a public `is_empty` method. - #[cfg(any(test, feature = "__test_support"))] - pub fn is_empty(&self) -> bool { - self.entries.read().is_empty() + #[cfg(test)] + pub(super) fn set_memory_max_index_for_test( + &self, + value: u64, + ) { + self.memory_max_index.store(value, Ordering::Release); } } -impl Drop for BufferedRaftLog -where - T: TypeConfig, -{ - fn drop(&mut self) { - if let Err(e) = self.command_sender.clone().send(IOTask::Shutdown) { - debug!( - "Shutdown command send failed (receiver already closed): {:?}", - e - ); +/// Maximum number of historical term segments (one per leader election). +/// 1024 is far more than any realistic cluster lifetime. +const MAX_TERM_SEGMENTS: usize = 1024; + +/// Compact term boundary index for O(1) `entry_term()` lookups. +/// +/// Hot path (99%+ of queries): two `Acquire` loads, no lock, no CAS. +/// Cold path (election recovery): atomic array reverse-scan, no lock, no unsafe. +/// +/// When the array is full, new segments are silently dropped and `entry_term()` +/// falls back to the SkipMap (O(log n)) β€” correct but slower. +/// +/// Memory: 2 AtomicU64 + 1 AtomicUsize + 2Γ—1024 AtomicU64 = ~16KB, all inline. +pub(crate) struct TermSegments { + /// Term of the most-recently appended segment. + pub(crate) last_term: AtomicU64, + /// First log index belonging to `last_term`. + pub(crate) last_term_start: AtomicU64, + /// Number of valid historical segments stored in `seg_starts`/`seg_terms`. + seg_count: AtomicUsize, + /// First index of each historical segment, in append order. + seg_starts: [AtomicU64; MAX_TERM_SEGMENTS], + /// Term of each historical segment, parallel to `seg_starts`. + seg_terms: [AtomicU64; MAX_TERM_SEGMENTS], +} + +impl TermSegments { + pub(crate) fn new() -> Self { + Self { + last_term: AtomicU64::new(0), + last_term_start: AtomicU64::new(0), + seg_count: AtomicUsize::new(0), + seg_starts: std::array::from_fn(|_| AtomicU64::new(0)), + seg_terms: std::array::from_fn(|_| AtomicU64::new(0)), } } + + /// Return the term for `index`, or `None` if the log is empty or index is out of range. + /// + /// Hot path: two `Acquire` loads, no lock, no CAS β€” O(1). + /// Cold path: reverse-scan atomic array, k = election count (typically < 10) β€” O(k). + pub(crate) fn get( + &self, + index: u64, + ) -> Option { + let last_start = self.last_term_start.load(Ordering::Acquire); + let last_term = self.last_term.load(Ordering::Acquire); + if last_term == 0 { + return None; + } + if index >= last_start { + return Some(last_term); + } + // Cold path: reverse-scan historical segments. + let count = self.seg_count.load(Ordering::Acquire); + (0..count).rev().find_map(|i| { + let start = self.seg_starts[i].load(Ordering::Acquire); + if start <= index { + Some(self.seg_terms[i].load(Ordering::Acquire)) + } else { + None + } + }) + } + + /// Update after appending entries at the tail. Entries must be in ascending index order. + /// + /// Common case (same term): one `Acquire` load, no write β€” O(1). + /// Term change: two atomic stores to next slot, no lock β€” O(1). + pub(crate) fn on_append( + &self, + entries: &[Entry], + ) { + for entry in entries { + let lt = self.last_term.load(Ordering::Acquire); + if entry.term == lt { + // Same term: pull segment start back if needed (truncation + re-insert). + let ls = self.last_term_start.load(Ordering::Acquire); + if entry.index < ls { + self.last_term_start.store(entry.index, Ordering::Release); + } + continue; + } + if lt == 0 { + // First entries ever: initialise hot atomics. + self.last_term_start.store(entry.index, Ordering::Release); + self.last_term.store(entry.term, Ordering::Release); + continue; + } + // New term boundary: archive current segment into the next slot. + // When the array is full, skip the write β€” entry_term() falls back + // to the SkipMap cold path for any overflow segments. + let ls = self.last_term_start.load(Ordering::Acquire); + let i = self.seg_count.fetch_add(1, Ordering::AcqRel); + if i < MAX_TERM_SEGMENTS { + self.seg_starts[i].store(ls, Ordering::Release); + self.seg_terms[i].store(lt, Ordering::Release); + } + // Always update hot atomics so the current term remains O(1). + self.last_term_start.store(entry.index, Ordering::Release); + self.last_term.store(entry.term, Ordering::Release); + } + } + + /// Reset to empty. Called on log reset (snapshot install / full rewind). + pub(crate) fn clear(&self) { + self.seg_count.store(0, Ordering::Release); + self.last_term.store(0, Ordering::Release); + self.last_term_start.store(0, Ordering::Release); + } } -impl std::fmt::Debug for BufferedRaftLog +impl std::fmt::Debug for RaftLogCore where T: TypeConfig, { @@ -1463,74 +1161,90 @@ where &self, f: &mut std::fmt::Formatter<'_>, ) -> std::fmt::Result { - f.debug_struct("BufferedRaftLog").finish() + f.debug_struct("RaftLogCore").finish() } } #[cfg(test)] -#[path = "buffered_raft_log_test/basic_operations_test.rs"] +#[path = "raft_log_core_test/basic_operations_test.rs"] mod basic_operations_test; #[cfg(test)] -#[path = "buffered_raft_log_test/concurrent_fsync_test.rs"] +#[path = "raft_log_core_test/concurrent_fsync_test.rs"] mod concurrent_fsync_test; #[cfg(test)] -#[path = "buffered_raft_log_test/concurrent_operations_test.rs"] +#[path = "raft_log_core_test/concurrent_operations_test.rs"] mod concurrent_operations_test; #[cfg(test)] -#[path = "buffered_raft_log_test/drain_fsync_test.rs"] +#[path = "raft_log_core_test/drain_fsync_test.rs"] mod drain_fsync_test; #[cfg(test)] -#[path = "buffered_raft_log_test/durable_index_test.rs"] +#[path = "raft_log_core_test/durable_index_test.rs"] mod durable_index_test; #[cfg(test)] -#[path = "buffered_raft_log_test/edge_cases_test.rs"] +#[path = "raft_log_core_test/edge_cases_test.rs"] mod edge_cases_test; #[cfg(test)] -#[path = "buffered_raft_log_test/flush_strategy_test.rs"] +#[path = "raft_log_core_test/flush_strategy_test.rs"] mod flush_strategy_test; #[cfg(test)] -#[path = "buffered_raft_log_test/id_allocation_test.rs"] +#[path = "raft_log_core_test/id_allocation_test.rs"] mod id_allocation_test; #[cfg(test)] -#[path = "buffered_raft_log_test/performance_test.rs"] +#[path = "raft_log_core_test/performance_test.rs"] mod performance_test; #[cfg(test)] -#[path = "buffered_raft_log_test/pipeline_overlap_test.rs"] +#[path = "raft_log_core_test/durable_index_truncation_clamp_test.rs"] +mod durable_index_truncation_clamp_test; + +#[cfg(test)] +#[path = "raft_log_core_test/pipeline_overlap_test.rs"] mod pipeline_overlap_test; #[cfg(test)] -#[path = "buffered_raft_log_test/quorum_durability_test.rs"] +#[path = "raft_log_core_test/quorum_durability_test.rs"] mod quorum_durability_test; #[cfg(test)] -#[path = "buffered_raft_log_test/raft_properties_test.rs"] +#[path = "raft_log_core_test/raft_properties_test.rs"] mod raft_properties_test; #[cfg(test)] -#[path = "buffered_raft_log_test/remove_range_test.rs"] +#[path = "raft_log_core_test/remove_range_test.rs"] mod remove_range_test; #[cfg(test)] -#[path = "buffered_raft_log_test/shutdown_test.rs"] +#[path = "raft_log_core_test/replace_range_fsync_test.rs"] +mod replace_range_fsync_test; + +#[cfg(test)] +#[path = "raft_log_core_test/shutdown_test.rs"] mod shutdown_test; #[cfg(test)] -#[path = "buffered_raft_log_test/term_index_test.rs"] +#[path = "raft_log_core_test/term_index_test.rs"] mod term_index_test; #[cfg(test)] -#[path = "buffered_raft_log_test/term_segments_test.rs"] +#[path = "raft_log_core_test/term_segments_test.rs"] mod term_segments_test; #[cfg(test)] -#[path = "buffered_raft_log_test/worker_test.rs"] -mod worker_test; +#[path = "raft_log_core_test/truncation_fsync_fence_test.rs"] +mod truncation_fsync_fence_test; + +#[cfg(test)] +#[path = "raft_log_core_test/content_validated_watermark_test.rs"] +mod content_validated_watermark_test; + +#[cfg(test)] +#[path = "raft_log_core_test/prev_log_index_zero_idempotency_test.rs"] +mod prev_log_index_zero_idempotency_test; diff --git a/d-engine-core/src/storage/buffered_raft_log_test/basic_operations_test.rs b/d-engine-core/src/storage/raft_log_core_test/basic_operations_test.rs similarity index 75% rename from d-engine-core/src/storage/buffered_raft_log_test/basic_operations_test.rs rename to d-engine-core/src/storage/raft_log_core_test/basic_operations_test.rs index 6790f539..c445607f 100644 --- a/d-engine-core/src/storage/buffered_raft_log_test/basic_operations_test.rs +++ b/d-engine-core/src/storage/raft_log_core_test/basic_operations_test.rs @@ -1,4 +1,4 @@ -//! Basic BufferedRaftLog operations tests +//! Basic RaftLogCore operations tests //! //! Tests verify core CRUD functionality: //! - Entry insertion and retrieval @@ -9,11 +9,10 @@ use d_engine_proto::common::{Entry, LogId}; +use crate::RaftLog; use crate::test_utils::{ - BufferedRaftLogTestContext, mock_empty_entries, simulate_delete_command, - simulate_insert_command, + RaftLogCoreTestContext, mock_empty_entries, simulate_delete_command, simulate_insert_command, }; -use crate::{FlushPolicy, PersistenceStrategy, RaftLog}; /// Test get_entries_range returns correct subset /// @@ -23,13 +22,7 @@ use crate::{FlushPolicy, PersistenceStrategy, RaftLog}; /// - Expected: Returns exactly 2 entries with correct indexes #[tokio::test] async fn test_get_entries_range_returns_correct_subset() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_get_entries_range_returns_correct_subset", - ); + let ctx = RaftLogCoreTestContext::new("test_get_entries_range_returns_correct_subset"); // Arrange: Insert 4 entries simulate_insert_command(&ctx.raft_log, vec![11, 12, 13, 14], 4).await; @@ -51,13 +44,7 @@ async fn test_get_entries_range_returns_correct_subset() { /// - Expected: Returns exactly 1 entry #[tokio::test] async fn test_get_entries_range_handles_large_range() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_get_entries_range_handles_large_range", - ); + let ctx = RaftLogCoreTestContext::new("test_get_entries_range_handles_large_range"); // Arrange: Insert 299 entries for i in 1..300 { @@ -85,13 +72,8 @@ async fn test_get_entries_range_handles_large_range() { /// - Expected: Entry 3 updated from term 2 to term 3 #[tokio::test] async fn test_filter_conflicts_removes_entries_with_different_term() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_filter_conflicts_removes_entries_with_different_term", - ); + let ctx = + RaftLogCoreTestContext::new("test_filter_conflicts_removes_entries_with_different_term"); // Arrange: Setup initial log ctx.raft_log.reset().await.unwrap(); @@ -132,13 +114,7 @@ async fn test_filter_conflicts_removes_entries_with_different_term() { /// - Expected: Conflicts resolved correctly in both cases #[tokio::test] async fn test_filter_conflicts_handles_multiple_scenarios() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_filter_conflicts_handles_multiple_scenarios", - ); + let ctx = RaftLogCoreTestContext::new("test_filter_conflicts_handles_multiple_scenarios"); // Arrange: Setup initial log ctx.raft_log.reset().await.unwrap(); @@ -192,13 +168,7 @@ async fn test_filter_conflicts_handles_multiple_scenarios() { /// - Expected: last_entry returns index 301 #[tokio::test] async fn test_last_entry_returns_highest_index() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_last_entry_returns_highest_index", - ); + let ctx = RaftLogCoreTestContext::new("test_last_entry_returns_highest_index"); // Arrange: Reset and insert entries ctx.raft_log.reset().await.unwrap(); @@ -217,13 +187,7 @@ async fn test_last_entry_returns_highest_index() { /// - Expected: last_entry.index == len() #[tokio::test] async fn test_last_entry_matches_buffer_length() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_last_entry_matches_buffer_length", - ); + let ctx = RaftLogCoreTestContext::new("test_last_entry_matches_buffer_length"); // Arrange: Insert 300 entries simulate_insert_command(&ctx.raft_log, (1..=300).collect(), 1).await; @@ -246,13 +210,7 @@ async fn test_last_entry_matches_buffer_length() { /// - Expected: Entry index is sequential (1), large payload ID handled correctly #[tokio::test] async fn test_last_entry_with_large_payload_id() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_last_entry_with_large_payload_id", - ); + let ctx = RaftLogCoreTestContext::new("test_last_entry_with_large_payload_id"); // Arrange: Insert command with large payload ID (not affecting entry index) let max = u64::MAX; @@ -271,13 +229,7 @@ async fn test_last_entry_with_large_payload_id() { /// - Expected: All entries present with correct sequential indexes #[tokio::test] async fn test_insert_batch_appends_entries_in_order() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_insert_batch_appends_entries_in_order", - ); + let ctx = RaftLogCoreTestContext::new("test_insert_batch_appends_entries_in_order"); // Arrange: Reset and insert 2000 entries ctx.raft_log.reset().await.unwrap(); @@ -304,13 +256,7 @@ async fn test_insert_batch_appends_entries_in_order() { /// - Expected: Correct counts for each range #[tokio::test] async fn test_get_entries_range_multiple_bounds() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_get_entries_range_multiple_bounds", - ); + let ctx = RaftLogCoreTestContext::new("test_get_entries_range_multiple_bounds"); // Arrange: Insert 2000 entries ctx.raft_log.reset().await.unwrap(); @@ -359,13 +305,7 @@ async fn test_get_entries_range_multiple_bounds() { /// - Expected: Treated as separate events, not duplicates #[tokio::test] async fn test_insert_duplicate_commands_as_separate_events() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_insert_duplicate_commands_as_separate_events", - ); + let ctx = RaftLogCoreTestContext::new("test_insert_duplicate_commands_as_separate_events"); // Arrange: Reset and insert first batch ctx.raft_log.reset().await.unwrap(); @@ -394,13 +334,7 @@ async fn test_insert_duplicate_commands_as_separate_events() { /// - Expected: Log length reflects all operations #[tokio::test] async fn test_purge_after_insert_maintains_consistency() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_purge_after_insert_maintains_consistency", - ); + let ctx = RaftLogCoreTestContext::new("test_purge_after_insert_maintains_consistency"); // Arrange: Insert initial entries ctx.raft_log.reset().await.unwrap(); @@ -427,13 +361,7 @@ async fn test_purge_after_insert_maintains_consistency() { /// - Expected: Entries 1-3 removed, 4-9 remain #[tokio::test] async fn test_purge_logs_removes_entries_up_to_index() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_purge_logs_removes_entries_up_to_index", - ); + let ctx = RaftLogCoreTestContext::new("test_purge_logs_removes_entries_up_to_index"); // Arrange: Insert 9 entries ctx.raft_log.reset().await.unwrap(); @@ -475,13 +403,7 @@ async fn test_purge_logs_removes_entries_up_to_index() { /// - Expected: All complete successfully without data corruption #[tokio::test] async fn test_concurrent_purge_operations_are_safe() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_concurrent_purge_operations_are_safe", - ); + let ctx = RaftLogCoreTestContext::new("test_concurrent_purge_operations_are_safe"); // Arrange: Insert 10 entries ctx.raft_log.reset().await.unwrap(); @@ -514,13 +436,7 @@ async fn test_concurrent_purge_operations_are_safe() { /// - Expected: first_entry_id reflects new starting index #[tokio::test] async fn test_first_entry_id_after_purge_updates() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_first_entry_id_after_purge_updates", - ); + let ctx = RaftLogCoreTestContext::new("test_first_entry_id_after_purge_updates"); // Arrange: Insert 10 entries ctx.raft_log.reset().await.unwrap(); @@ -546,13 +462,7 @@ async fn test_first_entry_id_after_purge_updates() { /// - Expected: Entry persisted with correct index #[tokio::test] async fn test_single_entry_insert_succeeds() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_single_entry_insert_succeeds", - ); + let ctx = RaftLogCoreTestContext::new("test_single_entry_insert_succeeds"); // Act: Insert single entry simulate_insert_command(&ctx.raft_log, vec![1], 1).await; @@ -568,13 +478,7 @@ async fn test_single_entry_insert_succeeds() { /// - Expected: is_empty() returns true #[tokio::test] async fn test_is_empty_returns_true_for_new_log() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_is_empty_returns_true_for_new_log", - ); + let ctx = RaftLogCoreTestContext::new("test_is_empty_returns_true_for_new_log"); // Assert: New log is empty assert!(ctx.raft_log.is_empty(), "New log should be empty"); @@ -587,13 +491,7 @@ async fn test_is_empty_returns_true_for_new_log() { /// - Expected: is_empty() returns false #[tokio::test] async fn test_is_empty_returns_false_after_append() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_is_empty_returns_false_after_append", - ); + let ctx = RaftLogCoreTestContext::new("test_is_empty_returns_false_after_append"); // Arrange: Insert entry simulate_insert_command(&ctx.raft_log, vec![1], 1).await; @@ -612,13 +510,7 @@ async fn test_is_empty_returns_false_after_append() { /// - Expected: Returns (0, 0) #[tokio::test] async fn test_last_log_id_for_empty_log() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_last_log_id_for_empty_log", - ); + let ctx = RaftLogCoreTestContext::new("test_last_log_id_for_empty_log"); // Act & Assert: Empty log returns default assert_eq!( @@ -635,13 +527,7 @@ async fn test_last_log_id_for_empty_log() { /// - Expected: last_log_id returns (1, 11) #[tokio::test] async fn test_last_log_id_after_appends() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_last_log_id_after_appends", - ); + let ctx = RaftLogCoreTestContext::new("test_last_log_id_after_appends"); // Arrange: Insert entry with term 11 simulate_insert_command(&ctx.raft_log, vec![1], 11).await; @@ -654,25 +540,20 @@ async fn test_last_log_id_after_appends() { ); } -/// Test drop shuts down workers gracefully +/// Test dropping the log with pending writes does not panic /// /// # Scenario /// - Insert entry and explicitly drop context -/// - Expected: Workers shut down without panic +/// - Expected: No panic (RaftLogCore has no custom Drop β€” this just guards +/// against a panic during normal field teardown) #[tokio::test] -async fn test_drop_shuts_down_workers_gracefully() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_drop_shuts_down_workers_gracefully", - ); +async fn test_drop_with_pending_writes_does_not_panic() { + let ctx = RaftLogCoreTestContext::new("test_drop_with_pending_writes_does_not_panic"); // Arrange: Insert entry simulate_insert_command(&ctx.raft_log, vec![1], 1).await; - // Act: Drop context (triggers shutdown) + // Act: Drop context drop(ctx); // Assert: No panic during drop (implicit) @@ -685,20 +566,8 @@ async fn test_drop_shuts_down_workers_gracefully() { /// - Expected: All entries before index 5 are identical #[tokio::test] async fn test_same_index_and_term_implies_identical_prefix() { - let ctx1 = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_log_matching_1", - ); - let ctx2 = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_log_matching_2", - ); + let ctx1 = RaftLogCoreTestContext::new("test_log_matching_1"); + let ctx2 = RaftLogCoreTestContext::new("test_log_matching_2"); // Arrange: Create identical prefix in both logs ctx1.raft_log.reset().await.unwrap(); @@ -725,13 +594,7 @@ async fn test_same_index_and_term_implies_identical_prefix() { /// - Expected: Committed entry preserved #[tokio::test] async fn test_committed_entry_present_in_future_leaders() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_committed_entry_present", - ); + let ctx = RaftLogCoreTestContext::new("test_committed_entry_present"); // Arrange: Insert and commit entry ctx.raft_log.reset().await.unwrap(); @@ -757,13 +620,7 @@ async fn test_committed_entry_present_in_future_leaders() { /// - Expected: last_entry always reflects most recent append #[tokio::test] async fn test_append_updates_last_entry() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_append_updates_last_entry", - ); + let ctx = RaftLogCoreTestContext::new("test_append_updates_last_entry"); // Arrange: Reset log ctx.raft_log.reset().await.unwrap(); @@ -783,13 +640,7 @@ async fn test_append_updates_last_entry() { /// - Expected: No error, log unchanged #[tokio::test] async fn test_insert_batch_with_empty_list() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_insert_batch_with_empty_list", - ); + let ctx = RaftLogCoreTestContext::new("test_insert_batch_with_empty_list"); // Act: Insert empty batch let result = ctx.raft_log.insert_batch(vec![]).await; @@ -806,13 +657,7 @@ async fn test_insert_batch_with_empty_list() { /// - Expected: length, last_entry_id, last_log_id all correct #[tokio::test] async fn test_insert_batch_updates_metadata() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_insert_batch_updates_metadata", - ); + let ctx = RaftLogCoreTestContext::new("test_insert_batch_updates_metadata"); // Arrange: Reset log ctx.raft_log.reset().await.unwrap(); diff --git a/d-engine-core/src/storage/buffered_raft_log_test/concurrent_fsync_test.rs b/d-engine-core/src/storage/raft_log_core_test/concurrent_fsync_test.rs similarity index 53% rename from d-engine-core/src/storage/buffered_raft_log_test/concurrent_fsync_test.rs rename to d-engine-core/src/storage/raft_log_core_test/concurrent_fsync_test.rs index 178a9328..a30a16f5 100644 --- a/d-engine-core/src/storage/buffered_raft_log_test/concurrent_fsync_test.rs +++ b/d-engine-core/src/storage/raft_log_core_test/concurrent_fsync_test.rs @@ -2,15 +2,14 @@ //! //! Covers three correctness dimensions: //! - **Protocol**: Raft invariants must hold regardless of fsync timing -//! - **Logic**: `FsyncCoordinator` (submit / run_until_caught_up) / shutdown_fsync / -//! advance_durable_and_notify contract +//! - **Logic**: `FsyncWorker` (submit / run_until_caught_up) / try_advance_durable_index contract //! - **Concurrency**: Reset races, out-of-order completion, crash recovery -use crate::{ - BufferedRaftLog, FlushPolicy, InternalEvent, MockStorageEngine, MockTypeConfig, - PersistenceConfig, PersistenceStrategy, RaftLog, -}; +use crate::test_utils::drain_and_apply_fsync_completions; +use crate::test_utils::wait_for_durable_index; +use crate::{MockStorageEngine, MockTypeConfig, RaftLog, RaftLogCore}; use d_engine_proto::common::Entry; +use d_engine_proto::common::LogId; use std::sync::Arc; use std::time::Duration; use tokio::sync::mpsc; @@ -31,26 +30,18 @@ async fn test_durable_index_not_advanced_before_fsync_completes() { let (storage, flush_gate) = MockStorageEngine::not_durable_gated_flush( "durable_index_not_advanced_before_fsync_completes".into(), ); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - Arc::new(storage), - ); - let raft_log = raft_log.start(receiver, None); - std::thread::sleep(Duration::from_millis(10)); // ensure IO thread is ready + // A real log_flush_tx is required now: durable_index only advances when + // something drains InternalEvent::FsyncCompleted and calls + // try_advance_durable_index β€” see drain_and_apply_fsync_completions. + let (log_flush_tx, mut log_flush_rx) = mpsc::unbounded_channel(); + let raft_log = + RaftLogCore::::new(1, Arc::new(storage), Some(log_flush_tx), 5000); let pre_write_durable_index = raft_log.durable_index(); - // append_entries β†’ write_notify.notify_one() β†’ IO thread picks it up β†’ - // FsyncCoordinator::submit() spawns run_until_caught_up() on the blocking - // pool β†’ log_store.flush() blocks on flush_gate. + // append_entries persists inline, then hands the mark to FsyncWorker::submit(), + // which spawns run_until_caught_up() on its own thread β€” log_store.flush() + // blocks on flush_gate there. raft_log .append_entries(vec![Entry { index: 1, @@ -60,7 +51,7 @@ async fn test_durable_index_not_advanced_before_fsync_completes() { .await .unwrap(); - // Give the IO thread + blocking task time to reach the gated flush() call. + // Give the fsync thread time to reach the gated flush() call. tokio::time::sleep(Duration::from_millis(50)).await; // While the gate is closed, durable_index must still equal @@ -71,12 +62,10 @@ async fn test_durable_index_not_advanced_before_fsync_completes() { "durable_index must not advance before fsync completes" ); - // Release the gate β€” flush() returns, advance_durable_and_notify(1) fires. + // Release the gate β€” flush() returns, notify_fsync_completed(1, 1) fires. flush_gate.send(()).unwrap(); - // Pick a polling/backoff strategy instead of a fixed sleep, - // to avoid flakiness under CI load. - tokio::time::sleep(Duration::from_millis(50)).await; + wait_for_durable_index(&raft_log, &mut log_flush_rx, 1, Duration::from_secs(5)).await; assert_eq!( raft_log.durable_index(), @@ -85,52 +74,29 @@ async fn test_durable_index_not_advanced_before_fsync_completes() { ); } -/// `calculate_majority_matched_index` uses the in-memory SkipMap (`last_entry_id`), -/// not `durable_index` β€” even when all fsyncs are stalled indefinitely. -/// -/// Stall every flush() call via a MockLogStore barrier, append entries, then verify -/// that majority-matched calculation returns the correct in-memory index. +/// `calculate_majority_matched_index` uses `durable_index` (fsync-confirmed), not the +/// in-memory `last_entry_id` β€” even when a follower already reports the index, the +/// leader's own contribution must not count toward quorum until it has itself fsynced. /// -/// Regression guard: if majority calculation ever changes to depend on `durable_index`, -/// this test will catch it before it reaches production. +/// Stall every flush() call via a MockLogStore barrier, append entries, then verify that +/// majority-matched calculation does NOT advance while fsync is stalled, and does advance +/// once fsync completes. /// -/// Expected: -/// - Append entries so `last_entry_id()` reaches N (e.g. 5) while fsync is -/// permanently stalled β€” `durable_index()` stays at its pre-write value -/// (0) throughout. -/// - Feed `calculate_majority_matched_index` a `match_index` map where enough -/// followers already report N to form a majority. -/// - Assert the returned majority-matched index equals N (matching -/// `last_entry_id()`) β€” NOT 0 (what it would return if it mistakenly used -/// `durable_index()` instead). +/// Regression guard: RPO=0 (#446) requires the leader's own copy to be durable before it +/// counts toward commit β€” if this ever reverts to using `last_entry_id`, this test will +/// catch it before it reaches production. #[tokio::test] -async fn test_majority_matched_index_uses_memory_not_durable_index() { - // Gate closed: the first flush() call will block until we send () on `flush_gate`. +async fn test_majority_matched_index_uses_durable_not_memory() { let (storage, flush_gate) = MockStorageEngine::not_durable_gated_flush( - "majority_matched_index_uses_memory_not_durable_index".into(), - ); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - // Safety-net disabled: only write_notify should trigger fsync here. - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - Arc::new(storage), + "majority_matched_index_uses_durable_not_memory".into(), ); - let raft_log = raft_log.start(receiver, None); - std::thread::sleep(Duration::from_millis(10)); // ensure IO thread is ready + let (log_flush_tx, mut log_flush_rx) = mpsc::unbounded_channel(); + let raft_log = + RaftLogCore::::new(1, Arc::new(storage), Some(log_flush_tx), 5000); let pre_write_durable_index = raft_log.durable_index(); let pre_last_entry_id = raft_log.last_entry_id(); - // append_entries β†’ write_notify.notify_one() β†’ IO thread picks it up β†’ - // FsyncCoordinator::submit() spawns run_until_caught_up() on the blocking - // pool β†’ log_store.flush() blocks on flush_gate. let entries = vec![ Entry { index: 1, @@ -146,11 +112,8 @@ async fn test_majority_matched_index_uses_memory_not_durable_index() { let size = entries.len() as u64; raft_log.append_entries(entries).await.unwrap(); - // Give the IO thread + blocking task time to reach the gated flush() call. tokio::time::sleep(Duration::from_millis(50)).await; - // While the gate is closed, durable_index must still equal - // its pre-write value β€” fsync hasn't physically completed yet. assert_eq!( raft_log.durable_index(), pre_write_durable_index, @@ -160,69 +123,61 @@ async fn test_majority_matched_index_uses_memory_not_durable_index() { assert_eq!( raft_log.last_entry_id(), pre_last_entry_id + size, - "durable_index must not advance before fsync completes" + "in-memory tail should still advance even while fsync is stalled" ); // One follower already matched index 2; the other is still behind at 0 β€” asymmetric - // on purpose. With only ONE follower at 2, the leader's own contribution decides - // whether the majority (2 out of 3 voters) reaches 2. If this ever regresses to use - // `durable_index()` (0, since fsync is still gated) instead of `last_entry_id()` (2), - // the median drops to 0 and the call returns `None` instead of `Some(2)`. + // on purpose. If the leader's own un-fsynced entry counted (the old MemFirst + // behavior), 2 out of 3 voters would reach index 2 β€” but RPO=0 requires the + // leader's own copy to be durable first, so this must return None while fsync + // is still gated. let result = raft_log.calculate_majority_matched_index(1, 1, vec![2, 0]); assert_eq!( - result, - Some(2), - "majority index must use last_entry_id (2), not durable_index (0)" + result, None, + "RPO=0: the leader's own un-fsynced entry must not count toward quorum, even \ + when a follower already reports it" ); - // Release the gate β€” flush() returns, advance_durable_and_notify(1) fires. flush_gate.send(()).unwrap(); - - // Pick a polling/backoff strategy instead of a fixed sleep, - // to avoid flakiness under CI load. - tokio::time::sleep(Duration::from_millis(50)).await; + wait_for_durable_index( + &raft_log, + &mut log_flush_rx, + pre_write_durable_index + size, + Duration::from_secs(5), + ) + .await; assert_eq!( raft_log.durable_index(), pre_write_durable_index + size, "durable_index must reach the expected index after fsync completes" ); + + let result_after_fsync = raft_log.calculate_majority_matched_index(1, 1, vec![2, 0]); + assert_eq!( + result_after_fsync, + Some(2), + "once the leader's own entry is durable, majority index must advance to 2" + ); } /// `entry_term()` returns the correct term during high-concurrency writes /// with artificially delayed fsyncs. /// /// Term correctness is a memory-only invariant (TermSegments / SkipMap). Fsync -/// timing must not affect it. Run 10 concurrent writers with a 5ms fsync delay, +/// timing must not affect it. Run 10 concurrent writers with a delayed fsync, /// then verify every entry's term matches what was written. /// -/// Expected: -/// - 10 concurrent writers each append entries with a known term (pick terms -/// that exercise a term boundary too, e.g. entries 1-5 at term 1, entries -/// 6-10 at term 2 β€” not just one flat term for all of them). -/// - `entry_term(index)` returns the correct term for every index, checked -/// BOTH while fsync is still delayed (in flight) and after it completes β€” -/// the answer must not depend on fsync having finished. +/// Doubles as a concurrency check for the inline architecture: `append_entries` +/// now executes `persist_pending_range` + `persisted_index.fetch_max` directly +/// on the caller's own task, so this test exercises genuinely concurrent +/// callers racing that path, not just concurrent readers. #[tokio::test] async fn test_entry_term_correct_during_concurrent_fsync_delay() { - // Gate closed: the first flush() call blocks until we send () on `flush_gate`. let (storage, flush_gate) = MockStorageEngine::not_durable_gated_flush( "entry_term_correct_during_concurrent_fsync_delay".into(), ); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - Arc::new(storage), - ); - let raft_log = raft_log.start(receiver, None); - std::thread::sleep(Duration::from_millis(10)); // ensure IO thread is ready + let raft_log = RaftLogCore::::new(1, Arc::new(storage), None, 5000); // 10 concurrent writers, each appending one entry. Term boundary at 5/6 // exercises TermSegments::on_append's "new term" branch, not just the @@ -275,69 +230,60 @@ async fn test_entry_term_correct_during_concurrent_fsync_delay() { // ── Logic correctness ───────────────────────────────────────────────────────── -/// `advance_durable_and_notify` is monotonic: a late-arriving lower index is a no-op. +/// `try_advance_durable_index` is monotonic: a late-arriving lower index is a no-op. /// -/// Directly call `advance_durable_and_notify(150)`, then `advance_durable_and_notify(100)`. +/// Directly call `try_advance_durable_index(150, 1)`, then `try_advance_durable_index(100, 1)`. /// Assert: /// - final `durable_index() == 150` (not 100) -/// - `LogFlushed` event fired exactly once (for 150), not twice +/// - the 150 call returns `Some(150)` (it fired), the 100 call returns `None` (no-op) /// /// Verifies the `fetch_max` invariant that makes out-of-order concurrent fsyncs safe. -/// -/// Expected: -/// - After `advance_durable_and_notify(150)`: `durable_index() == 150`. -/// - After the subsequent `advance_durable_and_notify(100)`: `durable_index()` -/// is STILL `150` (unchanged β€” 100 < 150 must be a no-op, not a regression). -/// - The flush-completion notification fires exactly once, carrying 150 β€” the -/// discarded 100 call must not fire a second notification. #[tokio::test] async fn test_durable_index_monotonic_when_fsyncs_complete_out_of_order() { - // Storage engine choice doesn't matter here β€” advance_durable_and_notify is + // Storage engine choice doesn't matter here β€” try_advance_durable_index is // called directly, bypassing the real fsync pipeline entirely. let storage = Arc::new(MockStorageEngine::with_id( "durable_index_monotonic_when_fsyncs_complete_out_of_order".into(), )); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - storage, - ); - let (log_flush_tx, mut log_flush_rx) = mpsc::unbounded_channel::(); - let raft_log = raft_log.start(receiver, Some(log_flush_tx)); - std::thread::sleep(Duration::from_millis(10)); // ensure IO thread is ready + let raft_log = RaftLogCore::::new(1, storage, None, 5000); + + // try_advance_durable_index() content-validates against entry_term(index) β€” + // needs real entries in memory, not just a raw max_index poke. + let entries: Vec = (1..=150) + .map(|index| Entry { + index, + term: 1, + payload: None, + }) + .collect(); + raft_log.append_entries(entries).await.unwrap(); // Simulates a fsync task completing with index 150, then a second, older // fsync task (dispatched earlier, finishing later) completing with 100. - raft_log.advance_durable_and_notify(150); - raft_log.advance_durable_and_notify(100); + let result_150 = raft_log.try_advance_durable_index(LogId { + term: 1, + index: 150, + }); + let result_100 = raft_log.try_advance_durable_index(LogId { + term: 1, + index: 100, + }); + assert_eq!( + result_150, + Some(150), + "the 150 call must fire β€” it's the first advance" + ); + assert_eq!( + result_100, None, + "the later, lower 100 call must be a no-op (None), not a regression" + ); assert_eq!( raft_log.durable_index(), 150, "durable_index must reflect the highest index seen (150), not the \ later-arriving lower one (100)" ); - - // Exactly one LogFlushed event must have fired, carrying 150 β€” the - // no-op 100 call must not have sent a second event. - let event = log_flush_rx.try_recv().expect("LogFlushed must fire for the 150 call"); - match event { - InternalEvent::LogFlushed { durable_index } => { - assert_eq!(durable_index, 150, "LogFlushed must carry 150, not 100"); - } - other => panic!("expected InternalEvent::LogFlushed, got {other:?}"), - } - assert!( - log_flush_rx.try_recv().is_err(), - "no second LogFlushed event should have fired for the no-op 100 call" - ); } /// A `flush()` caller receives `Ok(())` only after its batch is physically on disk. @@ -346,35 +292,14 @@ async fn test_durable_index_monotonic_when_fsyncs_complete_out_of_order() { /// is still pending while the gate is closed, open the gate, verify the future resolves /// to `Ok(())`. /// -/// Core contract of `FsyncCoordinator::run_until_caught_up`: the reply is sent -/// inside the `spawn_blocking` task, after `log_store.flush()` returns β€” never -/// before. -/// -/// Expected: -/// - While the gate is closed: polling the `flush()` future (e.g. with a -/// short `tokio::time::timeout`) shows it still pending β€” it must NOT -/// resolve before the gate opens. -/// - After releasing the gate: the SAME future resolves to `Ok(())`. +/// Core contract of `FsyncWorker::run_until_caught_up`: the reply is sent after +/// `log_store.flush()` returns β€” never before. #[tokio::test] async fn test_flush_caller_blocked_until_fsync_completes() { - // ── Setup (given) ────────────────────────────────────────────────────── let (storage, flush_gate) = MockStorageEngine::not_durable_gated_flush( "flush_caller_blocked_until_fsync_completes".into(), ); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - Arc::new(storage), - ); - let raft_log = raft_log.start(receiver, None); - std::thread::sleep(Duration::from_millis(10)); // ensure IO thread is ready + let raft_log = RaftLogCore::::new(1, Arc::new(storage), None, 5000); raft_log .append_entries(vec![Entry { @@ -404,56 +329,35 @@ async fn test_flush_caller_blocked_until_fsync_completes() { } /// `flush()` callers whose requests arrive WHILE a fsync task is already running -/// on the blocking pool are coalesced into that same in-flight task β€” not into a -/// second, competing `spawn_blocking` call. +/// are coalesced into that same in-flight task β€” not into a second, competing +/// physical fsync. /// -/// This is `FsyncCoordinator`'s actual new capability over the old inline-fsync -/// design (which could only coalesce callers that happened to land in the same -/// drain window, before the IO thread blocked on `flush()`). Now: `submit()` -/// records new work into `pending_max`/`pending_replies` and returns immediately -/// if `inflight` is already `true`; `run_until_caught_up` picks that accumulated -/// work up on its next loop iteration, before clearing `inflight`. +/// `FsyncWorker::submit()` records new work into `pending_max`/`pending_replies` +/// and returns immediately if `inflight` is already `true`; `run_until_caught_up` +/// picks that accumulated work up on its next loop iteration, before clearing +/// `inflight`. This is the mechanism this whole rearchitecture depends on for +/// keeping physical fsync counts low β€” see Β§9 in the design discussion. /// /// Configure a MockLogStore with a flush call counter and a release-gate on the -/// first `flush()` call. Append one entry β€” `write_notify` wakes the IO loop, -/// which submits a fsync round on its own (round 1), winning the CAS and -/// blocking on the gate. While round 1 is gated, call `raft_log.flush()` three -/// times (A, B, C) β€” none of them can win the CAS (round 1 is still in -/// flight), so all three just extend `pending_max`/`pending_replies` and wait. -/// Release the gate and assert: round 1 finishes and immediately picks up A/B/C -/// as a single round 2 (not one `spawn_blocking` call per caller), so +/// first `flush()` call. Append one entry β€” the inline persist path submits a +/// fsync round on its own (round 1), winning the CAS and blocking on the gate. +/// While round 1 is gated, call `raft_log.flush()` three times (A, B, C) β€” none +/// of them can win the CAS (round 1 is still in flight), so all three just +/// extend `pending_max`/`pending_replies` and wait. Release the gate and assert: +/// round 1 finishes and immediately picks up A/B/C as a single round 2, so /// `flush_call_count` is 2, not 4 β€” and all three futures resolve to `Ok(())`. -/// -/// Expected: -/// - `flush_call_count == 2`: round 1 (the append's own automatic fsync, -/// already in flight when A/B/C arrive) plus round 2 (A, B, and C served -/// together) β€” NOT 4 (one per append + one per explicit caller). -/// - All three `flush()` futures resolve to `Ok(())`. #[tokio::test] async fn test_flush_callers_arriving_during_inflight_fsync_are_coalesced() { let (storage, flush_gate, flush_call_count) = MockStorageEngine::not_durable_gated_flush_counted( "flush_callers_arriving_during_inflight_fsync_are_coalesced".into(), ); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - Arc::new(storage), - ); - let raft_log = raft_log.start(receiver, None); - std::thread::sleep(Duration::from_millis(10)); // ensure IO thread is ready + let raft_log = RaftLogCore::::new(1, Arc::new(storage), None, 5000); - // Append one entry: write_notify wakes the IO loop, which submits its own - // fsync round (round 1) and wins the CAS β€” flush() itself early-returns - // Ok(()) with nothing to do while max_index is still 0, so a real write - // is needed to get a round in flight for A/B/C to arrive during. + // Append one entry: the inline persist path submits its own fsync round + // (round 1) and wins the CAS β€” flush() itself early-returns Ok(()) with + // nothing to do while memory_max_index is still 0, so a real write is + // needed to get a round in flight for A/B/C to arrive during. raft_log .append_entries(vec![Entry { index: 1, @@ -463,9 +367,9 @@ async fn test_flush_callers_arriving_during_inflight_fsync_are_coalesced() { .await .unwrap(); - // Give the IO loop time to submit round 1 and block on the gate before - // A/B/C are dispatched β€” otherwise they could race the automatic round - // for the CAS instead of deterministically losing it. + // Give the fsync thread time to submit round 1 and block on the gate + // before A/B/C are dispatched β€” otherwise they could race the automatic + // round for the CAS instead of deterministically losing it. tokio::time::sleep(Duration::from_millis(50)).await; // A, B, C: all arrive while round 1 is still gated β€” none can win the @@ -505,34 +409,15 @@ async fn test_flush_callers_arriving_during_inflight_fsync_are_coalesced() { /// /// `close()`'s wait is bounded by `shutdown_timeout_ms` (a short value here so /// the test doesn't burn real wall-clock seconds); it does not cancel the -/// in-flight `FsyncCoordinator` task, which keeps running on the blocking pool -/// and eventually serves any callers queued behind it, independent of whether +/// in-flight `FsyncWorker` round, which keeps running on its own thread and +/// eventually serves any callers queued behind it, independent of whether /// `close()` has already returned. -/// -/// Expected: -/// - `close()` returns within roughly `shutdown_timeout_ms` even though the -/// fsync round is still gated β€” not hanging indefinitely. -/// - The queued caller's `flush()` future resolves to `Ok(())` once the -/// gate opens β€” even though `close()` already returned before that. #[tokio::test] async fn test_shutdown_with_pending_flush_caller_still_receives_ok_reply() { let (storage, flush_gate) = MockStorageEngine::not_durable_gated_flush( "shutdown_with_pending_flush_caller_still_receives_ok_reply".into(), ); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 100, - }, - Arc::new(storage), - ); - let raft_log = raft_log.start(receiver, None); - std::thread::sleep(Duration::from_millis(10)); // ensure IO thread is ready + let raft_log = RaftLogCore::::new(1, Arc::new(storage), None, 100); // Append triggers the automatic round (round 1), which wins the CAS and // blocks on the gate. @@ -547,8 +432,7 @@ async fn test_shutdown_with_pending_flush_caller_still_receives_ok_reply() { tokio::time::sleep(Duration::from_millis(50)).await; // X arrives while round 1 is gated β€” loses the CAS, gets queued into - // FsyncCoordinator's own pending_replies (not into batch_processor's - // shutdown path). + // FsyncWorker's own pending_replies. let raft_log_x = raft_log.clone(); let x = tokio::spawn(async move { raft_log_x.flush().await }); tokio::time::sleep(Duration::from_millis(50)).await; @@ -584,29 +468,12 @@ async fn test_shutdown_with_pending_flush_caller_still_receives_ok_reply() { /// Same as above, but the queued fsync round fails once unblocked β€” the /// caller must receive the real `Err`, not a silent channel-closed error. -/// -/// Expected: -/// - The queued caller's `flush()` future resolves to `Err(..)` carrying -/// the fsync failure after the gate opens. #[tokio::test] async fn test_shutdown_with_pending_flush_caller_still_receives_err_reply() { let (storage, flush_gate) = MockStorageEngine::not_durable_gated_flush_failing( "shutdown_with_pending_flush_caller_still_receives_err_reply".into(), ); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 100, - }, - Arc::new(storage), - ); - let raft_log = raft_log.start(receiver, None); - std::thread::sleep(Duration::from_millis(10)); // ensure IO thread is ready + let raft_log = RaftLogCore::::new(1, Arc::new(storage), None, 100); raft_log .append_entries(vec![Entry { @@ -626,8 +493,6 @@ async fn test_shutdown_with_pending_flush_caller_still_receives_err_reply() { "X's flush() must still be pending before shutdown" ); - // The gate stays closed here: close() must give up after - // shutdown_timeout_ms rather than hanging forever. tokio::time::timeout(Duration::from_secs(1), raft_log.close()) .await .expect("close() must return within shutdown_timeout_ms, not hang"); @@ -654,40 +519,19 @@ async fn test_shutdown_with_pending_flush_caller_still_receives_err_reply() { /// caller waiting on that stale round. /// /// Scenario: -/// 1. Write entry at index 1..100, gate the physical `flush()` call so the -/// round stays in flight (mirrors `not_durable_gated_flush` pattern used -/// throughout this file). -/// 2. While gated, call `reset()` β€” bumps `FsyncCoordinator`'s fence -/// generation, then clears `durable_index`/`max_index`/entries to 0. +/// 1. Write an entry, gate the physical `flush()` call so the round stays +/// in flight. +/// 2. While gated, call `reset()` β€” bumps `FsyncWorker`'s fence generation, +/// then clears `durable_index`/`memory_max_index`/entries to 0. /// 3. Release the gate β€” the stale round's `flush()` returns, but its /// generation no longer matches; it must discard its result instead of -/// calling `advance_durable_and_notify(100)`. -/// -/// Expected: -/// - `durable_index()` stays `0` after the gate releases β€” the stale round -/// must not resurrect the pre-reset value, not even transiently. -/// - If the stale round was carrying a `flush()` caller's reply, that -/// caller's future resolves to `Err(..)` β€” not a resurrected `Ok(())`, -/// and not a silently dropped channel. +/// resolving durable_index or the queued reply. #[tokio::test] async fn test_reset_during_inflight_fsync_does_not_resurrect_stale_durable_index() { let (storage, flush_gate) = MockStorageEngine::not_durable_gated_flush( "reset_during_inflight_fsync_does_not_resurrect_stale_durable_index".into(), ); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - Arc::new(storage), - ); - let raft_log = raft_log.start(receiver, None); - std::thread::sleep(Duration::from_millis(10)); // ensure IO thread is ready + let raft_log = RaftLogCore::::new(1, Arc::new(storage), None, 5000); // Append triggers the automatic round (round 1), which wins the CAS and // blocks on the gate. @@ -701,13 +545,13 @@ async fn test_reset_during_inflight_fsync_does_not_resurrect_stale_durable_index .unwrap(); tokio::time::sleep(Duration::from_millis(50)).await; - // X arrives while round 1 is gated β€” queues behind it in FsyncCoordinator. + // X arrives while round 1 is gated β€” queues behind it in FsyncWorker. let raft_log_x = raft_log.clone(); let x = tokio::spawn(async move { raft_log_x.flush().await }); tokio::time::sleep(Duration::from_millis(50)).await; // reset() while round 1 is still gated: bumps the fence generation, then - // clears durable_index/max_index/entries to 0. + // clears durable_index/memory_max_index/entries to 0. raft_log.reset().await.unwrap(); // Release the gate: round 1's flush() call returns, but its generation no @@ -736,40 +580,14 @@ async fn test_reset_during_inflight_fsync_does_not_resurrect_stale_durable_index /// (in a fresh round, not the stale one) must still advance `durable_index` /// normally β€” the fence only discards the round that was in flight *before* /// the reset, not everything that comes after it. -/// -/// Scenario: -/// 1. Same setup as above: gate a round, call `reset()` while it's stalled. -/// 2. Before releasing the gate, append new entries and call `flush()` β€” -/// this new round queues behind the still-gated stale round. -/// 3. Release the gate β€” the stale round discards itself (per the test -/// above); `run_until_caught_up` loops back, captures a FRESH generation -/// at the top of the next iteration, and processes the new round. -/// -/// Expected: -/// - The new (post-reset) `flush()` caller's future resolves to `Ok(())`. -/// - `durable_index()` advances to reflect the post-reset entries β€” the -/// fence must not discard legitimate work just because a reset happened -/// at some point in the task's lifetime; only the round that was -/// in-flight *at the moment of reset* is stale. #[tokio::test] async fn test_post_reset_writes_are_not_discarded_by_stale_fence() { let (storage, flush_gate) = MockStorageEngine::not_durable_gated_flush( "post_reset_writes_are_not_discarded_by_stale_fence".into(), ); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - Arc::new(storage), - ); - let raft_log = raft_log.start(receiver, None); - std::thread::sleep(Duration::from_millis(10)); // ensure IO thread is ready + let (log_flush_tx, mut log_flush_rx) = mpsc::unbounded_channel(); + let raft_log = + RaftLogCore::::new(1, Arc::new(storage), Some(log_flush_tx), 5000); // Append triggers the automatic round (round 1), which wins the CAS and // blocks on the gate. @@ -784,7 +602,7 @@ async fn test_post_reset_writes_are_not_discarded_by_stale_fence() { tokio::time::sleep(Duration::from_millis(50)).await; // reset() while round 1 is still gated: bumps the fence generation and - // drains anything already queued (fence_reset()). + // drains anything already queued. raft_log.reset().await.unwrap(); // New, post-reset entry β€” a fresh write, unrelated to the stale round. @@ -819,6 +637,7 @@ async fn test_post_reset_writes_are_not_discarded_by_stale_fence() { result.is_ok(), "Y's flush() must succeed β€” its data was written after reset, not stale" ); + drain_and_apply_fsync_completions(&raft_log, &mut log_flush_rx); assert_eq!( raft_log.durable_index(), 1, diff --git a/d-engine-core/src/storage/buffered_raft_log_test/concurrent_operations_test.rs b/d-engine-core/src/storage/raft_log_core_test/concurrent_operations_test.rs similarity index 88% rename from d-engine-core/src/storage/buffered_raft_log_test/concurrent_operations_test.rs rename to d-engine-core/src/storage/raft_log_core_test/concurrent_operations_test.rs index 2b345e79..616bae67 100644 --- a/d-engine-core/src/storage/buffered_raft_log_test/concurrent_operations_test.rs +++ b/d-engine-core/src/storage/raft_log_core_test/concurrent_operations_test.rs @@ -4,19 +4,12 @@ use futures::future::join_all; use tokio; use crate::storage::raft_log::RaftLog; -use crate::test_utils::{BufferedRaftLogTestContext, simulate_insert_command}; -use crate::{FlushPolicy, PersistenceStrategy}; +use crate::test_utils::{RaftLogCoreTestContext, simulate_insert_command}; use d_engine_proto::common::{Entry, LogId}; #[tokio::test] async fn test_remove_range_with_concurrent_reads() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_remove_range_with_concurrent_reads", - ); + let ctx = RaftLogCoreTestContext::new("test_remove_range_with_concurrent_reads"); ctx.raft_log.reset().await.expect("reset successfully!"); // Insert 1000 entries @@ -53,13 +46,7 @@ async fn test_remove_range_with_concurrent_reads() { #[tokio::test] async fn test_concurrent_append_and_purge() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 50, - }, - "test_concurrent_append_purge", - ); + let ctx = RaftLogCoreTestContext::new("test_concurrent_append_purge"); // Pre-populate simulate_insert_command(&ctx.raft_log, (1..=1000).collect(), 1).await; @@ -127,13 +114,7 @@ async fn test_get_entries_range_never_returns_torn_result_during_concurrent_purg const TOTAL: u64 = 2000; const ITERATIONS: usize = 500; - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_get_entries_range_never_returns_torn_result", - ); + let ctx = RaftLogCoreTestContext::new("test_get_entries_range_never_returns_torn_result"); ctx.raft_log.reset().await.expect("reset successfully!"); simulate_insert_command(&ctx.raft_log, (1..=TOTAL).collect(), 1).await; diff --git a/d-engine-core/src/storage/raft_log_core_test/content_validated_watermark_test.rs b/d-engine-core/src/storage/raft_log_core_test/content_validated_watermark_test.rs new file mode 100644 index 00000000..0f5d63d6 --- /dev/null +++ b/d-engine-core/src/storage/raft_log_core_test/content_validated_watermark_test.rs @@ -0,0 +1,185 @@ +//! Content-validated `durable_index` advance (#446 single-owner redesign). +//! Tests `try_advance_durable_index(index, term) -> Option`: +//! `Some(new)` only when it actually advanced, `None` when rejected as stale +//! (`entry_term(index) != Some(term)`) or already applied. +//! +//! Why no thread races / timing gates here (unlike `truncation_fsync_fence_test.rs`): +//! `durable_index` has one owner β€” raft.rs's event loop. The fsync-completion +//! report and a truncation are just two sequential calls on that thread, so no +//! interleaving inside a function body is possible. Each test drives one +//! arrival order directly. + +use std::sync::Arc; +use std::time::Duration; + +use d_engine_proto::common::Entry; +use d_engine_proto::common::LogId; + +use crate::storage::raft_log::RaftLog; +use crate::test_utils::RaftLogCoreTestContext; +use crate::{MockStorageEngine, MockTypeConfig, RaftLogCore}; + +fn entry( + index: u64, + term: u64, +) -> Entry { + Entry { + index, + term, + payload: None, + } +} + +async fn new_raft_log() -> Arc> { + let storage = Arc::new(MockStorageEngine::with_id( + "content_validated_watermark_test".into(), + )); + RaftLogCore::::new(1, storage, None, 5000) +} + +/// Business scenario: follower has entries 1..=100 under term 1. A physical +/// fsync for "up to 100" is still in flight when a new leader (term 2) +/// truncates 81..=100 and replaces it with its own entries. The in-flight +/// fsync's completion β€” a report for (index=100, term=1) β€” arrives after the +/// replacement. Index 100 still exists, but it's term 2 now: the report +/// describes content that's gone. +/// +/// Expected: rejected. `durable_index` must not move to 100. +#[tokio::test] +async fn test_stale_durable_report_rejected_when_term_no_longer_matches() { + let raft_log = new_raft_log().await; + + let term1_entries: Vec = (1..=100).map(|i| entry(i, 1)).collect(); + raft_log.append_entries(term1_entries).await.unwrap(); + + // New leader (term 2) truncates 81..=100 and replaces with its own tail. + let term2_tail: Vec = (81..=100).map(|i| entry(i, 2)).collect(); + raft_log.filter_out_conflicts_and_append(80, 1, term2_tail).await.unwrap(); + + // The stale in-flight fsync's report, generated before the truncation. + let result = raft_log.try_advance_durable_index(LogId { + term: 1, + index: 100, + }); + + assert_eq!( + result, None, + "a durable report for term=1 must be rejected once index 100 belongs to term=2" + ); + assert!( + raft_log.durable_index() < 81, + "durable_index ({}) must not advance into the replaced [81,100] range \ + on a rejected report", + raft_log.durable_index() + ); +} + +/// Sanity check: an unremarkable report (no truncation involved) must still +/// be applied. The new validation must not reject everything. +#[tokio::test] +async fn test_durable_report_accepted_when_term_still_matches() { + let raft_log = new_raft_log().await; + + let entries: Vec = (1..=100).map(|i| entry(i, 1)).collect(); + raft_log.append_entries(entries).await.unwrap(); + + let result = raft_log.try_advance_durable_index(LogId { + term: 1, + index: 100, + }); + + assert_eq!( + result, + Some(100), + "a report matching current log content must be applied" + ); + assert_eq!(raft_log.durable_index(), 100); +} + +/// Same scenario as `test_stale_durable_report_rejected_when_term_no_longer_matches`, +/// but the report arrives BEFORE the truncation instead of after β€” the other +/// possible arrival order. Under single ownership both orders must land on +/// the same final state, because the owner processes one event at a time +/// rather than racing a background write against a live update. +#[tokio::test] +async fn test_durable_report_then_truncation_is_order_independent() { + let raft_log = new_raft_log().await; + + let term1_entries: Vec = (1..=100).map(|i| entry(i, 1)).collect(); + raft_log.append_entries(term1_entries).await.unwrap(); + + // Report arrives first, while the log is still all term 1 β€” legitimately + // applied at this point in time. + let result = raft_log.try_advance_durable_index(LogId { + term: 1, + index: 100, + }); + assert_eq!(result, Some(100)); + assert_eq!(raft_log.durable_index(), 100); + + // Truncation arrives after β€” must still clamp durable_index down, + // exactly as it does today via `remove_range`'s existing fetch_min. + let term2_tail: Vec = (81..=100).map(|i| entry(i, 2)).collect(); + raft_log.filter_out_conflicts_and_append(80, 1, term2_tail).await.unwrap(); + + assert!( + raft_log.durable_index() < 81, + "truncation must clamp durable_index down to 80 regardless of the \ + earlier report having advanced it to 100, durable_index is {}", + raft_log.durable_index() + ); +} + +/// Regression test for the `flush()` short-circuit (`durable_index >= +/// memory_max_index`). This isn't proving a +/// live bug in the current design (`remove_range` clamps `durable_index` +/// synchronously, so the short-circuit's precondition always holds) β€” it's +/// pinning down that invariant so a future change that defers the clamp +/// doesn't silently reopen the RPO=0 violation +/// this whole fix was for: `flush()` returning `Ok(())` before the real, +/// post-truncation tail has actually been fsynced. +/// +/// Scenario: entries 1..=100 durable. New leader truncates 81..=100 (term 2 +/// tail 81..=85 replaces it) β€” `durable_index` clamps to 80, +/// `memory_max_index` becomes 85. Calling `flush()` right after must NOT +/// take the short-circuit (80 < 85) β€” it must dispatch a real physical +/// flush for the new, not-yet-synced tail. +#[tokio::test] +async fn test_flush_does_not_short_circuit_after_truncation_regrows_the_log() { + let (mut ctx, flush_count) = + RaftLogCoreTestContext::new_not_durable("flush_no_short_circuit_after_truncation"); + + ctx.append_entries(1, 100, 1).await; + ctx.raft_log.flush().await.unwrap(); + ctx.drain_fsync_completions(); + assert_eq!( + ctx.raft_log.durable_index(), + 100, + "baseline must be fully durable" + ); + + let flushes_before_truncation = flush_count.load(std::sync::atomic::Ordering::Relaxed); + + // New leader (term 2) truncates 81..=100, replaces with its own tail + // 81..=85 β€” durable_index clamps to 80, memory_max_index becomes 85. + let term2_tail: Vec = (81..=85).map(|i| entry(i, 2)).collect(); + ctx.raft_log.filter_out_conflicts_and_append(80, 1, term2_tail).await.unwrap(); + assert!(ctx.raft_log.durable_index() < 81, "clamp must have fired"); + assert_eq!(ctx.raft_log.last_entry_id(), 85); + + ctx.raft_log.flush().await.unwrap(); + tokio::time::sleep(Duration::from_millis(20)).await; + ctx.drain_fsync_completions(); + + let flushes_after = flush_count.load(std::sync::atomic::Ordering::Relaxed); + assert!( + flushes_after > flushes_before_truncation, + "flush() must dispatch a real physical flush for the new tail, not \ + short-circuit on a stale-looking durable_index β€” before={flushes_before_truncation}, after={flushes_after}" + ); + assert_eq!( + ctx.raft_log.durable_index(), + 85, + "the new tail must actually become durable, not just claimed so" + ); +} diff --git a/d-engine-core/src/storage/buffered_raft_log_test/drain_fsync_test.rs b/d-engine-core/src/storage/raft_log_core_test/drain_fsync_test.rs similarity index 59% rename from d-engine-core/src/storage/buffered_raft_log_test/drain_fsync_test.rs rename to d-engine-core/src/storage/raft_log_core_test/drain_fsync_test.rs index 838cc386..dcf995e5 100644 --- a/d-engine-core/src/storage/buffered_raft_log_test/drain_fsync_test.rs +++ b/d-engine-core/src/storage/raft_log_core_test/drain_fsync_test.rs @@ -11,42 +11,38 @@ //! 3. **Explicit flush barrier**: `flush()` waits until all entries written before //! the call are durable, regardless of how many internal fsyncs occurred. -use crate::BufferedRaftLog; use crate::Error; use crate::HardState; -use crate::IOTask; use crate::MockLogStore; use crate::MockMetaStore; use crate::MockStorageEngine; use crate::MockTypeConfig; -use crate::PersistenceConfig; -use crate::PersistenceStrategy; +use crate::RaftLogCore; use d_engine_proto::common::Entry; use d_engine_proto::common::LogId; use std::sync::Arc; +use std::sync::Mutex; +use std::sync::atomic::AtomicU64; use std::sync::atomic::Ordering; use tokio::sync::mpsc; use tokio::time::timeout; use tokio::time::{Duration, sleep}; -use crate::test_utils::BufferedRaftLogTestContext; -use crate::{FlushPolicy, RaftLog}; +use crate::RaftLog; +use crate::test_utils::RaftLogCoreTestContext; -/// The IO thread auto-fsyncs on each write_notify wakeup without any timer. +/// Each `append_entries` persists the new tail and auto-submits it to `FsyncWorker` +/// for fsync β€” no timer, no explicit `flush()` required. /// -/// After `append_entries`, the IO thread is notified via `write_notify`, reads -/// pending entries from the SkipMap, and calls fsync. `durable_index` advances -/// automatically β€” no explicit `flush()` required. +/// After `append_entries`, the new tail is written to the page cache and handed to +/// `FsyncWorker`, which fsyncs it. `durable_index` advances automatically once the +/// fsync completes and raft.rs drains the `FsyncCompleted` event. #[tokio::test] async fn test_writes_become_durable_via_io_thread() { - let (ctx, flush_count) = BufferedRaftLogTestContext::new_not_durable( - FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - "writes_become_durable_via_io_thread", - ); + let (mut ctx, flush_count) = + RaftLogCoreTestContext::new_not_durable("writes_become_durable_via_io_thread"); - // Append 5 entries β€” each calls write_notify.notify_one(). + // Append 5 entries β€” each persists and submits the new tail to FsyncWorker. for i in 1u64..=5 { ctx.raft_log .append_entries(vec![Entry { @@ -67,40 +63,36 @@ async fn test_writes_become_durable_via_io_thread() { ); } - // Give IO thread time to process write_notify wakeup and fsync. + // Give FsyncWorker time to process the submit and fsync. sleep(Duration::from_millis(50)).await; + ctx.drain_fsync_completions(); - // durable_index must have advanced via IO thread auto-fsync (no explicit flush). + // durable_index must have advanced via FsyncWorker auto-fsync (no explicit flush). assert_eq!( ctx.raft_log.durable_index(), 5, - "durable_index must advance via IO thread drain-then-fsync" + "durable_index must advance via FsyncWorker fsync" ); - // IO thread called log_store.flush() at least once. + // FsyncWorker called log_store.flush() at least once. assert!( flush_count.load(Ordering::Relaxed) >= 1, - "IO thread must have called flush at least once" + "FsyncWorker must have called flush at least once" ); } /// A single `append_entries` call with 100 entries batches into far fewer fsyncs /// than individual writes. /// -/// `append_entries` calls `write_notify.notify_one()` once regardless of how many -/// entries are in the batch. The IO thread wakes once, persists all entries to page -/// cache, then dispatches fsync via `FsyncCoordinator`. The explicit `flush()` call +/// `append_entries` persists the whole batch and submits one fsync regardless of how many +/// entries are in the batch. The new tail is written to page cache once, then fsynced via +/// `FsyncWorker`. The explicit `flush()` call /// may race with the spawned fsync task: if it observes `durable_index` before the /// first task completes, it submits a second round (coalesced by the coordinator). /// N entries in one call β†’ ≀2 fsyncs (not N), regardless of storage speed. #[tokio::test] async fn test_batch_append_produces_one_flush() { - let (ctx, flush_count) = BufferedRaftLogTestContext::new_not_durable( - FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - "batch_append_one_flush", - ); + let (mut ctx, flush_count) = RaftLogCoreTestContext::new_not_durable("batch_append_one_flush"); // All 100 entries in one append_entries call. let entries: Vec = (1u64..=100) @@ -113,11 +105,12 @@ async fn test_batch_append_produces_one_flush() { ctx.raft_log.append_entries(entries).await.unwrap(); ctx.raft_log.flush().await.unwrap(); + ctx.drain_fsync_completions(); assert_eq!(ctx.raft_log.durable_index(), 100); - // One notify_one() β†’ IO thread wakes once β†’ far fewer fsyncs than entries. - // With FsyncCoordinator the explicit flush() may add one extra round if it + // One append β†’ one fsync submit β†’ far fewer fsyncs than entries. + // With FsyncWorker the explicit flush() may add one extra round if it // races with the in-flight spawned task; the invariant is "not N flushes". let flushes = flush_count.load(Ordering::Relaxed); assert!( @@ -126,17 +119,15 @@ async fn test_batch_append_produces_one_flush() { ); } -/// `IOTask::Reset` must zero `pending_max`; stale value corrupts `durable_index`. +/// `reset()` must not let a stale `pending_max` corrupt `durable_index`. /// /// ## Background -/// `batch_processor` tracks `pending_max`: the highest log index written to the OS -/// page cache but not yet fsynced. After a successful `fsync_and_advance`, it is -/// zeroed (`pending_max = 0`). After a **failed** fsync, it is NOT zeroed β€” the -/// `else { pending_max = 0 }` branch is skipped. +/// `FsyncWorker::pending_max` holds the highest mark written to the OS page cache +/// but not yet fsynced. `reset()` calls `fence_reset()`, which zeroes it. /// -/// ## Original bug (fixed pre-#422) -/// `handle_non_write_cmd(IOTask::Reset)` wiped the on-disk log but did NOT zero -/// `pending_max`. On the next `write_notify` wakeup the IO thread would compute: +/// ## Original bug (fixed pre-#422, in the since-removed IO-thread design) +/// The IO thread's reset path wiped the on-disk log but did NOT zero +/// `pending_max`. On the next wakeup the IO thread would compute: /// ``` /// pending_max = pending_max.max(new_end) // stale 10 wins over new 3 /// fsync_and_advance(10) // advances durable_index to 10 β€” WRONG @@ -153,26 +144,13 @@ async fn test_pending_max_zeroed_on_reset_preventing_durable_index_corruption() let storage = Arc::new(MockStorageEngine::not_durable_first_flush_fails( "pending_max_zeroed_on_reset".into(), )); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - // Safety-net disabled: only write_notify triggers fsync. - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - storage, - ); - let raft_log = raft_log.start(receiver, None); - std::thread::sleep(Duration::from_millis(10)); // ensure IO thread is ready + let raft_log = RaftLogCore::::new(1, storage, None, 5000); + std::thread::sleep(Duration::from_millis(10)); // let the spawned fsync task start // Phase 1: append 10 entries. - // IO thread wakes on write_notify: persist succeeds, fsync FAILS (first call). + // FsyncWorker persists then fsyncs: persist succeeds, fsync FAILS (first call). // Failed fsync leaves pending_max = 10 (not zeroed β€” only success path zeros - // it) AND now poisons the log permanently (see FsyncCoordinator). + // it) AND now poisons the log permanently (see FsyncWorker). let entries: Vec = (1u64..=10) .map(|i| Entry { index: i, @@ -181,7 +159,7 @@ async fn test_pending_max_zeroed_on_reset_preventing_durable_index_corruption() }) .collect(); raft_log.append_entries(entries).await.unwrap(); - // Wait for IO thread to process write_notify (persist ok, fsync fails). + // Wait for FsyncWorker to process the submit (persist ok, fsync fails). sleep(Duration::from_millis(20)).await; assert!( raft_log.is_poisoned(), @@ -198,7 +176,7 @@ async fn test_pending_max_zeroed_on_reset_preventing_durable_index_corruption() // Phase 3: attempt to append 3 new entries starting from index 1 β€” this is // the exact sequence that used to reproduce the durable_index corruption. - // It must now be rejected outright, never reaching the IO thread at all. + // It must now be rejected outright, never reaching FsyncWorker at all. let new_entries: Vec = (1u64..=3) .map(|i| Entry { index: i, @@ -221,12 +199,8 @@ async fn test_pending_max_zeroed_on_reset_preventing_durable_index_corruption() /// flush() call must be durable when flush() returns, regardless of internal batching. #[tokio::test] async fn test_flush_is_strict_durability_barrier() { - let (ctx, _flush_count) = BufferedRaftLogTestContext::new_not_durable( - FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - "flush_durability_barrier", - ); + let (mut ctx, _flush_count) = + RaftLogCoreTestContext::new_not_durable("flush_durability_barrier"); // First batch. for i in 1u64..=20 { @@ -240,6 +214,7 @@ async fn test_flush_is_strict_durability_barrier() { .unwrap(); } ctx.raft_log.flush().await.unwrap(); + ctx.drain_fsync_completions(); assert_eq!( ctx.raft_log.durable_index(), 20, @@ -258,6 +233,7 @@ async fn test_flush_is_strict_durability_barrier() { .unwrap(); } ctx.raft_log.flush().await.unwrap(); + ctx.drain_fsync_completions(); assert_eq!( ctx.raft_log.durable_index(), 50, @@ -267,37 +243,24 @@ async fn test_flush_is_strict_durability_barrier() { /// flush() must return Err when the underlying fsync fails β€” not hang indefinitely. /// -/// ## Bug (pre-fix) -/// `flush()` sends `IOTask::FlushNow` (fire-and-forget), then calls `wait_durable()` -/// which registers a `WaitDurable` waiter. If the IO thread's fsync fails, it logs the -/// error and moves on β€” `durable_index` never advances, the waiter is never notified, -/// and `flush()` blocks forever. +/// ## Bug (pre-fix, in the since-removed IO-thread design) +/// `flush()` sent a fire-and-forget flush request, then registered a durable waiter. +/// If the IO thread's fsync failed, it logged the error and moved on β€” `durable_index` +/// never advanced, the waiter was never notified, and `flush()` blocked forever. /// -/// ## Fix (#331) -/// Replace `FlushNow` + `WaitDurable` with a single `IOTask::Flush(oneshot::Sender>)`. -/// The IO thread performs fsync and sends the result (Ok or Err) directly back to the -/// caller via the oneshot channel, so `flush()` always returns within bounded time. +/// ## Invariant (#331) +/// `flush()` submits to `FsyncWorker` together with a reply oneshot. The worker sends +/// the fsync result (Ok or Err) directly back through it, so `flush()` always returns +/// within bounded time. #[tokio::test] async fn test_flush_propagates_io_error() { // Every flush() call on the underlying log store returns an error. - // This covers both the auto-fsync triggered by write_notify and the explicit + // This covers both the auto-fsync triggered by append_entries and the explicit // flush() call, so timing between the two does not affect the outcome. let storage = Arc::new(MockStorageEngine::not_durable_always_failing_flush( "flush_propagates_io_error".into(), )); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - storage, - ); - let raft_log = raft_log.start(receiver, None); + let raft_log = RaftLogCore::::new(1, storage, None, 5000); std::thread::sleep(Duration::from_millis(10)); raft_log @@ -332,7 +295,7 @@ async fn test_flush_propagates_io_error() { } /// Test: a real fsync failure poisons the log end-to-end (via the actual -/// `FsyncCoordinator` failure path, not by forcing the flag directly like +/// `FsyncWorker` failure path, not by forcing the flag directly like /// `test_poisoned_survives_reset` does), and poisoning survives a /// subsequent `reset()` β€” writes afterward are rejected. /// @@ -347,22 +310,10 @@ async fn test_fsync_failure_poisons_and_rejects_writes_after_reset() { let storage = Arc::new(MockStorageEngine::not_durable_first_flush_fails( "fsync_failure_poisons_and_rejects_writes_after_reset".into(), )); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - storage, - ); - let raft_log = raft_log.start(receiver, None); - std::thread::sleep(Duration::from_millis(10)); // ensure IO thread is ready + let raft_log = RaftLogCore::::new(1, storage, None, 5000); + std::thread::sleep(Duration::from_millis(10)); // let the spawned fsync task start - // Trigger a real fsync failure via the actual FsyncCoordinator path. + // Trigger a real fsync failure via the actual FsyncWorker path. raft_log .append_entries(vec![Entry { index: 1, @@ -404,20 +355,8 @@ async fn test_replace_range_failure_poisons() { let storage = Arc::new(MockStorageEngine::not_durable_replace_range_fails( "replace_range_failure_poisons".into(), )); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - storage, - ); - let raft_log = raft_log.start(receiver, None); - std::thread::sleep(Duration::from_millis(10)); // ensure IO thread is ready + let raft_log = RaftLogCore::::new(1, storage, None, 5000); + std::thread::sleep(Duration::from_millis(10)); // let the spawned fsync task start // Base log: 3 entries at term 1. raft_log @@ -443,7 +382,7 @@ async fn test_replace_range_failure_poisons() { sleep(Duration::from_millis(20)).await; // A new leader (term 2) overwrites from index 2 onward β€” a real conflict - // that must truncate + replace, routing through IOTask::ReplaceRange. + // that must truncate + replace, routing through `replace_range_and_submit`. let result = raft_log .filter_out_conflicts_and_append( 1, @@ -486,27 +425,15 @@ async fn test_replace_range_failure_poisons() { } /// A `purge()` failure poisons the log, same as the other storage-layer -/// failures. `purge_logs_up_to()`'s own `done` channel is `oneshot::Sender<()>` -/// (no `Result`), so it always returns `Ok` regardless of the underlying -/// outcome β€” poisoning is the only way this failure becomes observable. +/// failures, and now propagates as a real `Err` to the caller (unlike the +/// old `oneshot::Sender<()>` done-channel, `purge_and_advance` returns +/// `Result<()>` directly). #[tokio::test] async fn test_purge_failure_poisons() { let storage = Arc::new(MockStorageEngine::not_durable_purge_fails( "purge_failure_poisons".into(), )); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - storage, - ); - let raft_log = raft_log.start(receiver, None); + let raft_log = RaftLogCore::::new(1, storage, None, 5000); std::thread::sleep(Duration::from_millis(10)); raft_log @@ -519,8 +446,11 @@ async fn test_purge_failure_poisons() { .unwrap(); sleep(Duration::from_millis(20)).await; - raft_log.purge_logs_up_to(LogId { term: 1, index: 1 }).await.unwrap(); // always Ok β€” see doc comment above - sleep(Duration::from_millis(20)).await; + let result = raft_log.purge_logs_up_to(LogId { term: 1, index: 1 }).await; + assert!( + result.is_err(), + "a real purge() failure must now be propagated to the caller" + ); assert!( raft_log.is_poisoned(), @@ -544,19 +474,7 @@ async fn test_reset_failure_poisons() { let storage = Arc::new(MockStorageEngine::not_durable_reset_fails( "reset_failure_poisons".into(), )); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - storage, - ); - let raft_log = raft_log.start(receiver, None); + let raft_log = RaftLogCore::::new(1, storage, None, 5000); std::thread::sleep(Duration::from_millis(10)); let result = raft_log.reset().await; @@ -593,19 +511,7 @@ async fn test_save_hard_state_failure_poisons() { let storage = Arc::new(MockStorageEngine::not_durable_save_hard_state_fails( "save_hard_state_failure_poisons".into(), )); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - storage, - ); - let raft_log = raft_log.start(receiver, None); + let raft_log = RaftLogCore::::new(1, storage, None, 5000); std::thread::sleep(Duration::from_millis(10)); let result = raft_log.save_hard_state(&HardState { @@ -642,19 +548,7 @@ async fn test_poisoned_rejects_save_hard_state() { let storage = Arc::new(MockStorageEngine::with_id( "poisoned_rejects_save_hard_state".into(), )); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - storage, - ); - let raft_log = raft_log.start(receiver, None); + let raft_log = RaftLogCore::::new(1, storage, None, 5000); std::thread::sleep(Duration::from_millis(10)); raft_log.poisoned.store(true, Ordering::SeqCst); @@ -674,7 +568,7 @@ async fn test_poisoned_rejects_save_hard_state() { } // ============================================================================ -// Gap fix: handle_non_write_cmd now checks is_poisoned() before executing +// Gap fix: run_storage_tasks now checks is_poisoned() before executing // ReplaceRange/Purge/Reset, instead of only checking it in run_batch_turn's // drain loop (which missed the direct-dispatch path in batch_processor's // top-level select, and the "just poisoned mid-turn" race). @@ -690,19 +584,7 @@ async fn test_poisoned_skips_replace_range() { let storage = Arc::new(MockStorageEngine::not_durable_replace_range_fails( "poisoned_skips_replace_range".into(), )); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - storage, - ); - let raft_log = raft_log.start(receiver, None); + let raft_log = RaftLogCore::::new(1, storage, None, 5000); std::thread::sleep(Duration::from_millis(10)); // Base log, written while still healthy. @@ -772,19 +654,7 @@ async fn test_poisoned_does_not_skip_reset() { let storage = Arc::new(MockStorageEngine::with_id( "poisoned_does_not_skip_reset".into(), )); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - storage, - ); - let raft_log = raft_log.start(receiver, None); + let raft_log = RaftLogCore::::new(1, storage, None, 5000); std::thread::sleep(Duration::from_millis(10)); raft_log.poisoned.store(true, Ordering::SeqCst); @@ -800,177 +670,35 @@ async fn test_poisoned_does_not_skip_reset() { ); } -/// Once already poisoned, `Purge` should be rejected immediately without -/// calling `log_store.purge()` β€” but this can only be checked weakly. -/// Unlike ReplaceRange/Reset, `Purge`'s `done` channel is -/// `oneshot::Sender<()>` and cannot carry error text, so -/// `purge_logs_up_to()` always returns `Ok` whether the underlying purge -/// ran or was skipped. This is a pre-existing API limitation, not something -/// this test can close β€” it only confirms poisoned stays true and the call -/// doesn't hang. +/// Once already poisoned, `Purge` must be rejected immediately without ever +/// calling `log_store.purge()`. Proven by checking the error message: if the +/// underlying mock's own failure had been reached, the message would differ +/// from the immediate "raft log storage is poisoned" rejection β€” the same +/// pattern `test_poisoned_skips_replace_range` uses. `purge_and_advance` now +/// returns `Result<()>` directly, so this is no longer just a weak check. #[tokio::test] async fn test_poisoned_skips_purge() { let storage = Arc::new(MockStorageEngine::not_durable_purge_fails( "poisoned_skips_purge".into(), )); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - storage, - ); - let raft_log = raft_log.start(receiver, None); + let raft_log = RaftLogCore::::new(1, storage, None, 5000); std::thread::sleep(Duration::from_millis(10)); + // try_advance_durable_index() clamps against memory_max_index β€” simulate + // a log that already has the entry this test purges up to. + raft_log.set_memory_max_index_for_test(1); raft_log.poisoned.store(true, Ordering::SeqCst); let result = raft_log.purge_logs_up_to(LogId { term: 1, index: 1 }).await; - assert!( - result.is_ok(), - "purge_logs_up_to() always returns Ok regardless of the underlying outcome" - ); - assert!(raft_log.is_poisoned(), "poisoned must remain true"); -} - -/// Test: a failed `IOTask::ReplaceRange` mid-batch-turn causes an explicit -/// `Err` reply to any `Flush` already queued in that same turn (fixed -/// 2026-07-19 β€” `run_batch_turn`'s drain loop now replies before returning, -/// instead of silently dropping the oneshot sender). -/// -/// Ordering is made deterministic (not timing-sensitive) by gating the IO -/// thread inside its first `persist_entries()` call. While it's blocked, an -/// `IOTask::Flush` is sent directly (guaranteed FIFO-first) followed by a -/// conflict-triggering `filter_out_conflicts_and_append` call (sends -/// `IOTask::ReplaceRange` second). Releasing the gate lets `run_batch_turn` -/// drain both in one pass, in that order. -#[tokio::test] -async fn test_run_batch_turn_replace_range_failure_replies_err_to_queued_flush() { - let (gate_tx, gate_rx) = std::sync::mpsc::channel::<()>(); - let gate_rx = std::sync::Mutex::new(Some(gate_rx)); - - let mut log_store = MockLogStore::new(); - log_store.expect_last_index().returning(|| 0); - log_store.expect_persist_entries().returning(move |_| { - // Only the very first call (the base-entries append) blocks. - if let Some(gate) = gate_rx.lock().unwrap().take() { - let _ = gate.recv(); - } - Ok(()) - }); - log_store.expect_replace_range().returning(|_, _| { - Err(crate::Error::Fatal( - "simulated replace_range failure".into(), - )) - }); - log_store.expect_entry().returning(|_| Ok(None)); - log_store.expect_get_entries().returning(|_| Ok(vec![])); - log_store.expect_purge().returning(|_| Ok(())); - log_store.expect_load_purge_boundary().returning(|| Ok(None)); - log_store.expect_reset().returning(|| Ok(())); - log_store.expect_truncate().returning(|_| Ok(())); - log_store.expect_is_write_durable().returning(|| true); - log_store.expect_flush().returning(|| Ok(())); - log_store.expect_flush_async().returning(|| Ok(())); - - let mut meta_store = MockMetaStore::new(); - meta_store.expect_save_hard_state().returning(|_| Ok(())); - meta_store.expect_load_hard_state().returning(|| Ok(None)); - meta_store.expect_flush().returning(|| Ok(())); - meta_store.expect_flush_async().returning(|| Ok(())); - - let storage = Arc::new(MockStorageEngine::from(log_store, meta_store)); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - storage, - ); - let raft_log = raft_log.start(receiver, None); - std::thread::sleep(Duration::from_millis(10)); - - // Base entries land in memory synchronously; the IO thread wakes and - // immediately blocks inside the gated persist_entries() call, before it - // ever drains the command queue. - raft_log - .append_entries(vec![ - Entry { - index: 1, - term: 1, - payload: None, - }, - Entry { - index: 2, - term: 1, - payload: None, - }, - Entry { - index: 3, - term: 1, - payload: None, - }, - ]) - .await - .unwrap(); - sleep(Duration::from_millis(20)).await; // let the IO thread reach the gate - - // Send Flush directly β€” guarantees it's enqueued before the ReplaceRange - // sent below, so it's the one already sitting in `replies` when the - // ReplaceRange failure triggers the early return. - let (flush_tx, flush_rx) = tokio::sync::oneshot::channel(); - raft_log.command_sender.send(IOTask::Flush(flush_tx)).unwrap(); - - let conflict_raft_log = raft_log.clone(); - let conflict_task = tokio::spawn(async move { - conflict_raft_log - .filter_out_conflicts_and_append( - 1, - 1, - vec![ - Entry { - index: 2, - term: 2, - payload: None, - }, - Entry { - index: 3, - term: 2, - payload: None, - }, - ], - ) - .await - }); - sleep(Duration::from_millis(50)).await; // let the ReplaceRange send land - gate_tx.send(()).unwrap(); - - let flush_result = timeout(Duration::from_secs(2), flush_rx) - .await - .expect("flush reply must not hang"); - match flush_result { - Ok(Err(e)) => { - // Expected: explicit Err reply, not a dropped-sender RecvError. - debug_assert!(!format!("{e:?}").is_empty()); - } - Ok(Ok(())) => panic!("a Flush queued alongside a fatal ReplaceRange must not succeed"), - Err(_recv_err) => panic!( - "flush reply sender was dropped without a reply β€” the fix should have \ - sent an explicit Err instead of silently closing the channel" + match result { + Err(Error::Fatal(msg)) => assert!( + msg.contains("poisoned"), + "expected the poisoned short-circuit to fire before purge() was \ + ever called, got: {msg}" ), + other => panic!("expected Err(Fatal(\"...poisoned...\")), got: {other:?}"), } - - let _ = timeout(Duration::from_secs(2), conflict_task).await; + assert!(raft_log.is_poisoned(), "poisoned must remain true"); } // ============================================================================ @@ -983,22 +711,10 @@ async fn test_run_batch_turn_replace_range_failure_replies_err_to_queued_flush() /// Guards against a constructor regression (e.g. a copy-paste default flip) /// that would make every node refuse writes from the very first call. #[tokio::test] -async fn test_new_buffered_raft_log_starts_unpoisoned() { +async fn test_new_raft_log_core_starts_unpoisoned() { let (storage, _flush_call_count) = - MockStorageEngine::not_durable("new_buffered_raft_log_starts_unpoisoned".into()); - let (raft_log, _receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - // Safety-net disabled: only write_notify triggers fsync. - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - Arc::new(storage), - ); + MockStorageEngine::not_durable("new_raft_log_core_starts_unpoisoned".into()); + let raft_log = RaftLogCore::::new(1, Arc::new(storage), None, 5000); assert!( !raft_log.is_poisoned(), "A freshly constructed log is never born poisoned" @@ -1006,7 +722,7 @@ async fn test_new_buffered_raft_log_starts_unpoisoned() { } /// Once poisoned, the state survives `reset()` β€” it must NOT be cleared by -/// `reset_internal()` / `FsyncCoordinator::fence_reset()`. +/// `reset_internal()` / `FsyncWorker::fence_reset()`. /// /// Why this matters: `reset()` is also invoked mid-flight for legitimate /// reasons (snapshot install, log conflict rewind). If poisoned were treated @@ -1016,24 +732,8 @@ async fn test_new_buffered_raft_log_starts_unpoisoned() { /// this whole fix exists to close. #[tokio::test] async fn test_poisoned_survives_reset() { - let storage = - MockStorageEngine::not_durable_first_flush_fails("poisoned_survives_reset".into()); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - // Safety-net disabled: only write_notify triggers fsync. - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - Arc::new(storage), - ); - - let raft_log = raft_log.start(receiver, None); - std::thread::sleep(Duration::from_millis(10)); // ensure IO thread is ready + let storage = MockStorageEngine::with_id("poisoned_survives_reset".into()); + let raft_log = RaftLogCore::::new(1, Arc::new(storage), None, 5000); raft_log.poisoned.store(true, Ordering::SeqCst); assert!(raft_log.reset().await.is_ok()); @@ -1045,8 +745,12 @@ async fn test_poisoned_survives_reset() { /// A `persist_entries()` (page-cache write) failure poisons the log, exactly /// like an fsync failure does β€” these are two independent failure surfaces -/// (see `persist_pending_range` vs `FsyncCoordinator::run_until_caught_up`) -/// and both must reach the same fatal outcome. +/// (`persist_pending_range` vs `FsyncWorker::run_until_caught_up`) and both +/// must reach the same fatal outcome. +/// +/// `persist_pending_range` now runs inline, awaited directly inside +/// `append_entries`, so the poison lands synchronously β€” this same call +/// returns `Err`, not just some later one. /// /// Without this test, a bug that only wires up ONE of the two poisoning /// paths (e.g. fsync failures poison correctly, but persist_entries @@ -1058,33 +762,22 @@ async fn test_persist_entries_failure_poisons() { let storage = MockStorageEngine::not_durable_first_persist_fails( "persist_entries_failure_poisons".into(), ); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - Arc::new(storage), - ); - let raft_log = raft_log.start(receiver, None); - std::thread::sleep(Duration::from_millis(10)); // ensure IO thread is ready + let raft_log = RaftLogCore::::new(1, Arc::new(storage), None, 5000); + std::thread::sleep(Duration::from_millis(10)); - // Triggers the IO thread's persist_pending_range call, which hits the - // mock's first (failing) persist_entries() β€” this is the - // persist_pending_range poisoning path, NOT FsyncCoordinator's. - raft_log + // persist_pending_range hits the mock's first (failing) persist_entries() + // inline β€” this call itself must fail, not a later one. + let result = raft_log .append_entries(vec![Entry { index: 1, term: 1, payload: None, }]) - .await - .unwrap(); - sleep(Duration::from_millis(20)).await; // let the IO thread process it + .await; + assert!( + result.is_err(), + "the append_entries() call whose persist_entries fails must itself return Err" + ); assert!( raft_log.is_poisoned(), @@ -1118,26 +811,14 @@ async fn test_persist_entries_failure_poisons() { #[tokio::test] async fn test_notify_fatal_channel_closed_still_poisons_and_logs() { // tracing-test's #[traced_test] only captures events on this test's own - // thread β€” the fsync failure and its log line happen on BufferedRaftLog's - // dedicated IO thread, so a cross-thread-capable global subscriber is + // thread β€” the fsync failure and its log line happen on FsyncWorker's + // own execution thread, so a cross-thread-capable global subscriber is // needed instead (see test_utils::log_capture). let logs = crate::test_utils::capture_logs_globally(); let storage = MockStorageEngine::not_durable_first_flush_fails( "notify_fatal_channel_closed_still_poisons_and_logs".into(), ); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - Arc::new(storage), - ); // Build the InternalEvent channel but drop the receiver immediately β€” // by the time notify_fatal() runs, log_flush_tx.send() hits a closed @@ -1145,8 +826,7 @@ async fn test_notify_fatal_channel_closed_still_poisons_and_logs() { let (tx, rx) = mpsc::unbounded_channel(); drop(rx); - let raft_log = raft_log.start(receiver, Some(tx)); - std::thread::sleep(Duration::from_millis(10)); // ensure IO thread is ready + let raft_log = RaftLogCore::::new(1, Arc::new(storage), Some(tx), 5000); raft_log .append_entries(vec![Entry { @@ -1156,7 +836,7 @@ async fn test_notify_fatal_channel_closed_still_poisons_and_logs() { }]) .await .unwrap(); - sleep(Duration::from_millis(20)).await; // let the IO thread hit the fsync failure + sleep(Duration::from_millis(20)).await; // let the spawned fsync task hit the fsync failure assert!( raft_log.is_poisoned(), @@ -1168,3 +848,156 @@ async fn test_notify_fatal_channel_closed_still_poisons_and_logs() { not a silent no-op β€” see notify_fatal()'s error! call" ); } + +/// Efficiency: the persist scan must start from its own page-cache +/// frontier, not from `durable_index`. Since #446 `durable_index` only advances +/// after an `FsyncCompleted` round-trips through raft.rs's event loop; under +/// load it lags far behind what has already been written. If the scan +/// restarted from `durable_index + 1` on every wakeup, each of N appends would +/// re-scan and re-`persist_entries` the whole not-yet-durable window β€” O(N^2) +/// total work. +/// +/// This test pins `durable_index` at 0 (no `log_flush_tx`, so no +/// `FsyncCompleted` is ever consumed) and appends N entries one at a time. The +/// total number of entries handed to `persist_entries` across all calls must +/// stay ~N, not ~N^2/2. +#[tokio::test] +async fn test_persist_scan_tracks_frontier_not_stuck_durable_index() { + let persisted_total = Arc::new(AtomicU64::new(0)); + let persisted_total_c = persisted_total.clone(); + + let mut log_store = MockLogStore::new(); + log_store.expect_last_index().returning(|| 0); + log_store.expect_persist_entries().returning(move |entries| { + persisted_total_c.fetch_add(entries.len() as u64, Ordering::Relaxed); + Ok(()) + }); + log_store.expect_entry().returning(|_| Ok(None)); + log_store.expect_get_entries().returning(|_| Ok(vec![])); + log_store.expect_purge().returning(|_| Ok(())); + log_store.expect_load_purge_boundary().returning(|| Ok(None)); + log_store.expect_reset().returning(|| Ok(())); + log_store.expect_truncate().returning(|_| Ok(())); + log_store.expect_replace_range().returning(|from, new_entries| { + Ok(new_entries.last().map(|e| e.index).unwrap_or(from.saturating_sub(1))) + }); + log_store.expect_is_write_durable().returning(|| false); + log_store.expect_flush().returning(|| Ok(())); + log_store.expect_flush_async().returning(|| Ok(())); + + let mut meta_store = MockMetaStore::new(); + meta_store.expect_save_hard_state().returning(|_| Ok(())); + meta_store.expect_load_hard_state().returning(|| Ok(None)); + meta_store.expect_flush().returning(|| Ok(())); + meta_store.expect_flush_async().returning(|| Ok(())); + + let storage = Arc::new(MockStorageEngine::from(log_store, meta_store)); + // No log_flush_tx: FsyncCompleted is never consumed, so durable_index + // stays pinned at 0 for the whole test. + let raft_log = RaftLogCore::::new(1, storage, None, 5000); + + const N: u64 = 100; + for i in 1..=N { + raft_log + .append_entries(vec![Entry { + index: i, + term: 1, + payload: None, + }]) + .await + .unwrap(); + // flush() forces the persist up to memory_max_index right + // now, so the scan boundary is exercised once per append β€” deterministic, + // no sleeps. + raft_log.flush().await.unwrap(); + } + + assert_eq!( + raft_log.durable_index(), + 0, + "durable_index must stay stuck for this test to be meaningful" + ); + let total = persisted_total.load(Ordering::Relaxed); + assert!( + total < 3 * N, + "persist_entries received {total} entries for {N} appends; a frontier-tracking \ + scan is ~{N}, a durable_index-relative scan would be ~{} (O(N^2))", + N * (N + 1) / 2 + ); +} + +/// Cold start: after a restart, `durable_index` starts at the disk length and +/// the persist frontier must start *past* it. The first write's +/// persist scan begins at `durable_index + 1` β€” an already-durable entry on +/// disk must never be handed back to `persist_entries`. +/// +/// Guards the frontier initialization (`= durable_index`, scans use `+ 1`). +#[tokio::test] +async fn test_cold_start_persist_frontier_starts_past_durable_index() { + let persist_calls: Arc>>> = Arc::new(Mutex::new(Vec::new())); + + let mut log_store = MockLogStore::new(); + log_store.expect_last_index().returning(|| 5); // disk already holds 1..=5 + log_store.expect_get_entries().returning(|range| { + Ok(range + .map(|i| Entry { + index: i, + term: 1, + payload: None, + }) + .collect()) + }); + { + let calls = persist_calls.clone(); + log_store.expect_persist_entries().returning(move |entries| { + calls.lock().unwrap().push(entries.iter().map(|e| e.index).collect()); + Ok(()) + }); + } + log_store.expect_entry().returning(|_| Ok(None)); + log_store.expect_purge().returning(|_| Ok(())); + log_store.expect_load_purge_boundary().returning(|| Ok(None)); + log_store.expect_reset().returning(|| Ok(())); + log_store.expect_truncate().returning(|_| Ok(())); + log_store + .expect_replace_range() + .returning(|from, e| Ok(e.last().map(|x| x.index).unwrap_or(from.saturating_sub(1)))); + log_store.expect_is_write_durable().returning(|| false); + log_store.expect_flush().returning(|| Ok(())); + log_store.expect_flush_async().returning(|| Ok(())); + + let mut meta_store = MockMetaStore::new(); + meta_store.expect_save_hard_state().returning(|_| Ok(())); + meta_store.expect_load_hard_state().returning(|| Ok(None)); + meta_store.expect_flush().returning(|| Ok(())); + meta_store.expect_flush_async().returning(|| Ok(())); + + let storage = Arc::new(MockStorageEngine::from(log_store, meta_store)); + let raft_log = RaftLogCore::::new(1, storage, None, 5000); + std::thread::sleep(Duration::from_millis(10)); + + assert_eq!( + raft_log.durable_index(), + 5, + "restart: disk length 5 is treated as durable" + ); + + // First write after restart. Its persist scan must start at 6. + raft_log + .append_entries(vec![Entry { + index: 6, + term: 1, + payload: None, + }]) + .await + .unwrap(); + sleep(Duration::from_millis(50)).await; + + let calls = persist_calls.lock().unwrap().clone(); + assert!(!calls.is_empty(), "entry 6 must have been persisted"); + assert!( + calls.iter().flatten().all(|&idx| idx >= 6), + "cold start: the first persist must scan from durable_index+1 (6), never \ + re-scan already-durable entry 5. Got: {calls:?}" + ); +} diff --git a/d-engine-core/src/storage/buffered_raft_log_test/durable_index_test.rs b/d-engine-core/src/storage/raft_log_core_test/durable_index_test.rs similarity index 80% rename from d-engine-core/src/storage/buffered_raft_log_test/durable_index_test.rs rename to d-engine-core/src/storage/raft_log_core_test/durable_index_test.rs index 521338b3..5f78f1a3 100644 --- a/d-engine-core/src/storage/buffered_raft_log_test/durable_index_test.rs +++ b/d-engine-core/src/storage/raft_log_core_test/durable_index_test.rs @@ -5,22 +5,13 @@ use futures::future::join_all; use tokio::sync::mpsc; use crate::storage::raft_log::RaftLog; -use crate::test_utils::{BufferedRaftLogTestContext, MockStorageEngine, simulate_insert_command}; -use crate::{ - BufferedRaftLog, FlushPolicy, InternalEvent, MockTypeConfig, PersistenceConfig, - PersistenceStrategy, -}; +use crate::test_utils::{MockStorageEngine, RaftLogCoreTestContext, simulate_insert_command}; +use crate::{InternalEvent, MockTypeConfig, RaftLogCore}; use d_engine_proto::common::{Entry, LogId}; #[tokio::test] async fn test_durable_index_monotonic_under_concurrency() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_durable_index_monotonic", - ); + let mut ctx = RaftLogCoreTestContext::new("test_durable_index_monotonic"); let mut handles = vec![]; @@ -43,6 +34,7 @@ async fn test_durable_index_monotonic_under_concurrency() { // Wait for flush to complete tokio::time::sleep(Duration::from_millis(200)).await; + ctx.drain_fsync_completions(); // Verify monotonicity let durable = ctx.raft_log.durable_index(); @@ -54,13 +46,7 @@ async fn test_durable_index_monotonic_under_concurrency() { #[tokio::test] async fn test_durable_index_with_non_contiguous_entries() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_durable_index_non_contiguous", - ); + let ctx = RaftLogCoreTestContext::new("test_durable_index_non_contiguous"); // Create entries with non-contiguous indexes: 4, 7, 8, 10 let entries = vec![ @@ -123,24 +109,13 @@ async fn test_purge_does_not_regress_durable_index_already_ahead() { let storage = Arc::new(MockStorageEngine::with_id( "test_purge_does_not_regress_durable_index".into(), )); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - storage, - ); let (log_flush_tx, mut log_flush_rx) = mpsc::unbounded_channel::(); - let raft_log = raft_log.start(receiver, Some(log_flush_tx)); + let raft_log = RaftLogCore::::new(1, storage, Some(log_flush_tx), 5000); // Arrange: entries 1..=100, all flushed β€” durable_index reaches 100 and // fires LogFlushed(100). simulate_insert_command(&raft_log, (1..=100).collect(), 1).await; + crate::test_utils::drain_and_apply_fsync_completions(&raft_log, &mut log_flush_rx); assert_eq!(raft_log.durable_index(), 100); // Drain the LogFlushed(100) from the insert+flush above β€” not what this diff --git a/d-engine-core/src/storage/raft_log_core_test/durable_index_truncation_clamp_test.rs b/d-engine-core/src/storage/raft_log_core_test/durable_index_truncation_clamp_test.rs new file mode 100644 index 00000000..f4e7b69d --- /dev/null +++ b/d-engine-core/src/storage/raft_log_core_test/durable_index_truncation_clamp_test.rs @@ -0,0 +1,484 @@ +//! `durable_index` must never claim more of the log survived to disk than the +//! log actually holds right now. The danger case: a term-conflict truncation +//! shrinks the log while a persist / fsync for the old, longer log is still in +//! flight β€” the stale in-flight write must not push `durable_index` past the +//! truncation point. Two guards cover this: `try_advance_durable_index`'s term +//! check, and the `FsyncWorker` generation fence bumped by `remove_range`. + +use std::sync::Arc; +use std::sync::Mutex; +use std::time::Duration; + +use d_engine_proto::common::Entry; +use d_engine_proto::common::LogId; + +use crate::storage::raft_log::RaftLog; +use crate::test_utils::RaftLogCoreTestContext; +use crate::{MockLogStore, MockMetaStore, MockStorageEngine, MockTypeConfig, RaftLogCore}; + +fn entry( + index: u64, + term: u64, +) -> Entry { + Entry { + index, + term, + payload: None, + } +} + +/// After a drastic truncate-then-regrow, `durable_index` must land exactly on +/// the new tail β€” never above it (would claim durability for discarded +/// entries), never stuck below it (the new tail must actually become durable). +#[tokio::test] +async fn test_durable_index_lands_on_new_tail_after_truncate_and_resync() { + let mut ctx = + RaftLogCoreTestContext::new("durable_index_lands_on_new_tail_after_truncate_and_resync"); + + // Old leader (term 1) replicates 1..=10. append_entries inserts them into + // memory and notifies the IO thread; nothing is fsync-confirmed until the + // FsyncCompleted events are drained below. + ctx.append_entries(1, 10, 1).await; + assert_eq!(ctx.raft_log.last_entry_id(), 10); + assert_eq!( + ctx.raft_log.durable_index(), + 0, + "no fsync report drained yet" + ); + + // New leader (term 2): index 2 conflicts, so the log is truncated from 2 + // and replaced with a single new entry β€” real log becomes [1, 2]. Slow + // path: remove_range(2..) drops memory_max_index to 1 and clamps + // durable_index down, then the new index 2 is inserted. + ctx.raft_log + .filter_out_conflicts_and_append(1, 1, vec![entry(2, 2)]) + .await + .unwrap(); + assert_eq!(ctx.raft_log.last_entry_id(), 2, "log is now [1, 2]"); + + // flush() only returns after its own fsync-completion event is enqueued, + // so draining right here is deterministic β€” no sleep needed. + ctx.raft_log.flush().await.unwrap(); + ctx.drain_fsync_completions(); + + // The stale report for index 10 must be rejected (index 10 no longer + // exists); the report for index 2 must be accepted. + assert!( + ctx.raft_log.durable_index() <= ctx.raft_log.last_entry_id(), + "durable_index ({}) must not exceed last_entry_id ({})", + ctx.raft_log.durable_index(), + ctx.raft_log.last_entry_id() + ); + assert_eq!( + ctx.raft_log.durable_index(), + 2, + "durable_index must reach the true tail (2), not a stale pre-truncation watermark" + ); +} + +/// A persist whose entry set was captured *before* a truncation but finishes +/// *after* it must not let a stale fsync-completion report advance +/// `durable_index` into the range the truncation discarded. +/// +/// Timeline (deterministic via the persist gate): +/// 1. Old leader (term 1) replicates 1..=10 β€” on its own task, since the +/// inline persist path now blocks the calling task on the gate (unlike +/// the old dedicated-IO-thread design, where `append_entries` returned +/// immediately and the gated persist ran elsewhere). +/// 2. New leader (term 2): index 2 conflicts, on a second task, concurrent +/// with the still-gated persist from step 1. In-memory truncation +/// (`remove_range`) is synchronous, so it's visible immediately; the +/// `replace_range_and_submit` await behind it does not depend on step 1's +/// gate at all β€” the two tasks touch independent code paths, only +/// `entries`'s `RwLock` briefly serializes them. +/// 3. Release the gate: the stale persist (step 1) completes and reports a +/// stale fsync-completion for index 10; a flush() drives one legitimate +/// fsync-completion for index 2. +/// +/// Expected: draining the fsync completions advances `durable_index` to 2 and +/// rejects the stale report for index 10. +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn test_stale_persist_after_truncation_does_not_advance_durable_index() { + let (storage, persist_gate) = + MockStorageEngine::not_durable_gated_persist("stale_persist_after_truncation".into()); + let (log_flush_tx, mut log_flush_rx) = tokio::sync::mpsc::unbounded_channel(); + let raft_log = + RaftLogCore::::new(1, Arc::new(storage), Some(log_flush_tx), 5000); + + // Step 1: replicate 1..=10 on its own task β€” this task blocks on the + // gated persist_entries call until step 3 releases it. + let entries: Vec = (1..=10).map(|i| entry(i, 1)).collect(); + let persist = { + let raft_log = raft_log.clone(); + tokio::spawn(async move { raft_log.append_entries(entries).await }) + }; + tokio::time::sleep(Duration::from_millis(50)).await; + + // Step 2: term conflict at index 2 β€” on a second, independent task. Its + // in-memory truncation does not wait on step 1's gate at all. + let truncate = { + let raft_log = raft_log.clone(); + tokio::spawn(async move { + raft_log.filter_out_conflicts_and_append(1, 1, vec![entry(2, 2)]).await + }) + }; + tokio::time::sleep(Duration::from_millis(50)).await; + assert_eq!( + raft_log.last_entry_id(), + 2, + "in-memory truncation is synchronous β€” visible without waiting on step 1's gate" + ); + + // Step 3: release the stale persist, let both tasks finish, then flush. + persist_gate + .send(()) + .expect("step 1's task should still be waiting on the gate"); + persist.await.unwrap().unwrap(); + truncate.await.unwrap().unwrap(); + raft_log.flush().await.unwrap(); + tokio::time::sleep(Duration::from_millis(50)).await; + + // Drain fsync completions the way raft.rs's event loop would. + while let Ok(event) = log_flush_rx.try_recv() { + if let crate::InternalEvent::FsyncCompleted { mark, sent_at: _ } = event { + raft_log.try_advance_durable_index(mark); + } + } + + assert!( + raft_log.durable_index() <= raft_log.last_entry_id(), + "durable_index ({}) must never exceed last_entry_id ({}) β€” the stale persist \ + for 1..=10 must not be reported durable after truncation shrank the log to [1, 2]", + raft_log.durable_index(), + raft_log.last_entry_id() + ); + assert_eq!( + raft_log.durable_index(), + 2, + "durable_index must land on the post-truncation tail (2), not the stale 10" + ); +} + +/// `persist_pending_range` must report the highest index it *actually wrote*, +/// not the upper scan bound it was handed. The two differ during a truncation +/// race: the IO thread latched `memory_max_index` = 10 (an old leader had sent +/// 8, 9, 10), then a term-conflict truncation removed everything above 7 before +/// the SkipMap scan ran. Asking to persist `(4, 10]` then writes only 5, 6, 7. +/// +/// Returning the bound (10) would push the caller's `persisted_index` and the +/// fsync target past entries that never reached disk β€” a redundant fdatasync +/// plus a spurious `FsyncCompleted{10}` that the term check then has to reject. +/// Reporting 7 keeps every downstream watermark on real data. +#[tokio::test] +async fn test_persist_pending_range_reports_written_max_not_scan_bound() { + let storage = Arc::new(MockStorageEngine::with_id( + "persist_pending_range_reports_written_max".into(), + )); + let raft_log = RaftLogCore::::new(1, storage, None, 5000); + + raft_log.append_entries((1..=7).map(|i| entry(i, 1)).collect()).await.unwrap(); + + // Scan bound is 10 (stale latch); the SkipMap holds only 1..=7. + let written = raft_log.persist_pending_range(5, 10, "test").await.unwrap(); + + assert_eq!( + written, + Some(LogId { term: 1, index: 7 }), + "must report the highest index actually written (7), not the scan bound (10)" + ); +} + +/// After a term-conflict truncation, the persist frontier (`persisted_index`) must land +/// *past* the new tail β€” not on it. `replace_range_and_submit` already wrote the new +/// tail via `replace_range` and submitted its fsync; the next write's persist scan must +/// start at `new_tail + 1`. If the frontier is left *at* `new_tail`, every +/// subsequent write re-scans and re-`persist_entries` that one boundary entry +/// (and re-submits a redundant fsync for it) β€” the exact waste #446 removes. +/// +/// Guards the "highest-persisted" watermark semantics: `ReplaceRange` sets the +/// watermark to `new_tail`, and scans start at `watermark + 1`. +#[tokio::test] +async fn test_persist_frontier_skips_new_tail_after_truncation() { + // Records the index list of every persist_entries() call. + let persist_calls: Arc>>> = Arc::new(Mutex::new(Vec::new())); + + let mut log_store = MockLogStore::new(); + log_store.expect_last_index().returning(|| 0); + { + let calls = persist_calls.clone(); + log_store.expect_persist_entries().returning(move |entries| { + calls.lock().unwrap().push(entries.iter().map(|e| e.index).collect()); + Ok(()) + }); + } + log_store.expect_replace_range().returning(|from, new_entries| { + Ok(new_entries.last().map(|e| e.index).unwrap_or(from.saturating_sub(1))) + }); + log_store.expect_truncate().returning(|_| Ok(())); + log_store.expect_entry().returning(|_| Ok(None)); + log_store.expect_get_entries().returning(|_| Ok(vec![])); + log_store.expect_purge().returning(|_| Ok(())); + log_store.expect_load_purge_boundary().returning(|| Ok(None)); + log_store.expect_reset().returning(|| Ok(())); + log_store.expect_is_write_durable().returning(|| false); + log_store.expect_flush().returning(|| Ok(())); + log_store.expect_flush_async().returning(|| Ok(())); + + let mut meta_store = MockMetaStore::new(); + meta_store.expect_save_hard_state().returning(|_| Ok(())); + meta_store.expect_load_hard_state().returning(|| Ok(None)); + meta_store.expect_flush().returning(|| Ok(())); + meta_store.expect_flush_async().returning(|| Ok(())); + + let storage = Arc::new(MockStorageEngine::from(log_store, meta_store)); + let raft_log = RaftLogCore::::new(1, storage, None, 5000); + + // Old leader (term 1): entries 1..=10, persisted. + raft_log.append_entries((1..=10).map(|i| entry(i, 1)).collect()).await.unwrap(); + raft_log.flush().await.unwrap(); + + // New leader (term 2): conflict at index 6 β†’ truncate [6..], replace with + // [6, 7] (term 2). `filter_out_conflicts_and_append` awaits + // `replace_range_and_submit`, so the frontier is at new-tail 7 on return. + raft_log + .filter_out_conflicts_and_append(5, 1, vec![entry(6, 2), entry(7, 2)]) + .await + .unwrap(); + + // Only care about persist calls from here on β€” no flush() in between, so the + // next append's persist is the first thing to touch the frontier. + persist_calls.lock().unwrap().clear(); + + // Next write extends the log. Its persist scan must start at 8, not 7. + raft_log.append_entries((8..=10).map(|i| entry(i, 2)).collect()).await.unwrap(); + + // Wait until the new entries were actually handed to persist_entries; an empty + // record would make the "no re-persist" check below pass without observing anything. + let deadline = tokio::time::Instant::now() + Duration::from_secs(5); + while !persist_calls.lock().unwrap().iter().flatten().any(|&idx| idx == 10) { + assert!( + tokio::time::Instant::now() < deadline, + "entries 8..=10 were never passed to persist_entries. Got calls: {:?}", + persist_calls.lock().unwrap() + ); + tokio::time::sleep(Duration::from_millis(10)).await; + } + + let calls = persist_calls.lock().unwrap().clone(); + let re_persisted_tail = calls.iter().flatten().any(|&idx| idx <= 7); + assert!( + !re_persisted_tail, + "after ReplaceRange set the frontier at new-tail 7, the next persist must \ + start at 8 β€” entry 7 (or below) must not be handed to persist_entries again. \ + Got calls: {calls:?}" + ); +} + +/// Regression test for a bug where a *shrinking* truncation left +/// `persisted_index` stuck above the new (shorter) tail. `replace_range_and_submit` +/// used to update the frontier with `fetch_max(new_tail)`, which β€” unlike a +/// plain store β€” can never pull a stale higher value back down. Every append +/// after the truncation then computed `start = persisted_index + 1 > end = +/// memory_max_index`, so `persist_pending_range` silently no-op'd and the new +/// entry was never handed to `persist_entries` at all. Worse: `flush()` would +/// still see `persisted_index >= target` and report the entry durable off pure +/// (stale) bookkeeping β€” `durable_index` advancing past data that was never +/// actually written to disk. The fix is a plain `store(new_tail)`, matching +/// `reset_internal`'s treatment of the same "authoritative reset, not an +/// advance" situation. +#[tokio::test] +async fn test_persisted_index_corrected_downward_after_shrinking_truncation() { + let persist_calls: Arc>>> = Arc::new(Mutex::new(Vec::new())); + + let mut log_store = MockLogStore::new(); + log_store.expect_last_index().returning(|| 0); + { + let calls = persist_calls.clone(); + log_store.expect_persist_entries().returning(move |entries| { + calls.lock().unwrap().push(entries.iter().map(|e| e.index).collect()); + Ok(()) + }); + } + log_store.expect_replace_range().returning(|from, new_entries| { + Ok(new_entries.last().map(|e| e.index).unwrap_or(from.saturating_sub(1))) + }); + log_store.expect_truncate().returning(|_| Ok(())); + log_store.expect_entry().returning(|_| Ok(None)); + log_store.expect_get_entries().returning(|_| Ok(vec![])); + log_store.expect_purge().returning(|_| Ok(())); + log_store.expect_load_purge_boundary().returning(|| Ok(None)); + log_store.expect_reset().returning(|| Ok(())); + log_store.expect_is_write_durable().returning(|| false); + log_store.expect_flush().returning(|| Ok(())); + log_store.expect_flush_async().returning(|| Ok(())); + + let mut meta_store = MockMetaStore::new(); + meta_store.expect_save_hard_state().returning(|_| Ok(())); + meta_store.expect_load_hard_state().returning(|| Ok(None)); + meta_store.expect_flush().returning(|| Ok(())); + meta_store.expect_flush_async().returning(|| Ok(())); + + let storage = Arc::new(MockStorageEngine::from(log_store, meta_store)); + let raft_log = RaftLogCore::::new(1, storage, None, 5000); + + // Old leader (term 1): 1..=10, all persisted β€” persisted_index reaches 10. + raft_log.append_entries((1..=10).map(|i| entry(i, 1)).collect()).await.unwrap(); + raft_log.flush().await.unwrap(); + + // New leader (term 2): conflict at index 3, replaced with a SHORT tail + // ([3] only, term 2) β€” new_tail (3) lands far below the old persisted_index + // (10). This is realistic, not contrived: a new leader's AppendEntries + // batch is often much shorter than a stale follower's uncommitted tail. + raft_log.filter_out_conflicts_and_append(2, 1, vec![entry(3, 2)]).await.unwrap(); + assert_eq!(raft_log.last_entry_id(), 3, "log is now [1, 2, 3]"); + + persist_calls.lock().unwrap().clear(); + + // Append one more entry past the new (shorter) tail. + raft_log.append_entries(vec![entry(4, 2)]).await.unwrap(); + + let calls = persist_calls.lock().unwrap().clone(); + assert!( + calls.iter().flatten().any(|&idx| idx == 4), + "entry 4 must have been persisted β€” persisted_index must have been \ + corrected down to the new (shorter) tail (3) after the truncation, not \ + left stuck above it at the pre-truncation value (10). Got persist \ + calls: {calls:?}" + ); +} + +/// `persist_pending_range` must refuse to touch storage once the log is +/// poisoned β€” this is its *own* guard, not something callers must each +/// remember to check first. `flush()` is the one `RaftLog` method that calls +/// `persist_pending_range` without checking `is_poisoned()` itself, so this +/// guard being missing would let `flush()` write to an already-untrusted disk. +/// +/// The setup relies on `append_entries`'s own ordering: `insert_to_memory` +/// runs *before* the persist attempt, so a failed persist leaves +/// `memory_max_index` ahead of `persisted_index` β€” exactly the condition +/// `flush()` needs to decide there's something to persist and reach +/// `persist_pending_range` at all. +#[tokio::test] +async fn test_flush_after_poisoned_does_not_call_persist_entries() { + let persist_calls: Arc>>> = Arc::new(Mutex::new(Vec::new())); + let persist_should_fail = Arc::new(std::sync::atomic::AtomicBool::new(false)); + + let mut log_store = MockLogStore::new(); + log_store.expect_last_index().returning(|| 0); + { + let calls = persist_calls.clone(); + let should_fail = persist_should_fail.clone(); + log_store.expect_persist_entries().returning(move |entries| { + calls.lock().unwrap().push(entries.iter().map(|e| e.index).collect()); + if should_fail.load(std::sync::atomic::Ordering::Acquire) { + Err(crate::Error::Fatal("simulated disk failure".into())) + } else { + Ok(()) + } + }); + } + log_store.expect_replace_range().returning(|_, _| Ok(0)); + log_store.expect_truncate().returning(|_| Ok(())); + log_store.expect_entry().returning(|_| Ok(None)); + log_store.expect_get_entries().returning(|_| Ok(vec![])); + log_store.expect_purge().returning(|_| Ok(())); + log_store.expect_load_purge_boundary().returning(|| Ok(None)); + log_store.expect_reset().returning(|| Ok(())); + log_store.expect_is_write_durable().returning(|| false); + log_store.expect_flush().returning(|| Ok(())); + log_store.expect_flush_async().returning(|| Ok(())); + + let mut meta_store = MockMetaStore::new(); + meta_store.expect_save_hard_state().returning(|_| Ok(())); + meta_store.expect_load_hard_state().returning(|| Ok(None)); + meta_store.expect_flush().returning(|| Ok(())); + meta_store.expect_flush_async().returning(|| Ok(())); + + let storage = Arc::new(MockStorageEngine::from(log_store, meta_store)); + let raft_log = RaftLogCore::::new(1, storage, None, 5000); + + // First append succeeds β€” persisted_index catches up to memory_max_index (1). + raft_log.append_entries(vec![entry(1, 1)]).await.unwrap(); + + // Now make persist_entries fail. append_entries calls insert_to_memory + // (memory_max_index -> 2) *before* the persist attempt, so the failure + // leaves persisted_index at 1 β€” one behind memory_max_index β€” and poisons. + persist_should_fail.store(true, std::sync::atomic::Ordering::Release); + let append_result = raft_log.append_entries(vec![entry(2, 1)]).await; + assert!(append_result.is_err(), "failed persist must propagate"); + assert!( + raft_log.is_poisoned(), + "failed persist_entries must poison the log" + ); + assert_eq!( + raft_log.last_entry_id(), + 2, + "entry 2 is in memory even though its persist failed" + ); + + persist_calls.lock().unwrap().clear(); + + // flush() must refuse once poisoned. persisted_index (1) < memory_max_index + // (2) here, so without persist_pending_range's own is_poisoned() guard, + // flush() would call persist_entries again on the now-untrusted disk. + let flush_result = raft_log.flush().await; + assert!( + flush_result.is_err(), + "flush() must refuse to proceed once the log is poisoned" + ); + assert!( + persist_calls.lock().unwrap().is_empty(), + "persist_entries must never be called again once the log is poisoned" + ); +} + +/// `flush()` must surface the error of its own catch-up persist, not swallow it. +/// +/// Entries that sit in memory but were never persisted (`persisted_index < +/// memory_max_index`) make `flush()` persist them itself. If that persist fails, the +/// caller must get the underlying disk error directly. Before the explicit `?`, +/// the error was dropped and `flush()` only failed later, with a generic +/// "storage is poisoned" reply from the fsync worker β€” correct outcome, but it relied +/// on that indirect chain and hid the root cause. +#[tokio::test] +async fn test_flush_returns_catch_up_persist_error_instead_of_swallowing_it() { + let mut log_store = MockLogStore::new(); + log_store.expect_last_index().returning(|| 0); + log_store + .expect_persist_entries() + .returning(|_| Err(crate::Error::Fatal("simulated disk failure".into()))); + log_store.expect_replace_range().returning(|_, _| Ok(0)); + log_store.expect_truncate().returning(|_| Ok(())); + log_store.expect_entry().returning(|_| Ok(None)); + log_store.expect_get_entries().returning(|_| Ok(vec![])); + log_store.expect_purge().returning(|_| Ok(())); + log_store.expect_load_purge_boundary().returning(|| Ok(None)); + log_store.expect_reset().returning(|| Ok(())); + log_store.expect_is_write_durable().returning(|| false); + log_store.expect_flush().returning(|| Ok(())); + log_store.expect_flush_async().returning(|| Ok(())); + + let mut meta_store = MockMetaStore::new(); + meta_store.expect_save_hard_state().returning(|_| Ok(())); + meta_store.expect_load_hard_state().returning(|| Ok(None)); + meta_store.expect_flush().returning(|| Ok(())); + meta_store.expect_flush_async().returning(|| Ok(())); + + let storage = Arc::new(MockStorageEngine::from(log_store, meta_store)); + let raft_log = RaftLogCore::::new(1, storage, None, 5000); + + // In memory only: persisted_index (0) < memory_max_index (1), log not poisoned yet. + raft_log.insert_to_memory(&[entry(1, 1)]); + assert!(!raft_log.is_poisoned()); + + let err = raft_log.flush().await.expect_err("catch-up persist failed, flush must fail"); + + assert!( + format!("{err:?}").contains("simulated disk failure"), + "flush() must return the catch-up persist error itself, got: {err:?}" + ); + assert!( + raft_log.is_poisoned(), + "a failed persist must poison the log" + ); +} diff --git a/d-engine-core/src/storage/buffered_raft_log_test/edge_cases_test.rs b/d-engine-core/src/storage/raft_log_core_test/edge_cases_test.rs similarity index 77% rename from d-engine-core/src/storage/buffered_raft_log_test/edge_cases_test.rs rename to d-engine-core/src/storage/raft_log_core_test/edge_cases_test.rs index 60f00876..1243436e 100644 --- a/d-engine-core/src/storage/buffered_raft_log_test/edge_cases_test.rs +++ b/d-engine-core/src/storage/raft_log_core_test/edge_cases_test.rs @@ -1,19 +1,12 @@ use bytes::Bytes; use crate::storage::raft_log::RaftLog; -use crate::test_utils::BufferedRaftLogTestContext; -use crate::{FlushPolicy, PersistenceStrategy}; +use crate::test_utils::RaftLogCoreTestContext; use d_engine_proto::common::{Entry, EntryPayload, LogId}; #[tokio::test] async fn test_empty_log_operations() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_empty_log", - ); + let ctx = RaftLogCoreTestContext::new("test_empty_log"); assert!(ctx.raft_log.is_empty()); assert_eq!(ctx.raft_log.first_entry_id(), 0); @@ -26,13 +19,7 @@ async fn test_empty_log_operations() { #[tokio::test] async fn test_single_entry_operations() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_single_entry", - ); + let ctx = RaftLogCoreTestContext::new("test_single_entry"); // Append single entry let entry = Entry { @@ -55,13 +42,7 @@ async fn test_single_entry_operations() { #[tokio::test] async fn test_gap_handling_in_indexes() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_gap_handling", - ); + let ctx = RaftLogCoreTestContext::new("test_gap_handling"); // Create entries 1..=100 for i in 1..=100 { @@ -98,13 +79,7 @@ async fn test_gap_handling_in_indexes() { #[tokio::test] async fn test_extreme_boundary_conditions() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_extreme_boundary_conditions", - ); + let ctx = RaftLogCoreTestContext::new("test_extreme_boundary_conditions"); // Test with maximum index values let max_index = u64::MAX - 10; diff --git a/d-engine-core/src/storage/buffered_raft_log_test/flush_strategy_test.rs b/d-engine-core/src/storage/raft_log_core_test/flush_strategy_test.rs similarity index 68% rename from d-engine-core/src/storage/buffered_raft_log_test/flush_strategy_test.rs rename to d-engine-core/src/storage/raft_log_core_test/flush_strategy_test.rs index 44fd0040..754da743 100644 --- a/d-engine-core/src/storage/buffered_raft_log_test/flush_strategy_test.rs +++ b/d-engine-core/src/storage/raft_log_core_test/flush_strategy_test.rs @@ -12,8 +12,8 @@ use bytes::Bytes; use d_engine_proto::common::{Entry, EntryPayload}; use tokio::time::{Duration, sleep}; -use crate::test_utils::BufferedRaftLogTestContext; -use crate::{FlushPolicy, PersistenceStrategy, RaftLog}; +use crate::RaftLog; +use crate::test_utils::RaftLogCoreTestContext; /// Test MemFirst with threshold=1 persists entries after flush /// @@ -22,17 +22,12 @@ use crate::{FlushPolicy, PersistenceStrategy, RaftLog}; /// - Expected: durable_index == 5 after explicit flush() #[tokio::test] async fn test_mem_first_entries_durable_after_flush() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_mem_first_persists_immediately", - ); + let mut ctx = RaftLogCoreTestContext::new("test_mem_first_persists_immediately"); // Act: Append entries then wait for durability ctx.append_entries(1, 5, 1).await; ctx.raft_log.flush().await.unwrap(); + ctx.drain_fsync_completions(); // Assert: All entries durable after flush assert_eq!( @@ -51,13 +46,7 @@ async fn test_mem_first_entries_durable_after_flush() { /// - Expected: All 1000 entries durable after flush(), no data loss #[tokio::test] async fn test_mem_first_concurrent_writes_durable_after_flush() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_mem_first_concurrent_writes", - ); + let mut ctx = RaftLogCoreTestContext::new("test_mem_first_concurrent_writes"); // Act: Concurrent appends let mut handles = vec![]; @@ -83,6 +72,7 @@ async fn test_mem_first_concurrent_writes_durable_after_flush() { // Wait for all entries to become durable ctx.raft_log.flush().await.unwrap(); + ctx.drain_fsync_completions(); // Assert: All entries durable after flush assert_eq!( @@ -101,13 +91,7 @@ async fn test_mem_first_concurrent_writes_durable_after_flush() { /// - Expected: All 5 flushed entries recovered from MockStorage #[tokio::test] async fn test_mem_first_crash_recovery_restores_flushed_entries() { - let original_ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_mem_first_crash_recovery", - ); + let original_ctx = RaftLogCoreTestContext::new("test_mem_first_crash_recovery"); // Arrange: Append and flush for i in 1..=5 { @@ -148,13 +132,7 @@ async fn test_mem_first_crash_recovery_restores_flushed_entries() { /// - Expected: Entries in buffer but not yet durable #[tokio::test] async fn test_mem_first_buffers_entries_before_flush() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1000, - }, - "test_mem_first_buffers_entries", - ); + let ctx = RaftLogCoreTestContext::new("test_mem_first_buffers_entries"); // Act: Append 5 entries (below threshold) ctx.append_entries(1, 5, 1).await; @@ -172,13 +150,7 @@ async fn test_mem_first_buffers_entries_before_flush() { /// - Expected: Entries become durable after flush #[tokio::test] async fn test_mem_first_flushes_asynchronously() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_mem_first_async_flush", - ); + let mut ctx = RaftLogCoreTestContext::new("test_mem_first_async_flush"); // Arrange: Append entries ctx.append_entries(1, 5, 1).await; @@ -186,6 +158,7 @@ async fn test_mem_first_flushes_asynchronously() { // Act: Explicit flush ctx.raft_log.flush().await.unwrap(); sleep(Duration::from_millis(100)).await; // Allow async flush + ctx.drain_fsync_completions(); // Assert: Entries now durable assert!( @@ -201,13 +174,7 @@ async fn test_mem_first_flushes_asynchronously() { /// - Expected: All entries buffered correctly #[tokio::test] async fn test_mem_first_concurrent_buffering() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 5000, - }, - "test_mem_first_concurrent", - ); + let ctx = RaftLogCoreTestContext::new("test_mem_first_concurrent"); // Act: Concurrent appends let mut handles = vec![]; @@ -235,58 +202,6 @@ async fn test_mem_first_concurrent_buffering() { assert_eq!(ctx.raft_log.len(), 100, "All entries should be buffered"); } -/// Test Batched flushes at threshold -/// -/// # Scenario -/// - Set threshold=5, append 5 entries -/// - Expected: Flush triggered at threshold -#[tokio::test] -async fn test_batched_flushes_at_threshold() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 10000, // High interval to test threshold trigger - }, - "test_batched_threshold", - ); - - // Act: Append exactly threshold entries - ctx.append_entries(1, 5, 1).await; - sleep(Duration::from_millis(100)).await; // Allow flush - - // Assert: Entries should be flushed - assert!( - ctx.raft_log.durable_index() >= 5, - "Entries should be flushed at threshold" - ); -} - -/// Test Batched flushes at interval -/// -/// # Scenario -/// - Set interval=50ms, append 2 entries, wait -/// - Expected: Flush triggered by timer -#[tokio::test] -async fn test_batched_flushes_at_interval() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 50, - }, - "test_batched_interval", - ); - - // Act: Append few entries and wait for interval - ctx.append_entries(1, 2, 1).await; - sleep(Duration::from_millis(200)).await; // Wait for interval flush - - // Assert: Entries flushed by timer - assert!( - ctx.raft_log.durable_index() > 0, - "Entries should be flushed by interval" - ); -} - /// Test Batched partial flush after crash /// /// # Scenario @@ -294,13 +209,7 @@ async fn test_batched_flushes_at_interval() { /// - Expected: Only flushed entries recovered #[tokio::test] async fn test_batched_partial_flush_recovery() { - let original_ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_batched_partial_flush", - ); + let original_ctx = RaftLogCoreTestContext::new("test_batched_partial_flush"); // Arrange: Append 5 entries (triggers flush) original_ctx.append_entries(1, 5, 1).await; diff --git a/d-engine-core/src/storage/buffered_raft_log_test/id_allocation_test.rs b/d-engine-core/src/storage/raft_log_core_test/id_allocation_test.rs similarity index 91% rename from d-engine-core/src/storage/buffered_raft_log_test/id_allocation_test.rs rename to d-engine-core/src/storage/raft_log_core_test/id_allocation_test.rs index 095db56d..4ec756ff 100644 --- a/d-engine-core/src/storage/buffered_raft_log_test/id_allocation_test.rs +++ b/d-engine-core/src/storage/raft_log_core_test/id_allocation_test.rs @@ -10,27 +10,11 @@ use std::ops::RangeInclusive; use std::sync::Arc; use std::sync::atomic::Ordering; -use crate::{ - BufferedRaftLog, FlushPolicy, MockStorageEngine, MockTypeConfig, PersistenceConfig, - PersistenceStrategy, RaftLog, -}; +use crate::{MockStorageEngine, MockTypeConfig, RaftLog, RaftLogCore}; -fn setup_memory() -> Arc> { +fn setup_memory() -> Arc> { let storage = Arc::new(MockStorageEngine::new()); - let (raft_log, _receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - storage, - ); - - Arc::new(raft_log) + RaftLogCore::::new(1, storage, None, 5000) } fn is_empty_range(range: &RangeInclusive) -> bool { diff --git a/d-engine-core/src/storage/raft_log_core_test/mod.rs b/d-engine-core/src/storage/raft_log_core_test/mod.rs new file mode 100644 index 00000000..a6f05a6b --- /dev/null +++ b/d-engine-core/src/storage/raft_log_core_test/mod.rs @@ -0,0 +1,28 @@ +//! RaftLogCore unit tests +//! +//! Ported from `buffered_raft_log_test` β€” same functional coverage, adapted +//! for `RaftLogCore`'s inline (no dedicated IO thread) execution model. +//! +//! These tests use `MockStorageEngine` to verify algorithm correctness without real disk I/O. +//! Integration tests with `FileStorageEngine` are in `d-engine-server/tests/integration/`. + +mod basic_operations_test; +mod concurrent_fsync_test; +mod concurrent_operations_test; +mod content_validated_watermark_test; +mod durable_index_test; +mod durable_index_truncation_clamp_test; +mod edge_cases_test; +mod flush_strategy_test; +mod id_allocation_test; +mod performance_test; +mod pipeline_overlap_test; +mod prev_log_index_zero_idempotency_test; +mod quorum_durability_test; +mod raft_properties_test; +mod remove_range_test; +mod replace_range_fsync_test; +mod shutdown_test; +mod term_index_test; +mod term_segments_test; +mod truncation_fsync_fence_test; diff --git a/d-engine-core/src/storage/raft_log_core_test/performance_test.rs b/d-engine-core/src/storage/raft_log_core_test/performance_test.rs new file mode 100644 index 00000000..c11ea47d --- /dev/null +++ b/d-engine-core/src/storage/raft_log_core_test/performance_test.rs @@ -0,0 +1,171 @@ +//! Performance tests for RaftLogCore with controllable delays +//! +//! These tests verify RaftLogCore performance behavior during concurrent +//! operations like flush, using MockStorageEngine with controllable delays. + +use std::sync::Arc; +use std::time::Duration; + +use bytes::Bytes; +use tokio::sync::Barrier; +use tokio::time::Instant; + +use crate::{MockLogStore, MockMetaStore, MockStorageEngine, MockTypeConfig, RaftLog, RaftLogCore}; +use d_engine_proto::common::{Entry, EntryPayload}; + +// Test helper: Creates storage with controllable delay +fn create_delayed_storage(delay_ms: u64) -> Arc { + let mut log_store = MockLogStore::new(); + log_store.expect_last_index().returning(|| 0); + log_store.expect_load_purge_boundary().returning(|| Ok(None)); + log_store.expect_truncate().returning(|_| Ok(())); + log_store.expect_reset().returning(|| Ok(())); + log_store.expect_is_write_durable().returning(|| true); + log_store.expect_flush().returning(|| Ok(())); + + // Add controllable delay to persist_entries + log_store.expect_persist_entries().returning(move |_| { + let delay = Duration::from_millis(delay_ms); + std::thread::sleep(delay); + Ok(()) + }); + + Arc::new(MockStorageEngine::from(log_store, MockMetaStore::new())) +} + +// Tests reset performance during active flush +#[tokio::test] +async fn test_reset_performance_during_active_flush() { + // persist_entries mock sleeps for FLUSH_DELAY_MS. + // reset() fences the in-flight fsync (generation bump) rather than waiting + // on it. The test verifies reset completes within a bounded time (3x the + // flush delay) and does not block indefinitely. + const FLUSH_DELAY_MS: u64 = 200; + let max_reset_duration_ms = FLUSH_DELAY_MS * 3; // 600ms: accounts for scheduling overhead + + let storage = create_delayed_storage(FLUSH_DELAY_MS); + let log = RaftLogCore::::new(1, storage, None, 5000); + let barrier = Arc::new(Barrier::new(2)); + + // Start long-running append+flush in background (slow due to persist_entries delay) + let flush_log = log.clone(); + let flush_barrier = barrier.clone(); + tokio::spawn(async move { + flush_barrier.wait().await; // Sync point + let entries: Vec = (1..=10) + .map(|i| Entry { + index: i, + term: 1, + payload: None, + }) + .collect(); + let _ = flush_log.append_entries(entries).await; + let _ = flush_log.flush().await; + }); + + // Wait for flush to start + barrier.wait().await; + + // Measure reset performance during active flush + let start = Instant::now(); + log.reset().await.unwrap(); + let duration = start.elapsed(); + + assert!( + duration.as_millis() < max_reset_duration_ms as u128, + "Reset took {}ms during active flush", + duration.as_millis(), + ); +} + +// Tests filter_out_conflicts performance with active flush +#[tokio::test] +async fn test_filter_conflicts_performance_during_flush() { + let is_ci = std::env::var("CI").is_ok(); + // Relax time limit in CI environment + let max_duration_ms = if is_ci { 500 } else { 50 }; + + const FLUSH_DELAY_MS: u64 = 300; + + let storage = create_delayed_storage(FLUSH_DELAY_MS); + let log = RaftLogCore::::new(1, storage, None, 5000); + let barrier = Arc::new(Barrier::new(2)); + + // Populate with test data + let mut entries = vec![]; + for i in 1..=1000 { + entries.push(Entry { + index: i, + term: 1, + payload: Some(EntryPayload::command(Bytes::from(vec![0; 256]))), + }); + } + log.append_entries(entries).await.unwrap(); + + // Start long flush in background (slow due to persist_entries delay) + let flush_log = log.clone(); + let flush_barrier = barrier.clone(); + tokio::spawn(async move { + flush_barrier.wait().await; + let _ = flush_log.flush().await; + }); + + // Wait for flush to start + barrier.wait().await; + + // Measure performance during active flush + let start = Instant::now(); + log.filter_out_conflicts_and_append( + 500, + 1, + vec![Entry { + index: 501, + term: 1, + payload: Some(EntryPayload::command(Bytes::from(vec![1; 256]))), + }], + ) + .await + .unwrap(); + + let duration = start.elapsed(); + assert!( + duration.as_millis() < max_duration_ms as u128, + "Operation took {}ms during flush", + duration.as_millis(), + ); +} + +// Tests fresh cluster performance consistency +#[tokio::test] +async fn test_fresh_cluster_performance_consistency() { + let is_ci = std::env::var("CI").is_ok(); + // Relax time limit in CI environment + let max_duration_ms = if is_ci { 50 } else { 5 }; + + let mut log_store = MockLogStore::new(); + log_store.expect_is_write_durable().returning(|| true); + log_store.expect_flush().return_once(|| Ok(())); + log_store.expect_last_index().returning(|| 0); + log_store.expect_load_purge_boundary().returning(|| Ok(None)); + log_store.expect_truncate().returning(|_| Ok(())); + log_store.expect_persist_entries().returning(|_| Ok(())); + log_store.expect_reset().returning(|| Ok(())); + + let log = RaftLogCore::::new( + 1, + Arc::new(MockStorageEngine::from(log_store, MockMetaStore::new())), + None, + 5000, + ); + + // Measure reset performance in fresh cluster + let start = Instant::now(); + log.reset().await.unwrap(); + let duration = start.elapsed(); + + assert!( + duration.as_millis() < max_duration_ms as u128, + "Fresh cluster reset took {}ms", + duration.as_millis(), + ); +} diff --git a/d-engine-core/src/storage/buffered_raft_log_test/pipeline_overlap_test.rs b/d-engine-core/src/storage/raft_log_core_test/pipeline_overlap_test.rs similarity index 90% rename from d-engine-core/src/storage/buffered_raft_log_test/pipeline_overlap_test.rs rename to d-engine-core/src/storage/raft_log_core_test/pipeline_overlap_test.rs index efb54b80..53808657 100644 --- a/d-engine-core/src/storage/buffered_raft_log_test/pipeline_overlap_test.rs +++ b/d-engine-core/src/storage/raft_log_core_test/pipeline_overlap_test.rs @@ -10,20 +10,11 @@ use std::sync::atomic::{AtomicU64, Ordering}; use d_engine_proto::common::Entry; -use crate::test_utils::BufferedRaftLogTestContext; -use crate::{ - BufferedRaftLog, FlushPolicy, MockLogStore, MockMetaStore, MockStorageEngine, MockTypeConfig, - PersistenceConfig, PersistenceStrategy, RaftLog, -}; - -fn ctx(name: &str) -> BufferedRaftLogTestContext { - BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 50, - }, - name, - ) +use crate::test_utils::RaftLogCoreTestContext; +use crate::{MockLogStore, MockMetaStore, MockStorageEngine, MockTypeConfig, RaftLog, RaftLogCore}; + +fn ctx(name: &str) -> RaftLogCoreTestContext { + RaftLogCoreTestContext::new(name) } fn entry( @@ -298,7 +289,7 @@ async fn test_filter_conflicts_bench_pattern_no_truncation() { /// must restore the post-conflict entries, not the original pre-conflict ones. /// /// # Why this matters -/// A real term conflict routes through IOTask::ReplaceRange in the IO thread +/// A real term conflict routes through `replace_range_and_submit` /// (truncate old entries from disk, persist new ones). If this path is not correctly /// flushed, crash recovery would replay the original WAL and restore the stale entries β€” /// violating the Raft log matching property (Β§5.3). @@ -322,7 +313,7 @@ async fn test_replace_range_conflict_persists_correctly_after_crash() { // Act: leader sends conflict β€” index=4 was term=1, leader has term=2 // filter_out_conflicts_and_append detects conflict at index=4, - // sends IOTask::ReplaceRange(truncate_from=4, new_entries=[4(t2), 5(t2)]) + // calls replace_range_and_submit(truncate_from=4, new_entries=[4(t2), 5(t2)]) let result = ctx .raft_log .filter_out_conflicts_and_append(2, 1, vec![entry(3, 1), entry(4, 2), entry(5, 2)]) @@ -330,7 +321,7 @@ async fn test_replace_range_conflict_persists_correctly_after_crash() { .unwrap(); assert_eq!(result.unwrap().index, 5); - // Flush ensures IOTask::ReplaceRange is fully processed and persisted by IO thread + // Flush makes the replaced tail durable before the simulated crash ctx.raft_log.flush().await.unwrap(); // Verify memory state before crash @@ -386,7 +377,7 @@ async fn test_replace_range_removes_entries_beyond_new_tail() { assert_eq!(ctx.raft_log.last_entry_id(), 5); // Act: leader conflict at index=3 (term mismatch); new suffix is only [3(t2), 4(t2)] - // IOTask::ReplaceRange(truncate_from=3, new_entries=[3(t2), 4(t2)]) β€” entry 5 must vanish + // replace_range_and_submit(truncate_from=3, new_entries=[3(t2), 4(t2)]) β€” entry 5 must vanish ctx.raft_log .filter_out_conflicts_and_append(2, 1, vec![entry(3, 2), entry(4, 2)]) .await @@ -418,7 +409,7 @@ async fn test_replace_range_removes_entries_beyond_new_tail() { assert_eq!(recovered.raft_log.entry(3).unwrap().unwrap().term, 2); } -/// IOTask::ReplaceRange must call LogStore::replace_range() as a single operation, +/// `replace_range_and_submit` must call LogStore::replace_range() as a single operation, /// not truncate() + persist_entries() separately. /// /// # Why @@ -426,11 +417,10 @@ async fn test_replace_range_removes_entries_beyond_new_tail() { /// (e.g. single RocksDB WriteBatch). Calling truncate + persist_entries separately /// breaks this contract regardless of what the backend implements. /// -/// # Red/Green -/// This test FAILS before implementation (truncate IS called separately). -/// It PASSES after IOTask::ReplaceRange is updated to call replace_range(). +/// # Invariant +/// One conflict resolution issues exactly one replace_range() call and never truncate(). #[tokio::test] -async fn test_io_task_replace_range_delegates_to_replace_range_not_truncate() { +async fn test_replace_range_and_submit_delegates_to_replace_range_not_truncate() { let replace_range_count = Arc::new(AtomicU64::new(0)); let truncate_count = Arc::new(AtomicU64::new(0)); @@ -438,12 +428,12 @@ async fn test_io_task_replace_range_delegates_to_replace_range_not_truncate() { // replace_range() must be called exactly once for one conflict resolution let rr_counter = replace_range_count.clone(); - log_store.expect_replace_range().returning(move |_from, _entries| { + log_store.expect_replace_range().returning(move |_from, new_entries| { rr_counter.fetch_add(1, Ordering::Relaxed); - Ok(()) + Ok(new_entries.last().map(|e| e.index).unwrap_or(0)) }); - // truncate() must NOT be called β€” IOTask::ReplaceRange owns the full operation + // truncate() must NOT be called β€” replace_range_and_submit owns the full operation let tr_counter = truncate_count.clone(); log_store.expect_truncate().returning(move |_| { tr_counter.fetch_add(1, Ordering::Relaxed); @@ -469,20 +459,7 @@ async fn test_io_task_replace_range_delegates_to_replace_range_not_truncate() { meta_store.expect_flush_async().returning(|| Ok(())); let storage = Arc::new(MockStorageEngine::from(log_store, meta_store)); - let (raft_log, receiver) = BufferedRaftLog::::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 60_000, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - storage, - ); - let raft_log = raft_log.start(receiver, None); - std::thread::sleep(std::time::Duration::from_millis(10)); + let raft_log = RaftLogCore::::new(1, storage, None, 5000); // Populate in-memory SkipMap with [1,2,3,4] term=1 so conflict detection fires raft_log @@ -490,7 +467,7 @@ async fn test_io_task_replace_range_delegates_to_replace_range_not_truncate() { .await .unwrap(); - // Trigger IOTask::ReplaceRange: conflict at index=3 (term mismatch 1 vs 2) + // Trigger replace_range_and_submit: conflict at index=3 (term mismatch 1 vs 2) raft_log .filter_out_conflicts_and_append(2, 1, vec![entry(3, 2), entry(4, 2)]) .await diff --git a/d-engine-core/src/storage/raft_log_core_test/prev_log_index_zero_idempotency_test.rs b/d-engine-core/src/storage/raft_log_core_test/prev_log_index_zero_idempotency_test.rs new file mode 100644 index 00000000..eb98ce8c --- /dev/null +++ b/d-engine-core/src/storage/raft_log_core_test/prev_log_index_zero_idempotency_test.rs @@ -0,0 +1,239 @@ +//! Tests for `filter_out_conflicts_and_append` when `prev_log_index == 0`. +//! +//! Background: `prev_log_index == 0` is Raft's sentinel for "nothing before the start of the +//! log" β€” Rule 2 (term-at-prev-index check) is trivially satisfied because there is no real +//! entry 0 to compare. That does NOT license skipping Rules 3/4: the receiver must still +//! compare incoming entries against whatever it already has, starting at index 1, and only +//! touch the entries that actually conflict (differing term at the same index). A batch that +//! fully matches existing content must be a no-op β€” this is exactly what +//! `pipeline_overlap_test.rs` already proves for `prev_log_index > 0`. +//! +//! The current implementation special-cases `prev_log_index == 0` to unconditionally +//! `reset()` (wipe the whole log, `durable_index` included) before re-appending β€” regardless +//! of whether the incoming entries are a pure duplicate of what's already durably stored. A +//! leader that resends a `prev_log_index=0` probe (no backpressure, a retry, a reconnect) before +//! learning the follower already caught up will repeatedly destroy real, already-durable +//! progress. These tests are RED until `prev_log_index == 0` is folded into the same +//! overlap/conflict comparison used for `prev_log_index > 0`. + +use crate::storage::raft_log::RaftLog; +use crate::test_utils::RaftLogCoreTestContext; +use d_engine_proto::common::Entry; +use std::time::Duration; + +fn ctx(name: &str) -> RaftLogCoreTestContext { + RaftLogCoreTestContext::new(name) +} + +fn entry( + index: u64, + term: u64, +) -> Entry { + Entry { + index, + term, + payload: None, + } +} + +/// A genuinely fresh follower (no prior `append_entries` calls at all) receiving its very +/// first `prev_log_index=0` probe must accept and append normally. +/// +/// # Why this needs its own test, not just implicit coverage +/// The other tests in this file all pre-populate the log before calling +/// `filter_out_conflicts_and_append`, so none of them exercise the case the +/// `is_virtual_log_start` guard exists to protect: without it, `entry_term(0)` returns `None` +/// unconditionally (there is no real entry 0 to look up), so `entry_term(0) != Some(0)` would +/// be true and this β€” the single most basic, legitimate case β€” would be wrongly rejected as a +/// conflict. This test pins that guard directly. +/// +/// # Expected (holds both before and after the fix β€” this is a regression guard for the +/// `is_virtual_log_start` skip, not a RED/GREEN discriminator for the reset() removal) +#[tokio::test] +async fn test_filter_conflicts_zero_prev_on_genuinely_empty_log_appends_all() { + let ctx = ctx("zero_prev_genuinely_empty_log_appends_all"); + assert_eq!( + ctx.raft_log.last_entry_id(), + 0, + "precondition: log must be untouched" + ); + + // Act: the very first AppendEntries this follower ever receives. + let result = ctx + .raft_log + .filter_out_conflicts_and_append( + 0, + 0, + vec![ + entry(1, 1), + entry(2, 1), + entry(3, 1), + entry(4, 1), + entry(5, 1), + ], + ) + .await + .unwrap(); + + assert_eq!(result.unwrap().index, 5); + assert_eq!(ctx.raft_log.last_entry_id(), 5); + for i in 1u64..=5 { + assert_eq!(ctx.raft_log.entry(i).unwrap().unwrap().term, 1); + } +} + +/// A leader resending `prev_log_index=0` with entries the follower already has β€” durably β€” +/// must be a no-op. This is the exact T4/T5 scenario from the #446 investigation: the +/// follower's first response to a `prev_log_index=0` probe is withheld pending its own +/// `durable_index` catching up (RPO=0); if the leader resends the identical probe before that +/// withheld ACK is released, the follower must not throw away the progress it already made. +/// +/// # Expected (RED until fixed) +/// `durable_index()` and `last_entry_id()` stay at 5 β€” the duplicate probe changes nothing. +#[tokio::test] +async fn test_filter_conflicts_zero_prev_duplicate_resend_preserves_durable_index() { + let mut ctx = ctx("zero_prev_duplicate_preserves_durable_index"); + + // Arrange: follower already durably has [1..5], all term=1 β€” simulating a prior + // `prev_log_index=0` probe that succeeded and finished fsyncing. + for i in 1u64..=5 { + ctx.raft_log.append_entries(vec![entry(i, 1)]).await.unwrap(); + } + tokio::time::sleep(Duration::from_millis(50)).await; + ctx.drain_fsync_completions(); + assert_eq!( + ctx.raft_log.durable_index(), + 5, + "precondition: [1..5] must be durable" + ); + + // Act: leader resends the identical prev_log_index=0 probe β€” same entries, same terms. + // This is what happens when the leader's next_index[peer] never advanced (the leader + // hasn't processed a response yet), not a genuinely new peer. + let result = ctx + .raft_log + .filter_out_conflicts_and_append( + 0, + 0, + vec![ + entry(1, 1), + entry(2, 1), + entry(3, 1), + entry(4, 1), + entry(5, 1), + ], + ) + .await + .unwrap(); + + // Assert: no-op β€” durable progress must survive a duplicate zero-prev probe. + assert_eq!(result.unwrap().index, 5); + assert_eq!( + ctx.raft_log.last_entry_id(), + 5, + "last_entry_id must be unchanged" + ); + assert_eq!( + ctx.raft_log.durable_index(), + 5, + "a duplicate prev_log_index=0 resend must not regress durable_index β€” this is what \ + re-arms the withheld-ACK deadlock (RPO=0 withhold never resolves once durable_index \ + is wiped out from under it)" + ); + for i in 1u64..=5 { + assert_eq!( + ctx.raft_log.entry(i).unwrap().unwrap().term, + 1, + "index={i} must not be touched by a duplicate zero-prev probe" + ); + } +} + +/// A `prev_log_index=0` batch that overlaps existing content but also carries genuinely new +/// entries beyond it must append only the new tail β€” mirrors +/// `pipeline_overlap_test::test_filter_conflicts_pipeline_overlap_no_truncation`, anchored at +/// prev=0 instead of prev>0, to prove the same comparison logic applies uniformly regardless +/// of which branch computed `prev_log_index`. +/// +/// # Expected (RED until fixed) +/// [1..5] untouched, [6,7] appended. +#[tokio::test] +async fn test_filter_conflicts_zero_prev_overlap_appends_only_new_tail() { + let ctx = ctx("zero_prev_overlap_appends_only_new_tail"); + + // Arrange: follower has [1..5], term=1 (not necessarily durable yet β€” overlap detection + // must work purely off in-memory content, independent of durability). + for i in 1u64..=5 { + ctx.raft_log.append_entries(vec![entry(i, 1)]).await.unwrap(); + } + assert_eq!(ctx.raft_log.last_entry_id(), 5); + + // Act: leader sends prev_log_index=0 with [1..7] β€” [1..5] match, [6,7] are new. + let new_entries: Vec<_> = (1u64..=7).map(|i| entry(i, 1)).collect(); + let result = ctx.raft_log.filter_out_conflicts_and_append(0, 0, new_entries).await.unwrap(); + + // Assert: existing [1..5] untouched, new tail [6,7] appended. + assert_eq!(result.unwrap().index, 7); + assert_eq!(ctx.raft_log.last_entry_id(), 7); + for i in 1u64..=5 { + assert_eq!( + ctx.raft_log.entry(i).unwrap().unwrap().term, + 1, + "index={i} was already present and must not be truncated" + ); + } + assert_eq!(ctx.raft_log.entry(6).unwrap().unwrap().term, 1); + assert_eq!(ctx.raft_log.entry(7).unwrap().unwrap().term, 1); +} + +/// A genuine conflict at index 1 (different term than what the follower already has) must +/// still truncate and replace β€” proves the fix is a precise reuse of the existing +/// overlap/conflict comparison, not "prev_log_index=0 always becomes a no-op." +/// +/// # Scenario +/// Follower has [1..5] term=1 (stale, from a since-superseded leader). A new leader with no +/// prior knowledge of this follower (or after a purge/snapshot boundary reset) sends +/// prev_log_index=0 with [1..3] all term=2 β€” a real conflict at index=1. +/// +/// # Expected (should already hold both before and after the fix β€” this is the control case) +/// [1..3] replaced with term=2; nothing beyond index=3 survives. +#[tokio::test] +async fn test_filter_conflicts_zero_prev_real_conflict_truncates_and_replaces() { + let ctx = ctx("zero_prev_real_conflict_truncates_and_replaces"); + + // Arrange: follower has [1..5], all term=1. + for i in 1u64..=5 { + ctx.raft_log.append_entries(vec![entry(i, 1)]).await.unwrap(); + } + assert_eq!(ctx.raft_log.last_entry_id(), 5); + + // Act: prev_log_index=0, entries=[1(t2), 2(t2), 3(t2)] β€” conflicts at index=1 immediately. + let result = ctx + .raft_log + .filter_out_conflicts_and_append(0, 0, vec![entry(1, 2), entry(2, 2), entry(3, 2)]) + .await + .unwrap(); + + // Assert: [1..3] replaced with term=2; stale [4,5] from the old leader must not survive. + assert_eq!(result.unwrap().index, 3); + assert_eq!( + ctx.raft_log.last_entry_id(), + 3, + "stale tail beyond the new leader's log must be gone" + ); + for i in 1u64..=3 { + assert_eq!( + ctx.raft_log.entry(i).unwrap().unwrap().term, + 2, + "index={i} must be term=2" + ); + } + assert!( + ctx.raft_log.entry(4).unwrap().is_none(), + "stale index=4 must not survive" + ); + assert!( + ctx.raft_log.entry(5).unwrap().is_none(), + "stale index=5 must not survive" + ); +} diff --git a/d-engine-core/src/storage/raft_log_core_test/quorum_durability_test.rs b/d-engine-core/src/storage/raft_log_core_test/quorum_durability_test.rs new file mode 100644 index 00000000..328830b0 --- /dev/null +++ b/d-engine-core/src/storage/raft_log_core_test/quorum_durability_test.rs @@ -0,0 +1,248 @@ +//! Quorum Durability Tests +//! +//! RPO=0 (#446): leader contributes `durable_index` (not `last_entry_id`) to quorum β€” +//! commit must not advance past what the leader itself has survived fsync for. +//! +//! Superseded design (kept here as history, do not resurrect): the old MemFirst model had +//! the leader contribute `last_entry_id` (in-memory) so IO persistence never sat on the +//! commit critical path. That traded away RPO=0 β€” a majority-acked write could still be +//! lost on correlated power loss before fsync. This file's tests now lock in the new +//! behavior instead of the old one. +//! +//! Follower ACK path (tracked separately, not yet landed): followers will ACK only after +//! their own durable_index catches up β€” so a follower's reported match_index is inherently +//! already durable by the time the leader sees it. +//! +//! Election-eligibility comparison must keep reading the in-memory log, never +//! `durable_index` β€” a separate, independent invariant from the durable-quorum change +//! above, but one a majority-count safety argument for #446 depends on. See +//! `test_election_eligibility_reads_memory_log_not_durable_index`. +//! +//! Note: these tests rely on `RaftLogCore`'s in-memory layer existing (they force a +//! gap between `last_entry_id` and `durable_index` via a gated mock flush). If that layer +//! is ever removed, this file's setup assumptions need revisiting β€” not a decided plan, +//! just a known dependency to check first. +//! +//! Tests that need a genuine, un-fsynced gap between `last_entry_id` and `durable_index` +//! use `MockStorageEngine::not_durable_gated_flush` β€” a real channel-based gate, not a +//! timing guess. An earlier version of this file relied on a long idle-flush timer +//! and assumed a background IO thread just wouldn't get scheduled before the +//! assertions ran; that's a real race (the IO thread is independent of the test's own +//! runtime), and it was intermittently losing under load β€” flaky, not broken logic. Do not +//! reintroduce that pattern here. + +use crate::MockStorageEngine; +use crate::MockTypeConfig; +use crate::RaftLogCore; +use crate::storage::raft_log::RaftLog; +use crate::test_utils::RaftLogCoreTestContext; +use d_engine_proto::common::Entry; +use d_engine_proto::common::LogId; +use std::sync::Arc; +use std::time::Duration; + +/// Entries `1..=n`, all at `term`, no payload β€” the shape these tests need. +fn entries( + n: u64, + term: u64, +) -> Vec { + (1..=n) + .map(|index| Entry { + index, + term, + payload: None, + }) + .collect() +} + +// ── Leader quorum uses durable_index, not last_entry_id (RPO=0) ── + +/// RPO=0: leader's quorum contribution is durable_index (fsync-confirmed), not +/// last_entry_id (in-memory). +/// +/// Even when a follower has already ACKed an index, the leader must not count its own +/// un-fsynced entry toward quorum β€” otherwise a majority-looking commit can still lose +/// data on correlated power loss (the leader's own copy was never actually durable). +/// +/// This test FAILS if calculate_majority_matched_index still uses last_entry_id (the old +/// MemFirst behavior, since revoked). It replaces +/// `test_memfirst_quorum_uses_last_entry_id_not_durable_index`, which asserted the exact +/// opposite of this on purpose β€” that assertion documented a since-revoked design decision. +#[tokio::test] +async fn test_quorum_uses_durable_index_not_last_entry_id() { + // Gate closed: the first flush() call blocks until we send () on `flush_gate` β€” fsync + // deterministically never completes until we say so, no timing involved. + let (storage, flush_gate) = + MockStorageEngine::not_durable_gated_flush("test_quorum_durable_index".into()); + let raft_log = RaftLogCore::::new(1, Arc::new(storage), None, 5000); + std::thread::sleep(Duration::from_millis(10)); // ensure IO thread is ready + + raft_log.append_entries(entries(1, 1)).await.unwrap(); + tokio::time::sleep(Duration::from_millis(50)).await; // let it reach the gate + + assert_eq!(raft_log.last_entry_id(), 1); + assert_eq!(raft_log.durable_index(), 0); // gate never released β€” fsync hasn't completed + + let result = raft_log.calculate_majority_matched_index( + 1, + 0, + vec![1], // one follower reports match=1 (already durable, post-Stage2 semantics) + ); + + // RPO=0: leader contributes durable_index=0, not last_entry_id=1. + // peer_matched_ids = [follower=1, leader=0], sorted desc = [1,0], median(len/2=1) = 0. + // majority_index=0 is not < commit_index=0, so falls through to the term check on + // entry(0) β€” index 0 is not a real entry (log is 1-indexed) β€” Ok(None) β€” result is None. + assert_eq!( + result, None, + "RPO=0: the leader's own un-fsynced entry must not count toward quorum, even when \ + a follower has already acked it β€” one follower alone isn't majority without the \ + leader's own durable contribution" + ); + + let _ = flush_gate.send(()); // release so the blocked IO thread doesn't linger +} + +/// Election-eligibility comparison (`last_log_id`, consumed by +/// `election_handler::handle_vote_request`) must read the in-memory log, never +/// `durable_index`. A follower with an un-fsynced tail must still be able to correctly +/// reject a candidate whose log is genuinely less up to date β€” voting eligibility and +/// commit-durability are two separate concerns and must not be conflated by sharing the +/// same index source. +#[tokio::test] +async fn test_election_eligibility_reads_memory_log_not_durable_index() { + let (storage, flush_gate) = + MockStorageEngine::not_durable_gated_flush("test_election_eligibility_memory_log".into()); + let raft_log = RaftLogCore::::new(1, Arc::new(storage), None, 5000); + std::thread::sleep(Duration::from_millis(10)); + + raft_log.append_entries(entries(10, 1)).await.unwrap(); // entries 1..=10, term=1 + tokio::time::sleep(Duration::from_millis(50)).await; + + assert_eq!(raft_log.durable_index(), 0, "nothing fsynced yet"); + assert_eq!( + raft_log.last_log_id(), + Some(LogId { index: 10, term: 1 }), + "election-eligibility comparison must see the un-fsynced tail, not fall back to \ + durable_index=0 β€” a candidate with a truly-shorter log must still be rejected" + ); + + let _ = flush_gate.send(()); +} + +/// `calculate_majority_matched_index`'s median-based calculation requires an actual +/// majority of `peer_matched_ids` to reach an index before it counts toward commit β€” a +/// minority (here: 2 of 5) reporting a higher index cannot move the result past what the +/// rest of the cluster last confirmed. +/// +/// This is a pure property of the median calculation itself β€” the function has no way to +/// know whether any of its inputs are stale or expired, so this test does not by itself +/// prove anything about stale reports being harmless. It only pins down the arithmetic +/// that a separate, broader safety argument for #446 relies on. +#[tokio::test] +async fn test_majority_matched_index_requires_actual_majority_of_reports() { + let mut ctx = RaftLogCoreTestContext::new("test_majority_requires_actual_majority"); + + ctx.append_entries(1, 10, 1).await; + ctx.raft_log.flush().await.unwrap(); // leader's own entries now durable through 10 + ctx.drain_fsync_completions(); + + assert_eq!(ctx.raft_log.durable_index(), 10); + + // 1 of 4 followers reports index 10; the other 3 are still at their last-known + // value, 9. + let result = ctx.raft_log.calculate_majority_matched_index(1, 9, vec![10, 9, 9, 9]); + + // peer_matched_ids after leader's own contribution = [10, 9, 9, 9, 10] + // sorted desc = [10,10,9,9,9], median(len/2=2) = 9 β€” majority stays at 9, entry(9) + // exists with term=1=current_term, so the result is the previously-safe Some(9), not 10. + assert_eq!( + result, + Some(9), + "a minority (2 of 5) reporting a higher index cannot move majority past what the \ + other 3 nodes last confirmed" + ); +} + +/// Once the leader flushes, quorum calculation should succeed. +/// +/// This test verifies the positive case: after flush, durable_index=1, +/// quorum should proceed normally. +#[tokio::test] +async fn test_quorum_succeeds_after_leader_flush() { + let mut ctx = RaftLogCoreTestContext::new("test_quorum_after_flush"); + + // Append entry 1 β€” threshold=1 so flush fires immediately + ctx.append_entries(1, 1, 1).await; + + // Wait for flush to complete + tokio::time::sleep(Duration::from_millis(50)).await; + ctx.drain_fsync_completions(); + + assert_eq!(ctx.raft_log.last_entry_id(), 1); + assert_eq!( + ctx.raft_log.durable_index(), + 1, + "durable_index must be 1 after threshold flush" + ); + + let new_commit = ctx.raft_log.calculate_majority_matched_index( + 1, + 0, + vec![1], // follower acked + ); + + // Correct: durable_index=1 = last_entry_id=1, quorum should pass + assert_eq!( + new_commit, + Some(1), + "quorum must succeed after leader flush" + ); +} + +// ── Bug 2: gap between last_entry_id and durable_index ── + +/// Demonstrates that after append_entries with a stalled fsync, last_entry_id and +/// durable_index diverge. +/// +/// This is the root condition enabling the bug: both values exist, +/// but quorum calculation only uses the unsafe one. +#[tokio::test] +async fn test_last_entry_id_diverges_from_durable_index_with_mem_first() { + let (storage, flush_gate) = + MockStorageEngine::not_durable_gated_flush("test_diverge_mem_first".into()); + let raft_log = RaftLogCore::::new(1, Arc::new(storage), None, 5000); + std::thread::sleep(Duration::from_millis(10)); + + raft_log.append_entries(entries(5, 1)).await.unwrap(); // entries 1..=5, no flush + tokio::time::sleep(Duration::from_millis(50)).await; + + assert_eq!(raft_log.last_entry_id(), 5, "memory index should be 5"); + assert_eq!( + raft_log.durable_index(), + 0, + "durable_index must remain 0: no flush has run" + ); + // This gap (5 vs 0) is exactly what the quorum bug exploits. + + let _ = flush_gate.send(()); +} + +/// After explicit flush, durable_index must equal last_entry_id. +#[tokio::test] +async fn test_durable_index_equals_last_entry_id_after_flush() { + let mut ctx = RaftLogCoreTestContext::new("test_no_diverge_after_flush"); + + ctx.append_entries(1, 5, 1).await; + ctx.raft_log.flush().await.unwrap(); + ctx.drain_fsync_completions(); + + let last = ctx.raft_log.last_entry_id(); + let durable = ctx.raft_log.durable_index(); + + assert_eq!(last, 5); + assert_eq!( + durable, last, + "durable_index must equal last_entry_id after flush" + ); +} diff --git a/d-engine-core/src/storage/buffered_raft_log_test/raft_properties_test.rs b/d-engine-core/src/storage/raft_log_core_test/raft_properties_test.rs similarity index 73% rename from d-engine-core/src/storage/buffered_raft_log_test/raft_properties_test.rs rename to d-engine-core/src/storage/raft_log_core_test/raft_properties_test.rs index f7acfe7f..bc4de169 100644 --- a/d-engine-core/src/storage/buffered_raft_log_test/raft_properties_test.rs +++ b/d-engine-core/src/storage/raft_log_core_test/raft_properties_test.rs @@ -1,17 +1,10 @@ use crate::storage::raft_log::RaftLog; -use crate::test_utils::{BufferedRaftLogTestContext, simulate_insert_command}; -use crate::{FlushPolicy, PersistenceStrategy}; +use crate::test_utils::{RaftLogCoreTestContext, simulate_insert_command}; use d_engine_proto::common::Entry; #[tokio::test] async fn test_log_matching_property() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_log_matching", - ); + let ctx = RaftLogCoreTestContext::new("test_log_matching"); // Build log: [1,1] [2,1] [3,2] [4,2] [5,3] ctx.append_entries(1, 2, 1).await; @@ -45,17 +38,12 @@ async fn test_log_matching_property() { #[tokio::test] async fn test_leader_completeness_property() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_leader_completeness", - ); + let mut ctx = RaftLogCoreTestContext::new("test_leader_completeness"); // Leader writes entries and flushes so durable_index = 10 ctx.append_entries(1, 10, 1).await; ctx.raft_log.flush().await.unwrap(); + ctx.drain_fsync_completions(); // Simulate majority replication scenario: // Leader has: [1,2,3,4,5,6,7,8,9,10], durable_index=10 @@ -83,13 +71,7 @@ async fn test_leader_completeness_property() { #[tokio::test] async fn test_calculate_majority_matched_index_case0() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_calculate_majority_matched_index_case0", - ); + let mut ctx = RaftLogCoreTestContext::new("test_calculate_majority_matched_index_case0"); ctx.raft_log.reset().await.expect("reset successfully!"); let current_term = 2; @@ -97,6 +79,7 @@ async fn test_calculate_majority_matched_index_case0() { simulate_insert_command(&ctx.raft_log, vec![1], 1).await; simulate_insert_command(&ctx.raft_log, vec![2, 3], 2).await; + ctx.drain_fsync_completions(); assert_eq!( Some(3), @@ -107,13 +90,7 @@ async fn test_calculate_majority_matched_index_case0() { #[tokio::test] async fn test_calculate_majority_matched_index_case1() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_calculate_majority_matched_index_case1", - ); + let ctx = RaftLogCoreTestContext::new("test_calculate_majority_matched_index_case1"); ctx.raft_log.reset().await.expect("reset successfully!"); // Case 1: majority matched index is 2, commit_index: 4, current_term is 3, @@ -131,13 +108,7 @@ async fn test_calculate_majority_matched_index_case1() { #[tokio::test] async fn test_calculate_majority_matched_index_case2() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_calculate_majority_matched_index_case2", - ); + let mut ctx = RaftLogCoreTestContext::new("test_calculate_majority_matched_index_case2"); ctx.raft_log.reset().await.expect("reset successfully!"); // Case 2: majority matched index is 3, commit_index: 2, current_term is 3, @@ -148,6 +119,7 @@ async fn test_calculate_majority_matched_index_case2() { simulate_insert_command(&ctx.raft_log, vec![1], 1).await; simulate_insert_command(&ctx.raft_log, vec![2], 2).await; simulate_insert_command(&ctx.raft_log, vec![3], 3).await; + ctx.drain_fsync_completions(); assert_eq!( Some(3), ctx.raft_log.calculate_majority_matched_index(ct, ci, vec![4, 2]) @@ -156,13 +128,7 @@ async fn test_calculate_majority_matched_index_case2() { #[tokio::test] async fn test_calculate_majority_matched_index_case3() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_calculate_majority_matched_index_case3", - ); + let ctx = RaftLogCoreTestContext::new("test_calculate_majority_matched_index_case3"); // Case 3: majority matched index is 3, commit_index: 2, current_term is 3, // while log(3) term is 2, return None @@ -179,13 +145,7 @@ async fn test_calculate_majority_matched_index_case3() { #[tokio::test] async fn test_calculate_majority_matched_index_case4() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_calculate_majority_matched_index_case4", - ); + let ctx = RaftLogCoreTestContext::new("test_calculate_majority_matched_index_case4"); // Case 4: majority matched index is 2, commit_index: 2, current_term is 3, // while log(2) term is 2, return None @@ -203,13 +163,7 @@ async fn test_calculate_majority_matched_index_case4() { #[tokio::test] async fn test_calculate_majority_matched_index_case5() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_calculate_majority_matched_index_case5", - ); + let mut ctx = RaftLogCoreTestContext::new("test_calculate_majority_matched_index_case5"); // Case 5: stress testing with 100,000 entries // Simulate 100,000 local log entries @@ -223,6 +177,7 @@ async fn test_calculate_majority_matched_index_case5() { let raft_log_entry_ids: Vec = (1..=raft_log_length).collect(); simulate_insert_command(&ctx.raft_log, raft_log_entry_ids, 1).await; + ctx.drain_fsync_completions(); assert_eq!( Some(peer2_match), diff --git a/d-engine-core/src/storage/buffered_raft_log_test/remove_range_test.rs b/d-engine-core/src/storage/raft_log_core_test/remove_range_test.rs similarity index 83% rename from d-engine-core/src/storage/buffered_raft_log_test/remove_range_test.rs rename to d-engine-core/src/storage/raft_log_core_test/remove_range_test.rs index 656e2c25..11ce3da4 100644 --- a/d-engine-core/src/storage/buffered_raft_log_test/remove_range_test.rs +++ b/d-engine-core/src/storage/raft_log_core_test/remove_range_test.rs @@ -1,8 +1,7 @@ use d_engine_proto::common::{Entry, LogId}; use crate::storage::raft_log::RaftLog; -use crate::test_utils::{BufferedRaftLogTestContext, simulate_insert_command}; -use crate::{FlushPolicy, PersistenceStrategy}; +use crate::test_utils::{RaftLogCoreTestContext, simulate_insert_command}; fn entry( index: u64, @@ -17,13 +16,7 @@ fn entry( #[tokio::test] async fn test_remove_middle_range() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_remove_middle_range", - ); + let ctx = RaftLogCoreTestContext::new("test_remove_middle_range"); ctx.raft_log.reset().await.expect("reset successfully!"); // Insert 100 entries @@ -49,13 +42,7 @@ async fn test_remove_middle_range() { #[tokio::test] async fn test_remove_from_start() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_remove_from_start", - ); + let ctx = RaftLogCoreTestContext::new("test_remove_from_start"); ctx.raft_log.reset().await.expect("reset successfully!"); // Insert 100 entries @@ -76,13 +63,7 @@ async fn test_remove_from_start() { #[tokio::test] async fn test_remove_to_end() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_remove_to_end", - ); + let ctx = RaftLogCoreTestContext::new("test_remove_to_end"); ctx.raft_log.reset().await.expect("reset successfully!"); // Insert 100 entries @@ -104,13 +85,7 @@ async fn test_remove_to_end() { #[tokio::test] async fn test_remove_empty_range() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_remove_empty_range", - ); + let ctx = RaftLogCoreTestContext::new("test_remove_empty_range"); ctx.raft_log.reset().await.expect("reset successfully!"); simulate_insert_command(&ctx.raft_log, vec![1, 2, 3], 1).await; @@ -125,13 +100,7 @@ async fn test_remove_empty_range() { #[tokio::test] async fn test_remove_entire_log() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_remove_entire_log", - ); + let ctx = RaftLogCoreTestContext::new("test_remove_entire_log"); ctx.raft_log.reset().await.expect("reset successfully!"); // Insert 100 entries @@ -149,13 +118,7 @@ async fn test_remove_entire_log() { #[tokio::test] async fn test_remove_single_entry() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_remove_single_entry", - ); + let ctx = RaftLogCoreTestContext::new("test_remove_single_entry"); ctx.raft_log.reset().await.expect("reset successfully!"); simulate_insert_command(&ctx.raft_log, vec![1, 2, 3], 1).await; @@ -184,13 +147,7 @@ async fn test_remove_single_entry() { /// Expected: first/last_index_for_term(2) = None; term=1 and term=3 unchanged #[tokio::test] async fn test_remove_range_clears_term_indexes_for_removed_entries() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_remove_range_clears_term_indexes", - ); + let ctx = RaftLogCoreTestContext::new("test_remove_range_clears_term_indexes"); ctx.raft_log.reset().await.unwrap(); // Arrange: term1=[1-3], term2=[4-6], term3=[7-9] @@ -251,13 +208,7 @@ async fn test_remove_range_clears_term_indexes_for_removed_entries() { /// `None` in between. #[tokio::test] async fn test_purge_prefix_removes_entries_and_records_boundary_together() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_purge_prefix_boundary_atomicity", - ); + let ctx = RaftLogCoreTestContext::new("test_purge_prefix_boundary_atomicity"); ctx.raft_log.reset().await.expect("reset successfully!"); // Arrange: entries 1..=100, all term 1. @@ -300,13 +251,7 @@ async fn test_purge_prefix_removes_entries_and_records_boundary_together() { /// the surviving term3 range is untouched. #[tokio::test] async fn test_purge_prefix_multi_term_cutoff_updates_term_indexes_and_boundary() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_purge_prefix_multi_term_cutoff", - ); + let ctx = RaftLogCoreTestContext::new("test_purge_prefix_multi_term_cutoff"); ctx.raft_log.reset().await.expect("reset successfully!"); // Arrange: term1=[1-30], term2=[31-60], term3=[61-90]. diff --git a/d-engine-core/src/storage/raft_log_core_test/replace_range_fsync_test.rs b/d-engine-core/src/storage/raft_log_core_test/replace_range_fsync_test.rs new file mode 100644 index 00000000..4ed48ee7 --- /dev/null +++ b/d-engine-core/src/storage/raft_log_core_test/replace_range_fsync_test.rs @@ -0,0 +1,65 @@ +//! `replace_range_and_submit` (term-conflict truncation, see +//! `filter_out_conflicts_and_append`'s slow path) must submit fsync itself, +//! directly, as part of the same call β€” it must not depend on a subsequent +//! `append_entries()` call to separately trigger persistence. +//! +//! This test pins down that invariant: a term-conflict truncation with no +//! append afterward must still become durable on its own. + +use std::time::Duration; + +use d_engine_proto::common::Entry; + +use crate::storage::raft_log::RaftLog; +use crate::test_utils::RaftLogCoreTestContext; + +fn entry( + index: u64, + term: u64, +) -> Entry { + Entry { + index, + term, + payload: None, + } +} + +/// A term-conflict truncation (`replace_range_and_submit`) must eventually +/// become durable even if no `append_entries()` call follows it. +#[tokio::test] +async fn test_replace_range_becomes_durable_without_a_following_append() { + let mut ctx = + RaftLogCoreTestContext::new("replace_range_becomes_durable_without_a_following_append"); + + // Arrange: log [1,2,3] all term=1, explicitly flushed durable. + ctx.append_entries(1, 3, 1).await; + ctx.raft_log.flush().await.unwrap(); + ctx.drain_fsync_completions(); + assert_eq!(ctx.raft_log.durable_index(), 3, "baseline must be durable"); + + // Act: leader (term=2) sends entries that conflict at index=2 and extend + // the log to index=4. filter_out_conflicts_and_append's slow path detects + // the term mismatch at index=2, truncates [2,3], and replaces with + // [2,3,4] (term=2) via `replace_range_and_submit` β€” with no append_entries() + // call afterward. + let result = ctx + .raft_log + .filter_out_conflicts_and_append(1, 1, vec![entry(2, 2), entry(3, 2), entry(4, 2)]) + .await + .unwrap(); + assert_eq!(result.unwrap().index, 4); + assert_eq!( + ctx.raft_log.last_entry_id(), + 4, + "memory must reflect the replace" + ); + + tokio::time::sleep(Duration::from_millis(50)).await; + ctx.drain_fsync_completions(); + + assert_eq!( + ctx.raft_log.durable_index(), + 4, + "replace_range_and_submit must submit fsync itself, without needing a following append" + ); +} diff --git a/d-engine-core/src/storage/raft_log_core_test/shutdown_test.rs b/d-engine-core/src/storage/raft_log_core_test/shutdown_test.rs new file mode 100644 index 00000000..ff337a2a --- /dev/null +++ b/d-engine-core/src/storage/raft_log_core_test/shutdown_test.rs @@ -0,0 +1,215 @@ +use std::sync::Arc; +use std::time::Duration; + +use bytes::Bytes; + +use crate::storage::raft_log::RaftLog; +use crate::test_utils::{RaftLogCoreTestContext, drain_and_apply_fsync_completions}; +use crate::{MockLogStore, MockMetaStore, MockStorageEngine, MockTypeConfig, RaftLogCore}; +use d_engine_proto::common::{Entry, EntryPayload}; + +fn entry( + index: u64, + term: u64, +) -> Entry { + Entry { + index, + term, + payload: None, + } +} + +/// Verifies `close()` completes without hanging and leaves pending writes +/// durable β€” the explicit shutdown path replaces what used to be verified via +/// `Drop` (there is no dedicated IO thread left to join; `close()` is the +/// only shutdown mechanism now). +#[tokio::test] +async fn test_close_completes_without_hanging() { + let mut ctx = RaftLogCoreTestContext::new("test_shutdown_channel"); + + for i in 1..=10 { + ctx.raft_log + .append_entries(vec![Entry { + index: i, + term: 1, + payload: None, + }]) + .await + .unwrap(); + } + + ctx.raft_log.close().await; + ctx.drain_fsync_completions(); + + assert_eq!( + ctx.raft_log.durable_index(), + 10, + "close() must flush pending writes to durable before returning" + ); +} + +/// Verifies `close()` waits for in-flight persistence work rather than +/// returning while writes are still pending. +#[tokio::test] +async fn test_close_awaits_pending_writes() { + let mut ctx = RaftLogCoreTestContext::new("test_shutdown_await_workers"); + + for i in 1..=50 { + ctx.raft_log + .append_entries(vec![Entry { + index: i, + term: 1, + payload: Some(EntryPayload::command(Bytes::from(vec![0u8; 100]))), + }]) + .await + .unwrap(); + } + + let close_start = std::time::Instant::now(); + ctx.raft_log.close().await; + let close_duration = close_start.elapsed(); + ctx.drain_fsync_completions(); + + assert_eq!( + ctx.raft_log.durable_index(), + 50, + "close() must have flushed all pending writes to durable" + ); + assert!( + close_duration < Duration::from_millis(500), + "close() took too long: {close_duration:?}", + ); +} + +/// Verifies `close()` completes in bounded time even with a full backlog of +/// unflushed writes. +#[tokio::test] +async fn test_close_completes_in_bounded_time() { + let storage = Arc::new(MockStorageEngine::with_id( + "test_shutdown_slow_workers".to_string(), + )); + let (log_flush_tx, mut log_flush_rx) = tokio::sync::mpsc::unbounded_channel(); + let raft_log = RaftLogCore::::new(1, storage, Some(log_flush_tx), 5000); + + for i in 1..=10 { + raft_log + .append_entries(vec![Entry { + index: i, + term: 1, + payload: Some(EntryPayload::command(Bytes::from(vec![0u8; 50]))), + }]) + .await + .unwrap(); + } + + let close_start = std::time::Instant::now(); + raft_log.close().await; + let close_duration = close_start.elapsed(); + drain_and_apply_fsync_completions(&raft_log, &mut log_flush_rx); + + assert_eq!(raft_log.durable_index(), 10); + assert!( + close_duration < Duration::from_millis(1000), + "close() with pending writes took too long" + ); +} + +/// Verifies `close()` still leaves everything durable after several prior +/// explicit `flush()` calls β€” closing is not a no-op just because the caller +/// already flushed once. +#[tokio::test] +async fn test_close_flushes_multiple_batches() { + let mut ctx = RaftLogCoreTestContext::new("test_shutdown_multiple_flushes"); + + for batch in 0..5 { + for i in 1..=10 { + let index = batch * 10 + i; + ctx.raft_log + .append_entries(vec![Entry { + index, + term: 1, + payload: Some(EntryPayload::command(Bytes::from(vec![0u8; 100]))), + }]) + .await + .unwrap(); + } + ctx.raft_log.flush().await.unwrap(); + } + ctx.drain_fsync_completions(); + + let close_start = std::time::Instant::now(); + ctx.raft_log.close().await; + let close_duration = close_start.elapsed(); + ctx.drain_fsync_completions(); + + assert_eq!( + ctx.raft_log.durable_index(), + 50, + "close() must leave all batches durable" + ); + assert!( + close_duration < Duration::from_millis(500), + "close() with multiple prior flushes took too long" + ); +} + +/// Verifies that a fatal `replace_range` failure: +/// 1. Propagates the error synchronously to the caller. +/// 2. Poisons the log β€” writes after the failure are rejected outright, not +/// silently accepted into memory with nothing left to ever persist them. +#[tokio::test] +async fn test_replace_range_failure_poisons_log_and_rejects_future_writes() { + let mut log_store = MockLogStore::new(); + + // replace_range always fails β€” simulates an unrecoverable disk error. + log_store + .expect_replace_range() + .returning(|_, _| Err(crate::Error::Fatal("simulated disk failure".into()))); + + log_store.expect_last_index().returning(|| 0); + log_store.expect_persist_entries().returning(|_| Ok(())); + log_store.expect_entry().returning(|_| Ok(None)); + log_store.expect_get_entries().returning(|_| Ok(vec![])); + log_store.expect_purge().returning(|_| Ok(())); + log_store.expect_load_purge_boundary().returning(|| Ok(None)); + log_store.expect_reset().returning(|| Ok(())); + log_store.expect_truncate().returning(|_| Ok(())); + log_store.expect_is_write_durable().returning(|| true); + log_store.expect_flush().returning(|| Ok(())); + log_store.expect_flush_async().returning(|| Ok(())); + + let mut meta_store = MockMetaStore::new(); + meta_store.expect_save_hard_state().returning(|_| Ok(())); + meta_store.expect_load_hard_state().returning(|| Ok(None)); + meta_store.expect_flush().returning(|| Ok(())); + meta_store.expect_flush_async().returning(|| Ok(())); + + let storage = Arc::new(MockStorageEngine::from(log_store, meta_store)); + let raft_log = RaftLogCore::::new(1, storage, None, 5000); + + // Append [1..4] term=1 β€” persisted inline. + raft_log + .append_entries(vec![entry(1, 1), entry(2, 1), entry(3, 1), entry(4, 1)]) + .await + .unwrap(); + + // Trigger replace_range: term conflict at index 3 (term 1 β†’ 2). + let result = raft_log + .filter_out_conflicts_and_append(2, 1, vec![entry(3, 2), entry(4, 2)]) + .await; + + assert!( + result.is_err(), + "expected replace_range failure to be propagated to caller" + ); + assert!( + raft_log.is_poisoned(), + "a replace_range() failure must poison the log" + ); + + let append_result = raft_log.append_entries(vec![entry(5, 2)]).await; + assert!( + append_result.is_err(), + "writes must be rejected once poisoned" + ); +} diff --git a/d-engine-core/src/storage/buffered_raft_log_test/term_index_test.rs b/d-engine-core/src/storage/raft_log_core_test/term_index_test.rs similarity index 83% rename from d-engine-core/src/storage/buffered_raft_log_test/term_index_test.rs rename to d-engine-core/src/storage/raft_log_core_test/term_index_test.rs index 5e37ef0d..6e797942 100644 --- a/d-engine-core/src/storage/buffered_raft_log_test/term_index_test.rs +++ b/d-engine-core/src/storage/raft_log_core_test/term_index_test.rs @@ -7,18 +7,12 @@ use d_engine_proto::common::Entry; -use crate::test_utils::BufferedRaftLogTestContext; -use crate::{FlushPolicy, PersistenceStrategy, RaftLog}; +use crate::RaftLog; +use crate::test_utils::RaftLogCoreTestContext; #[tokio::test] async fn test_first_index_for_term() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_first_index_for_term", - ); + let ctx = RaftLogCoreTestContext::new("test_first_index_for_term"); ctx.raft_log.reset().await.unwrap(); let entries = vec![ @@ -87,13 +81,7 @@ async fn test_first_index_for_term() { #[tokio::test] async fn test_last_index_for_term() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_last_index_for_term", - ); + let ctx = RaftLogCoreTestContext::new("test_last_index_for_term"); ctx.raft_log.reset().await.unwrap(); let entries = vec![ @@ -161,13 +149,7 @@ async fn test_last_index_for_term() { #[tokio::test] async fn test_term_index_functions_with_purged_logs() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_term_index_with_purged", - ); + let ctx = RaftLogCoreTestContext::new("test_term_index_with_purged"); ctx.raft_log.reset().await.unwrap(); let entries = vec![ @@ -206,20 +188,14 @@ async fn test_term_index_functions_with_purged_logs() { /// Sequential multi-term insertion correctness test. /// -/// Production invariant: all writes to BufferedRaftLog go through the single +/// Production invariant: all writes to RaftLogCore go through the single /// inbound event-loop task; there are no concurrent writers. The previous version /// of this test spawned multiple tasks writing concurrently, which is not a /// production scenario and masked the real invariant. This test verifies /// first/last index tracking across five consecutive term segments. #[tokio::test] async fn test_term_index_sequential_multi_term_insertion() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1000, - }, - "test_term_index_sequential_multi_term", - ); + let ctx = RaftLogCoreTestContext::new("test_term_index_sequential_multi_term"); ctx.raft_log.reset().await.unwrap(); // Insert five consecutive term segments sequentially (as the Raft loop would) @@ -249,7 +225,7 @@ async fn test_term_index_sequential_multi_term_insertion() { /// are correctly rebuilt from disk after a restart. /// /// # Why this matters -/// `BufferedRaftLog::new()` loads all entries from disk and rebuilds three +/// `RaftLogCore::new()` loads all entries from disk and rebuilds three /// in-memory term indexes from scratch. If any of them are incorrectly populated, /// queries like `first_index_for_term()` silently return `None` instead of the /// correct boundary β€” causing the #346 conflict-skip optimization to fall back @@ -262,13 +238,7 @@ async fn test_term_index_sequential_multi_term_insertion() { /// - Assert all three indexes return correct values on the recovered instance #[tokio::test] async fn test_term_indexes_rebuilt_correctly_after_restart() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_term_indexes_rebuilt_after_restart", - ); + let ctx = RaftLogCoreTestContext::new("test_term_indexes_rebuilt_after_restart"); // term 1: indices 1-3, term 2: indices 4-6, term 3: indices 7-9 let entries: Vec = (1u64..=9) @@ -281,7 +251,7 @@ async fn test_term_indexes_rebuilt_correctly_after_restart() { ctx.raft_log.insert_batch(entries).await.unwrap(); ctx.raft_log.flush().await.unwrap(); - // Simulate process restart: new BufferedRaftLog loads from same storage. + // Simulate process restart: new RaftLogCore loads from same storage. let recovered = ctx.recover_from_crash(); // first_index_for_term β€” used by #346 conflict-skip optimization @@ -303,13 +273,7 @@ async fn test_term_indexes_rebuilt_correctly_after_restart() { #[tokio::test] async fn test_term_index_performance_large_dataset() { - let ctx = BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 5000, - }, - "test_term_index_performance", - ); + let ctx = RaftLogCoreTestContext::new("test_term_index_performance"); ctx.raft_log.reset().await.unwrap(); let mut entries = vec![]; diff --git a/d-engine-core/src/storage/buffered_raft_log_test/term_segments_test.rs b/d-engine-core/src/storage/raft_log_core_test/term_segments_test.rs similarity index 95% rename from d-engine-core/src/storage/buffered_raft_log_test/term_segments_test.rs rename to d-engine-core/src/storage/raft_log_core_test/term_segments_test.rs index a385cd5d..53d77b65 100644 --- a/d-engine-core/src/storage/buffered_raft_log_test/term_segments_test.rs +++ b/d-engine-core/src/storage/raft_log_core_test/term_segments_test.rs @@ -14,9 +14,9 @@ use d_engine_proto::common::Entry; -use crate::storage::buffered_raft_log::TermSegments; -use crate::test_utils::BufferedRaftLogTestContext; -use crate::{FlushPolicy, PersistenceStrategy, RaftLog}; +use crate::RaftLog; +use crate::storage::raft_log_core::TermSegments; +use crate::test_utils::RaftLogCoreTestContext; // --------------------------------------------------------------------------- // Helpers @@ -35,14 +35,8 @@ fn entries( .collect() } -fn ctx(name: &str) -> BufferedRaftLogTestContext { - BufferedRaftLogTestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 50, - }, - name, - ) +fn ctx(name: &str) -> RaftLogCoreTestContext { + RaftLogCoreTestContext::new(name) } // --------------------------------------------------------------------------- diff --git a/d-engine-core/src/storage/raft_log_core_test/truncation_fsync_fence_test.rs b/d-engine-core/src/storage/raft_log_core_test/truncation_fsync_fence_test.rs new file mode 100644 index 00000000..0b44fba8 --- /dev/null +++ b/d-engine-core/src/storage/raft_log_core_test/truncation_fsync_fence_test.rs @@ -0,0 +1,82 @@ +//! `RaftLogCore::try_advance_durable_index`'s content check +//! (`entry_term(mark.index) != Some(mark.term)`) is what makes a stale, +//! in-flight fsync's completion report safe to ignore after a truncation: +//! `remove_range` bumps `FsyncWorker`'s generation AND β€” decisively β€” the +//! truncated entry no longer exists at the index the stale report names, so +//! the content check rejects it even if the generation check somehow didn't. +//! +//! Scenario this pins down: a follower has 10 entries synchronously written +//! to its storage engine but not yet fsynced β€” a physical fsync for "up to +//! index 10" is already dispatched and running in the background. Before +//! that fsync returns, a new leader tells the follower its log from index=2 +//! onward is wrong; the follower truncates and replaces it, ending up with +//! only entries [1, 2]. The in-flight fsync then completes and reports +//! "index 10 is durable" β€” `try_advance_durable_index` must reject it: +//! `entry_term(10)` is now `None`, so `durable_index` must stay at or below +//! `last_entry_id()`, never claiming durability for entries [3..=10], which +//! no longer exist in this follower's log. + +use std::sync::Arc; +use std::time::Duration; + +use d_engine_proto::common::Entry; + +use crate::storage::raft_log::RaftLog; +use crate::{MockStorageEngine, MockTypeConfig, RaftLogCore}; + +fn entry( + index: u64, + term: u64, +) -> Entry { + Entry { + index, + term, + payload: None, + } +} + +/// `durable_index()` must never exceed `last_entry_id()` β€” a follower must +/// never claim durability for log entries a truncation has already discarded. +#[tokio::test] +async fn test_durable_index_does_not_adopt_a_stale_fsync_after_truncation() { + // Gate closed: the first flush() call β€” for the original 10-entry batch β€” + // blocks here until we release it, letting us deterministically truncate + // the log while that fsync is still "in flight". + let (storage, flush_gate) = MockStorageEngine::not_durable_gated_flush( + "durable_index_does_not_adopt_a_stale_fsync_after_truncation".into(), + ); + let raft_log = RaftLogCore::::new(1, Arc::new(storage), None, 5000); + + // Old leader (term=1) replicates entries 1..=10. append_entries() persists + // the range inline and submits a physical fsync for "up to index=10" to + // FsyncWorker β€” that fsync is now running in the background, blocked on + // flush_gate. + let entries: Vec = (1..=10).map(|i| entry(i, 1)).collect(); + raft_log.append_entries(entries).await.unwrap(); + + // Give the IO thread + blocking task time to reach the gated flush() call. + tokio::time::sleep(Duration::from_millis(50)).await; + + // New leader (term=2): index=2 conflicts, truncate and replace β€” the + // stale fsync (still blocked on the gate) has no way to observe this. + raft_log.filter_out_conflicts_and_append(1, 1, vec![entry(2, 2)]).await.unwrap(); + assert_eq!( + raft_log.last_entry_id(), + 2, + "log must be truncated and replaced down to [1, 2] before the stale fsync completes" + ); + + // Release the gate β€” the stale fsync (dispatched for index=10, before the + // truncation) now completes. + flush_gate.send(()).unwrap(); + tokio::time::sleep(Duration::from_millis(50)).await; + + assert!( + raft_log.durable_index() <= raft_log.last_entry_id(), + "durable_index ({}) must never exceed last_entry_id ({}) β€” the stale \ + fsync for index=10 must not be adopted after truncation shrank the \ + log to [1, 2]", + raft_log.durable_index(), + raft_log.last_entry_id() + ); +} diff --git a/d-engine-core/src/storage/storage_engine.rs b/d-engine-core/src/storage/storage_engine.rs index 29a61830..5feedb5c 100644 --- a/d-engine-core/src/storage/storage_engine.rs +++ b/d-engine-core/src/storage/storage_engine.rs @@ -79,16 +79,19 @@ pub trait LogStore: Send + Sync + 'static { /// /// Default implementation calls `truncate` then `persist_entries` sequentially /// (non-atomic). Override with a single WriteBatch for true crash atomicity. + /// Returns the highest log index on disk after the operation: the last of + /// `new_entries`, or `from_index - 1` when `new_entries` is empty. async fn replace_range( &self, from_index: u64, new_entries: Vec, - ) -> Result<(), Error> { + ) -> Result { + let new_last = new_entries.last().map(|e| e.index).unwrap_or(from_index.saturating_sub(1)); self.truncate(from_index).await?; if !new_entries.is_empty() { self.persist_entries(new_entries).await?; } - Ok(()) + Ok(new_last) } /// Whether a single `persist_entries` call is crash-safe without an explicit `flush()`. @@ -123,7 +126,7 @@ pub trait LogStore: Send + Sync + 'static { /// Load the purge boundary persisted by the last `purge()` call. /// /// Returns the `LogId` (index + term) of the highest entry ever purged, - /// or `None` if `purge()` has never been called. `BufferedRaftLog::new()` + /// or `None` if `purge()` has never been called. `RaftLogCore::new()` /// uses this to restore `last_purged_index/term` after a restart so that /// `entry_term(last_purged_index)` returns the correct term even though /// the entry has been removed from the in-memory log. diff --git a/d-engine-core/src/test_utils/buffered_raft_log_test_helpers.rs b/d-engine-core/src/test_utils/buffered_raft_log_test_helpers.rs deleted file mode 100644 index 9bcfaa07..00000000 --- a/d-engine-core/src/test_utils/buffered_raft_log_test_helpers.rs +++ /dev/null @@ -1,239 +0,0 @@ -//! Test helpers for BufferedRaftLog testing -//! -//! Provides utilities to simplify BufferedRaftLog unit tests: -//! - Test context management -//! - Mock entry generation -//! - Crash recovery simulation - -use std::sync::Arc; -use std::sync::atomic::AtomicU64; - -use bytes::Bytes; -use d_engine_proto::common::{Entry, EntryPayload}; - -use crate::{ - BufferedRaftLog, FlushPolicy, MockStorageEngine, MockTypeConfig, PersistenceConfig, - PersistenceStrategy, RaftLog, -}; - -/// Test context for BufferedRaftLog tests -pub struct BufferedRaftLogTestContext { - pub raft_log: Arc>, - pub storage: Arc, - pub strategy: PersistenceStrategy, - pub flush_policy: FlushPolicy, - pub instance_id: String, -} - -impl BufferedRaftLogTestContext { - /// Create a new test context with specified strategy and flush policy - pub fn new( - strategy: PersistenceStrategy, - flush_policy: FlushPolicy, - instance_id: &str, - ) -> Self { - let storage = Arc::new(MockStorageEngine::with_id(instance_id.to_string())); - - let (raft_log, receiver) = BufferedRaftLog::new( - 1, - PersistenceConfig { - strategy: strategy.clone(), - flush_policy: flush_policy.clone(), - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - storage.clone(), - ); - let raft_log = raft_log.start(receiver, None); - - // Small delay to ensure processor is ready - std::thread::sleep(std::time::Duration::from_millis(10)); - - Self { - raft_log, - storage, - strategy, - flush_policy, - instance_id: instance_id.to_string(), - } - } - - /// Helper to append a batch of entries with specified range and term - pub async fn append_entries( - &self, - start: u64, - count: u64, - term: u64, - ) { - let entries: Vec<_> = (start..start + count) - .map(|index| Entry { - index, - term, - payload: Some(EntryPayload::command(Bytes::from(b"data".to_vec()))), - }) - .collect(); - - self.raft_log.append_entries(entries).await.unwrap(); - } - - /// Create a context where `is_write_durable()=false`. - /// - /// Returns the context and a counter incremented on every `flush()` call. - /// Use this to verify: - /// - `durable_index` advances only after flush, not after write - /// - Multiple rapid writes are batched into fewer flushes - pub fn new_not_durable( - flush_policy: FlushPolicy, - instance_id: &str, - ) -> (Self, Arc) { - let (storage, flush_count) = MockStorageEngine::not_durable(instance_id.to_string()); - let storage = Arc::new(storage); - - let (raft_log, receiver) = BufferedRaftLog::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: flush_policy.clone(), - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - storage.clone(), - ); - let raft_log = raft_log.start(receiver, None); - std::thread::sleep(std::time::Duration::from_millis(10)); - - let ctx = Self { - raft_log, - storage, - strategy: PersistenceStrategy::MemFirst, - flush_policy, - instance_id: instance_id.to_string(), - }; - (ctx, flush_count) - } - - /// Simulate crash recovery from the same storage instance - pub fn recover_from_crash(&self) -> Self { - // Use same instance ID to recover data from thread_local storage - let storage = Arc::new(MockStorageEngine::with_id(self.instance_id.clone())); - - let (raft_log, receiver) = BufferedRaftLog::new( - 1, - PersistenceConfig { - strategy: self.strategy.clone(), - flush_policy: self.flush_policy.clone(), - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - storage.clone(), - ); - let raft_log = raft_log.start(receiver, None); - - // Small delay to ensure processor is ready - std::thread::sleep(std::time::Duration::from_millis(10)); - - Self { - raft_log, - storage, - strategy: self.strategy.clone(), - flush_policy: self.flush_policy.clone(), - instance_id: self.instance_id.clone(), - } - } -} - -/// Generate mock log entries with sequential indexes -pub fn mock_entries( - start: u64, - count: u64, - term: u64, -) -> Vec { - (start..start + count) - .map(|index| Entry { - index, - term, - payload: Some(EntryPayload::command(Bytes::from( - format!("data_{index}").into_bytes(), - ))), - }) - .collect() -} - -/// Generate empty mock entries (no payload) -pub fn mock_empty_entries( - start: u64, - count: u64, - term: u64, -) -> Vec { - (start..start + count) - .map(|index| Entry { - index, - term, - payload: None, - }) - .collect() -} - -/// Insert single entry helper -pub async fn insert_single_entry( - raft_log: &Arc>, - index: u64, - term: u64, -) { - let entry = Entry { - index, - term, - payload: None, - }; - raft_log.insert_batch(vec![entry]).await.expect("insert should succeed"); -} - -/// Generate mock insert command payload bytes -fn mock_insert_command_payload(ids: Vec) -> Bytes { - let commands: Vec = ids.iter().map(|id| format!("insert_{id}")).collect(); - Bytes::from(commands.join(",")) -} - -/// Simulate inserting command entries into the log -/// -/// Creates command entries with pre-allocated indexes and appends them to the log. -/// Each ID becomes a command payload with the given term. -pub async fn simulate_insert_command( - raft_log: &Arc>, - ids: Vec, - term: u64, -) { - let mut entries = Vec::new(); - for id in ids { - let entry = Entry { - index: raft_log.pre_allocate_raft_logs_next_index(), - term, - payload: Some(EntryPayload::command(mock_insert_command_payload(vec![id]))), - }; - entries.push(entry); - } - raft_log.insert_batch(entries).await.unwrap(); - raft_log.flush().await.unwrap(); -} - -/// Simulate deleting entries from the log for a range of IDs -/// -/// Creates delete command entries for each ID in the specified range and appends -/// them to the log. Each ID in the range becomes a separate delete command entry. -pub async fn simulate_delete_command( - raft_log: &Arc>, - id_range: std::ops::RangeInclusive, - term: u64, -) { - let mut entries = Vec::new(); - for id in id_range { - let entry = Entry { - index: raft_log.pre_allocate_raft_logs_next_index(), - term, - payload: Some(EntryPayload::command(Bytes::from(format!("delete_{id}")))), - }; - entries.push(entry); - } - raft_log.insert_batch(entries).await.unwrap(); - raft_log.flush().await.unwrap(); -} diff --git a/d-engine-core/src/test_utils/log_capture.rs b/d-engine-core/src/test_utils/log_capture.rs index cae2848a..a04420dc 100644 --- a/d-engine-core/src/test_utils/log_capture.rs +++ b/d-engine-core/src/test_utils/log_capture.rs @@ -2,7 +2,7 @@ //! //! `tracing-test`'s `#[traced_test]` only captures events emitted on the //! annotated test's own thread. Some code under test runs on a dedicated OS -//! thread with its own tokio runtime (e.g. `BufferedRaftLog`'s IO thread), +//! thread with its own tokio runtime (e.g. a dedicated IO thread), //! whose log output `#[traced_test]` cannot see. This installs a //! process-wide subscriber instead, visible from any thread. //! diff --git a/d-engine-core/src/test_utils/metrics_capture.rs b/d-engine-core/src/test_utils/metrics_capture.rs new file mode 100644 index 00000000..cf393804 --- /dev/null +++ b/d-engine-core/src/test_utils/metrics_capture.rs @@ -0,0 +1,227 @@ +use metrics::{ + Counter, CounterFn, Gauge, GaugeFn, Histogram, HistogramFn, Key, KeyName, Metadata, Recorder, + SharedString, Unit, +}; +use std::collections::HashMap; +use std::sync::{Arc, Mutex}; + +#[derive(Default)] +struct Store { + gauges: HashMap, + counters: HashMap, + histograms: HashMap>, +} + +fn key_string(key: &Key) -> String { + let mut labels: Vec = + key.labels().map(|l| format!("{}={}", l.key(), l.value())).collect(); + labels.sort(); + if labels.is_empty() { + key.name().to_string() + } else { + format!("{}{{{}}}", key.name(), labels.join(",")) + } +} + +struct GaugeHandle { + store: Arc>, + key: String, +} + +impl GaugeFn for GaugeHandle { + fn increment( + &self, + value: f64, + ) { + let mut s = self.store.lock().unwrap(); + *s.gauges.entry(self.key.clone()).or_default() += value; + } + fn decrement( + &self, + value: f64, + ) { + let mut s = self.store.lock().unwrap(); + *s.gauges.entry(self.key.clone()).or_default() -= value; + } + fn set( + &self, + value: f64, + ) { + self.store.lock().unwrap().gauges.insert(self.key.clone(), value); + } +} + +struct CounterHandle { + store: Arc>, + key: String, +} + +impl CounterFn for CounterHandle { + fn increment( + &self, + value: u64, + ) { + let mut s = self.store.lock().unwrap(); + *s.counters.entry(self.key.clone()).or_default() += value; + } + fn absolute( + &self, + value: u64, + ) { + self.store.lock().unwrap().counters.insert(self.key.clone(), value); + } +} + +struct HistogramHandle { + store: Arc>, + key: String, +} + +impl HistogramFn for HistogramHandle { + fn record( + &self, + value: f64, + ) { + self.store + .lock() + .unwrap() + .histograms + .entry(self.key.clone()) + .or_default() + .push(value); + } +} + +/// In-memory metrics recorder for tests. +/// +/// Install with `metrics::set_global_recorder(capture.clone())` or use +/// `with_recorder` to scope it to a single thread. +pub struct MetricsCapture { + store: Arc>, +} + +impl Clone for MetricsCapture { + fn clone(&self) -> Self { + Self { + store: Arc::clone(&self.store), + } + } +} + +impl Default for MetricsCapture { + fn default() -> Self { + Self { + store: Arc::new(Mutex::new(Store::default())), + } + } +} + +impl MetricsCapture { + pub fn new() -> Self { + Self::default() + } + + /// Returns the last `set()` value for a gauge identified by name + labels. + /// + /// `labels` is a slice of `("key", "value")` pairs; order does not matter. + pub fn gauge( + &self, + name: &str, + labels: &[(&str, &str)], + ) -> Option { + let k = build_key_string(name, labels); + self.store.lock().unwrap().gauges.get(&k).copied() + } + + /// Returns the cumulative increment for a counter identified by name + labels. + pub fn counter( + &self, + name: &str, + labels: &[(&str, &str)], + ) -> u64 { + let k = build_key_string(name, labels); + self.store.lock().unwrap().counters.get(&k).copied().unwrap_or(0) + } + + /// Every value recorded into a histogram, in recording order. + pub fn histogram( + &self, + name: &str, + labels: &[(&str, &str)], + ) -> Vec { + let k = build_key_string(name, labels); + self.store.lock().unwrap().histograms.get(&k).cloned().unwrap_or_default() + } +} + +fn build_key_string( + name: &str, + labels: &[(&str, &str)], +) -> String { + if labels.is_empty() { + return name.to_string(); + } + let mut pairs: Vec = labels.iter().map(|(k, v)| format!("{}={}", k, v)).collect(); + pairs.sort(); + format!("{}{{{}}}", name, pairs.join(",")) +} + +impl Recorder for MetricsCapture { + fn describe_counter( + &self, + _key: KeyName, + _unit: Option, + _description: SharedString, + ) { + } + fn describe_gauge( + &self, + _key: KeyName, + _unit: Option, + _description: SharedString, + ) { + } + fn describe_histogram( + &self, + _key: KeyName, + _unit: Option, + _description: SharedString, + ) { + } + + fn register_counter( + &self, + key: &Key, + _metadata: &Metadata<'_>, + ) -> Counter { + let k = key_string(key); + // pre-insert so counter() returns 0 before first increment + self.store.lock().unwrap().counters.entry(k.clone()).or_default(); + Counter::from_arc(Arc::new(CounterHandle { + store: Arc::clone(&self.store), + key: k, + })) + } + + fn register_gauge( + &self, + key: &Key, + _metadata: &Metadata<'_>, + ) -> Gauge { + Gauge::from_arc(Arc::new(GaugeHandle { + store: Arc::clone(&self.store), + key: key_string(key), + })) + } + + fn register_histogram( + &self, + key: &Key, + _metadata: &Metadata<'_>, + ) -> Histogram { + Histogram::from_arc(Arc::new(HistogramHandle { + store: Arc::clone(&self.store), + key: key_string(key), + })) + } +} diff --git a/d-engine-core/src/test_utils/mock/mock_raft_builder.rs b/d-engine-core/src/test_utils/mock/mock_raft_builder.rs index 7c8cbee6..4e932800 100644 --- a/d-engine-core/src/test_utils/mock/mock_raft_builder.rs +++ b/d-engine-core/src/test_utils/mock/mock_raft_builder.rs @@ -358,7 +358,6 @@ pub fn mock_raft_log() -> MockRaftLog { raft_log.expect_save_hard_state().returning(|_| Ok(())); raft_log.expect_calculate_majority_matched_index().returning(|_, _, _| None); raft_log.expect_close().returning(|| ()); - raft_log.expect_is_poisoned().returning(|| false); raft_log } diff --git a/d-engine-core/src/test_utils/mock/mock_storage_engine.rs b/d-engine-core/src/test_utils/mock/mock_storage_engine.rs index b21f91da..171b7815 100644 --- a/d-engine-core/src/test_utils/mock/mock_storage_engine.rs +++ b/d-engine-core/src/test_utils/mock/mock_storage_engine.rs @@ -263,7 +263,7 @@ impl MockStorageEngine { } else { data.insert(last_key, new_last_index.to_be_bytes().to_vec()); } - Ok(()) + Ok(new_last_index) }); } @@ -319,10 +319,8 @@ impl MockStorageEngine { /// Create a MockStorageEngine where `is_write_durable()=false` and the **first** /// `flush()` call returns an error, simulating a transient fsync failure. /// - /// After a failed fsync, `batch_processor` logs the error and does NOT zero - /// `pending_max` (the success branch `else { pending_max = 0 }` is not taken). - /// This is the deterministic pre-condition needed to exercise the bug where - /// `handle_non_write_cmd(IOTask::Reset)` forgets to zero `pending_max`. + /// A failed fsync poisons the log permanently (see `FsyncWorker`). This is the + /// deterministic way to reach the poisoned state through a real fsync failure. pub fn not_durable_first_flush_fails(id: String) -> Self { let mut mock_log_store = MockLogStore::new(); let mut mock_meta_store = MockMetaStore::new(); @@ -356,8 +354,8 @@ impl MockStorageEngine { /// /// Distinct from `not_durable_first_flush_fails`, which only fails the later /// fsync step β€” `persist_entries()` and `flush()` are two independent failure - /// surfaces in `BufferedRaftLog` (`persist_pending_range` vs - /// `FsyncCoordinator::run_until_caught_up`). `flush()` itself always succeeds + /// surfaces in `RaftLogCore` (`persist_pending_range` vs + /// `FsyncWorker::run_until_caught_up`). `flush()` itself always succeeds /// here, keeping the two surfaces isolated. All subsequent `persist_entries()` /// calls succeed. pub fn not_durable_first_persist_fails(id: String) -> Self { @@ -393,8 +391,8 @@ impl MockStorageEngine { /// Create a MockStorageEngine where `replace_range()` always fails, /// simulating a fatal storage error during conflict-resolution - /// (truncate + write). `handle_non_write_cmd`'s `IOTask::ReplaceRange` - /// arm treats this as unrecoverable β€” disk state is now uncertain. + /// (truncate + write). `replace_range_and_submit` treats this as + /// unrecoverable and poisons the log β€” disk state is now uncertain. pub fn not_durable_replace_range_fails(id: String) -> Self { let mut mock_log_store = MockLogStore::new(); let mut mock_meta_store = MockMetaStore::new(); @@ -635,6 +633,57 @@ impl MockStorageEngine { (engine, tx) } + /// Create a MockStorageEngine where the first `persist_entries()` call blocks + /// until the returned sender fires. `flush()`/`is_write_durable()` are left at + /// their always-succeeds default (`configure_durable`) β€” this gate is only + /// about the write-to-storage-engine step, not fsync. + /// + /// Use this to freeze the IO thread mid-persist so a concurrent truncation + /// can be driven deterministically β€” see + /// `durable_index_truncation_clamp_test.rs`. + pub fn not_durable_gated_persist(id: String) -> (Self, std::sync::mpsc::Sender<()>) { + let (tx, rx) = std::sync::mpsc::channel::<()>(); + let rx = Mutex::new(Some(rx)); + + let mut mock_log_store = MockLogStore::new(); + let mut mock_meta_store = MockMetaStore::new(); + + Self::configure_mocks(&mut mock_log_store, &mut mock_meta_store, &id); + // persist_entries is gated below instead of via configure_persist_entries_success. + Self::configure_replace_range_success(&mut mock_log_store, &id); + Self::configure_purge_success(&mut mock_log_store); + Self::configure_reset_success(&mut mock_log_store, &id); + Self::configure_save_hard_state_success(&mut mock_meta_store, &id); + Self::configure_durable(&mut mock_log_store); + + let instance_id_ref = id.clone(); + mock_log_store.expect_persist_entries().returning(move |entries| { + // Only the first call blocks β€” take() leaves None for subsequent calls. + if let Some(gate) = rx.lock().unwrap().take() { + let _ = gate.recv(); // blocks until the test sends () + } + let mut data = MOCK_STORAGE_DATA.lock().unwrap(); + for entry in &entries { + let key = format!("{instance_id_ref}_entry_{}", entry.index); + let value = bincode::serialize(entry).unwrap(); + data.insert(key, value); + } + if let Some(last_entry) = entries.last() { + let key = format!("{instance_id_ref}_last_index"); + data.insert(key, last_entry.index.to_be_bytes().to_vec()); + } + Ok(()) + }); + + let engine = Self { + log_store: Arc::new(mock_log_store), + meta_store: Arc::new(mock_meta_store), + instance_id: id, + }; + + (engine, tx) + } + /// Configure `is_write_durable=true` and no-op flush (durable mock). fn configure_durable(log_store: &mut MockLogStore) { log_store.expect_is_write_durable().returning(|| true); diff --git a/d-engine-core/src/test_utils/mod.rs b/d-engine-core/src/test_utils/mod.rs index 9607c92a..4041c103 100644 --- a/d-engine-core/src/test_utils/mod.rs +++ b/d-engine-core/src/test_utils/mod.rs @@ -1,9 +1,9 @@ //! the test_utils folder here will share utils or test components between unit //! tests and integration tests -mod buffered_raft_log_test_helpers; mod common; mod entry_builder; pub mod mock; +mod raft_log_core_test_helpers; mod replication_test_helpers; mod snapshot; @@ -19,10 +19,16 @@ mod log_capture; #[cfg(any(test, feature = "__test_support"))] pub use log_capture::*; -pub use buffered_raft_log_test_helpers::*; +#[cfg(any(test, feature = "__test_support"))] +mod metrics_capture; + +#[cfg(any(test, feature = "__test_support"))] +pub use metrics_capture::MetricsCapture; + pub use common::*; pub use entry_builder::*; pub use mock::*; +pub use raft_log_core_test_helpers::*; pub use replication_test_helpers::*; pub use snapshot::*; diff --git a/d-engine-core/src/test_utils/raft_log_core_test_helpers.rs b/d-engine-core/src/test_utils/raft_log_core_test_helpers.rs new file mode 100644 index 00000000..3d8be0a4 --- /dev/null +++ b/d-engine-core/src/test_utils/raft_log_core_test_helpers.rs @@ -0,0 +1,243 @@ +//! Test helpers for RaftLogCore testing +//! +//! Provides utilities to simplify RaftLogCore unit tests. Entry-generation +//! and log-mutation helpers (`mock_entries`, `simulate_insert_command`, ...) +//! are generic over `L: RaftLog`, so they live here once instead of being +//! duplicated per struct. + +use std::sync::Arc; +use std::sync::atomic::AtomicU64; + +use bytes::Bytes; +use d_engine_proto::common::{Entry, EntryPayload}; + +use crate::{MockStorageEngine, MockTypeConfig, RaftLog, RaftLogCore}; + +/// Test context for RaftLogCore tests +pub struct RaftLogCoreTestContext { + pub raft_log: Arc>, + pub storage: Arc, + pub instance_id: String, + log_flush_rx: tokio::sync::mpsc::UnboundedReceiver, +} + +impl RaftLogCoreTestContext { + /// Create a new test context. + /// + /// `RaftLogCore::new()` executes inline, synchronously β€” there is no + /// dedicated IO thread to wait on, so no startup sleep is needed here. + pub fn new(instance_id: &str) -> Self { + let storage = Arc::new(MockStorageEngine::with_id(instance_id.to_string())); + let (log_flush_tx, log_flush_rx) = tokio::sync::mpsc::unbounded_channel(); + let raft_log = RaftLogCore::new(1, storage.clone(), Some(log_flush_tx), 5000); + + Self { + raft_log, + storage, + instance_id: instance_id.to_string(), + log_flush_rx, + } + } + + /// Stands in for `raft.rs`'s `InternalEvent::FsyncCompleted` handler, + /// which isn't running in these `RaftLogCore`-only unit tests. Call + /// after any operation that should make `durable_index` advance + /// (`append_entries`, `flush`, truncation + resync, ...) and before + /// asserting on `durable_index()`. + pub fn drain_fsync_completions(&mut self) { + drain_and_apply_fsync_completions(&self.raft_log, &mut self.log_flush_rx); + } + + /// Helper to append a batch of entries with specified range and term + pub async fn append_entries( + &self, + start: u64, + count: u64, + term: u64, + ) { + let entries: Vec<_> = (start..start + count) + .map(|index| d_engine_proto::common::Entry { + index, + term, + payload: Some(d_engine_proto::common::EntryPayload::command( + bytes::Bytes::from(b"data".to_vec()), + )), + }) + .collect(); + + self.raft_log.append_entries(entries).await.unwrap(); + } + + /// Create a context where `is_write_durable()=false`. + /// + /// Returns the context and a counter incremented on every `flush()` call. + pub fn new_not_durable(instance_id: &str) -> (Self, Arc) { + let (storage, flush_count) = MockStorageEngine::not_durable(instance_id.to_string()); + let storage = Arc::new(storage); + let (log_flush_tx, log_flush_rx) = tokio::sync::mpsc::unbounded_channel(); + let raft_log = RaftLogCore::new(1, storage.clone(), Some(log_flush_tx), 5000); + + let ctx = Self { + raft_log, + storage, + instance_id: instance_id.to_string(), + log_flush_rx, + }; + (ctx, flush_count) + } + + /// Simulate crash recovery from the same storage instance + pub fn recover_from_crash(&self) -> Self { + let storage = Arc::new(MockStorageEngine::with_id(self.instance_id.clone())); + let (log_flush_tx, log_flush_rx) = tokio::sync::mpsc::unbounded_channel(); + let raft_log = RaftLogCore::new(1, storage.clone(), Some(log_flush_tx), 5000); + + Self { + raft_log, + storage, + instance_id: self.instance_id.clone(), + log_flush_rx, + } + } +} + +/// Generate mock log entries with sequential indexes +pub fn mock_entries( + start: u64, + count: u64, + term: u64, +) -> Vec { + (start..start + count) + .map(|index| Entry { + index, + term, + payload: Some(EntryPayload::command(Bytes::from( + format!("data_{index}").into_bytes(), + ))), + }) + .collect() +} + +/// Generate empty mock entries (no payload) +pub fn mock_empty_entries( + start: u64, + count: u64, + term: u64, +) -> Vec { + (start..start + count) + .map(|index| Entry { + index, + term, + payload: None, + }) + .collect() +} + +/// Insert single entry helper. Generic over `L: RaftLog`. +pub async fn insert_single_entry( + raft_log: &Arc, + index: u64, + term: u64, +) { + let entry = Entry { + index, + term, + payload: None, + }; + raft_log.insert_batch(vec![entry]).await.expect("insert should succeed"); +} + +/// Generate mock insert command payload bytes +fn mock_insert_command_payload(ids: Vec) -> Bytes { + let commands: Vec = ids.iter().map(|id| format!("insert_{id}")).collect(); + Bytes::from(commands.join(",")) +} + +/// Simulate inserting command entries into the log +/// +/// Creates command entries with pre-allocated indexes and appends them to the +/// log. Each ID becomes a command payload with the given term. +pub async fn simulate_insert_command( + raft_log: &Arc, + ids: Vec, + term: u64, +) { + let mut entries = Vec::new(); + for id in ids { + let entry = Entry { + index: raft_log.pre_allocate_raft_logs_next_index(), + term, + payload: Some(EntryPayload::command(mock_insert_command_payload(vec![id]))), + }; + entries.push(entry); + } + raft_log.insert_batch(entries).await.unwrap(); + raft_log.flush().await.unwrap(); +} + +/// Simulate deleting entries from the log for a range of IDs +/// +/// Creates delete command entries for each ID in the specified range and +/// appends them to the log. Each ID in the range becomes a separate delete +/// command entry. +pub async fn simulate_delete_command( + raft_log: &Arc, + id_range: std::ops::RangeInclusive, + term: u64, +) { + let mut entries = Vec::new(); + for id in id_range { + let entry = Entry { + index: raft_log.pre_allocate_raft_logs_next_index(), + term, + payload: Some(EntryPayload::command(Bytes::from(format!("delete_{id}")))), + }; + entries.push(entry); + } + raft_log.insert_batch(entries).await.unwrap(); + raft_log.flush().await.unwrap(); +} + +/// Stands in for `raft.rs`'s `InternalEvent::FsyncCompleted` handler, which +/// `RaftLogCore`-only unit tests don't have running. `durable_index` only +/// advances when something calls `try_advance_durable_index(mark)` in +/// response to that event β€” `FsyncWorker::notify_fsync_completed` only +/// *sends* the event, it never writes `durable_index` itself. A test that +/// registers a `log_flush_tx` and wants to see `durable_index()` advance +/// must drain that channel through this helper. +pub fn drain_and_apply_fsync_completions( + raft_log: &Arc, + log_flush_rx: &mut tokio::sync::mpsc::UnboundedReceiver, +) { + while let Ok(event) = log_flush_rx.try_recv() { + if let crate::InternalEvent::FsyncCompleted { mark, sent_at: _ } = event { + raft_log.try_advance_durable_index(mark); + } + } +} + +/// Like `drain_and_apply_fsync_completions`, but waits: applies `FsyncCompleted` +/// events as they arrive until `durable_index() >= target`. Panics after `timeout`, +/// so a fsync that never completes fails loudly instead of racing a fixed sleep. +pub async fn wait_for_durable_index( + raft_log: &Arc, + log_flush_rx: &mut tokio::sync::mpsc::UnboundedReceiver, + target: u64, + timeout: std::time::Duration, +) { + let deadline = tokio::time::Instant::now() + timeout; + while raft_log.durable_index() < target { + let event = tokio::time::timeout_at(deadline, log_flush_rx.recv()) + .await + .unwrap_or_else(|_| { + panic!( + "durable_index stuck at {} (< {target}) after {timeout:?}", + raft_log.durable_index() + ) + }) + .expect("log_flush channel closed before durable_index reached target"); + if let crate::InternalEvent::FsyncCompleted { mark, sent_at: _ } = event { + raft_log.try_advance_durable_index(mark); + } + } +} diff --git a/d-engine-core/src/utils/scoped_timer.rs b/d-engine-core/src/utils/scoped_timer.rs index e4c7d9a5..dbfa2f5a 100644 --- a/d-engine-core/src/utils/scoped_timer.rs +++ b/d-engine-core/src/utils/scoped_timer.rs @@ -19,5 +19,7 @@ impl Drop for ScopedTimer { fn drop(&mut self) { let elapsed = self.start.elapsed(); trace!(target: "timing", "[TIMING] {} took {} ms", self.name, elapsed.as_millis()); + metrics::histogram!("core.timing.scoped_duration_ms", "phase" => self.name) + .record(elapsed.as_secs_f64() * 1_000.0); } } diff --git a/d-engine-core/src/watch/mod.rs b/d-engine-core/src/watch/mod.rs index 0998f2c0..32fcc4ba 100644 --- a/d-engine-core/src/watch/mod.rs +++ b/d-engine-core/src/watch/mod.rs @@ -87,6 +87,7 @@ //! watcher_buffer_size: 256, //! enable_metrics: true, //! max_watcher_count: 5000, +//! heartbeat_interval_ms: 30_000, //! }; //! ``` //! diff --git a/d-engine-proto/go/go.mod b/d-engine-proto/go/go.mod index 68f93120..b5dd4aa2 100644 --- a/d-engine-proto/go/go.mod +++ b/d-engine-proto/go/go.mod @@ -3,13 +3,13 @@ module github.com/deventlab/d-engine/proto go 1.25.0 require ( - google.golang.org/grpc v1.80.0 + google.golang.org/grpc v1.83.2 google.golang.org/protobuf v1.36.11 ) require ( - golang.org/x/net v0.53.0 // indirect - golang.org/x/sys v0.43.0 // indirect - golang.org/x/text v0.36.0 // indirect - google.golang.org/genproto/googleapis/rpc v0.0.0-20260406210006-6f92a3bedf2d // indirect + golang.org/x/net v0.58.0 // indirect + golang.org/x/sys v0.47.0 // indirect + golang.org/x/text v0.41.0 // indirect + google.golang.org/genproto/googleapis/rpc v0.0.0-20260526163538-3dc84a4a5aaa // indirect ) diff --git a/d-engine-proto/go/go.sum b/d-engine-proto/go/go.sum index adb3ad1c..2d3ad8d1 100644 --- a/d-engine-proto/go/go.sum +++ b/d-engine-proto/go/go.sum @@ -12,27 +12,27 @@ github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0= github.com/google/uuid v1.6.0/go.mod h1:TIyPZe4MgqvfeYDBFedMoGGpEw/LqOeaOT+nhxU+yHo= go.opentelemetry.io/auto/sdk v1.2.1 h1:jXsnJ4Lmnqd11kwkBV2LgLoFMZKizbCi5fNZ/ipaZ64= go.opentelemetry.io/auto/sdk v1.2.1/go.mod h1:KRTj+aOaElaLi+wW1kO/DZRXwkF4C5xPbEe3ZiIhN7Y= -go.opentelemetry.io/otel v1.39.0 h1:8yPrr/S0ND9QEfTfdP9V+SiwT4E0G7Y5MO7p85nis48= -go.opentelemetry.io/otel v1.39.0/go.mod h1:kLlFTywNWrFyEdH0oj2xK0bFYZtHRYUdv1NklR/tgc8= -go.opentelemetry.io/otel/metric v1.39.0 h1:d1UzonvEZriVfpNKEVmHXbdf909uGTOQjA0HF0Ls5Q0= -go.opentelemetry.io/otel/metric v1.39.0/go.mod h1:jrZSWL33sD7bBxg1xjrqyDjnuzTUB0x1nBERXd7Ftcs= -go.opentelemetry.io/otel/sdk v1.39.0 h1:nMLYcjVsvdui1B/4FRkwjzoRVsMK8uL/cj0OyhKzt18= -go.opentelemetry.io/otel/sdk v1.39.0/go.mod h1:vDojkC4/jsTJsE+kh+LXYQlbL8CgrEcwmt1ENZszdJE= -go.opentelemetry.io/otel/sdk/metric v1.39.0 h1:cXMVVFVgsIf2YL6QkRF4Urbr/aMInf+2WKg+sEJTtB8= -go.opentelemetry.io/otel/sdk/metric v1.39.0/go.mod h1:xq9HEVH7qeX69/JnwEfp6fVq5wosJsY1mt4lLfYdVew= -go.opentelemetry.io/otel/trace v1.39.0 h1:2d2vfpEDmCJ5zVYz7ijaJdOF59xLomrvj7bjt6/qCJI= -go.opentelemetry.io/otel/trace v1.39.0/go.mod h1:88w4/PnZSazkGzz/w84VHpQafiU4EtqqlVdxWy+rNOA= -golang.org/x/net v0.53.0 h1:d+qAbo5L0orcWAr0a9JweQpjXF19LMXJE8Ey7hwOdUA= -golang.org/x/net v0.53.0/go.mod h1:JvMuJH7rrdiCfbeHoo3fCQU24Lf5JJwT9W3sJFulfgs= -golang.org/x/sys v0.43.0 h1:Rlag2XtaFTxp19wS8MXlJwTvoh8ArU6ezoyFsMyCTNI= -golang.org/x/sys v0.43.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw= -golang.org/x/text v0.36.0 h1:JfKh3XmcRPqZPKevfXVpI1wXPTqbkE5f7JA92a55Yxg= -golang.org/x/text v0.36.0/go.mod h1:NIdBknypM8iqVmPiuco0Dh6P5Jcdk8lJL0CUebqK164= +go.opentelemetry.io/otel v1.44.0 h1:JjwHmHpA4iZ3wBxluu2fbbE7j4kqlE8jXyAyPXH7HqU= +go.opentelemetry.io/otel v1.44.0/go.mod h1:BMgjTHL9WPRlRjL2oZCBTL4whCGtXch2H4BhOPIAyYc= +go.opentelemetry.io/otel/metric v1.44.0 h1:1w0gILTcHdr3YI+ixLyjemwrVnsMURbTZFrSYCdDdmc= +go.opentelemetry.io/otel/metric v1.44.0/go.mod h1:8O7hanEPBNgEMmybD3s2VBKcgWOCsA6tzHBPODAiquo= +go.opentelemetry.io/otel/sdk v1.44.0 h1:nHYwb9lK+fJPU/dnT6s7W7Z8itMWyqrnVfbheVYrZ58= +go.opentelemetry.io/otel/sdk v1.44.0/go.mod h1:Osuydd3Se74nqjAKxid74N5eC+jfEqfTegHRnq58oK0= +go.opentelemetry.io/otel/sdk/metric v1.44.0 h1:3LlKgI+VjbVsjNRFZJZAJ30WjXC5VkNRks6si09iEfI= +go.opentelemetry.io/otel/sdk/metric v1.44.0/go.mod h1:5B5pMARnXxKhltooO4xUuCBorl65a4EpnTalObqOigA= +go.opentelemetry.io/otel/trace v1.44.0 h1:jxF5CsGYCe74MCRx2X4g7WsY/VBKRqqpNvXlX/6gtIk= +go.opentelemetry.io/otel/trace v1.44.0/go.mod h1:oLl1jrMQAVo6v3GAggN+1VH9VIz9iUSvW53sW1Q8PIE= +golang.org/x/net v0.58.0 h1:ynWG7rqYi4ccpTEuPZ2QGWHktVEM9DMCj9yzDE0Q7To= +golang.org/x/net v0.58.0/go.mod h1:YwCddHnFlT7eLQqVprV19OnhLGtc5xOKgE0RyqgfWAU= +golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs= +golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw= +golang.org/x/text v0.41.0 h1:vz/seA0lnX87Othu2f/0L24RcgrXD9/YFTSuGjj3rH8= +golang.org/x/text v0.41.0/go.mod h1:jvf1O8ajNzZqhSrQBPbutR/EB83Cc0CFrezNQIwbb5M= gonum.org/v1/gonum v0.17.0 h1:VbpOemQlsSMrYmn7T2OUvQ4dqxQXU+ouZFQsZOx50z4= gonum.org/v1/gonum v0.17.0/go.mod h1:El3tOrEuMpv2UdMrbNlKEh9vd86bmQ6vqIcDwxEOc1E= -google.golang.org/genproto/googleapis/rpc v0.0.0-20260406210006-6f92a3bedf2d h1:wT2n40TBqFY6wiwazVK9/iTWbsQrgk5ZfCSVFLO9LQA= -google.golang.org/genproto/googleapis/rpc v0.0.0-20260406210006-6f92a3bedf2d/go.mod h1:4Hqkh8ycfw05ld/3BWL7rJOSfebL2Q+DVDeRgYgxUU8= -google.golang.org/grpc v1.80.0 h1:Xr6m2WmWZLETvUNvIUmeD5OAagMw3FiKmMlTdViWsHM= -google.golang.org/grpc v1.80.0/go.mod h1:ho/dLnxwi3EDJA4Zghp7k2Ec1+c2jqup0bFkw07bwF4= +google.golang.org/genproto/googleapis/rpc v0.0.0-20260526163538-3dc84a4a5aaa h1:mZHHdPZl0dbGHCflZgAq/Q468DWVFcU2whhB2KAo8fk= +google.golang.org/genproto/googleapis/rpc v0.0.0-20260526163538-3dc84a4a5aaa/go.mod h1:4Hqkh8ycfw05ld/3BWL7rJOSfebL2Q+DVDeRgYgxUU8= +google.golang.org/grpc v1.83.2 h1:EManeRomTObA0BU7I8vXgg/78uE5MJ9M8B39EX2WscU= +google.golang.org/grpc v1.83.2/go.mod h1:YPI1hK3kDked6iHvgX3tR0y+nX/qpMFKhPgFsokw1S8= google.golang.org/protobuf v1.36.11 h1:fV6ZwhNocDyBLK0dj+fg8ektcVegBBuEolpbTQyBNVE= google.golang.org/protobuf v1.36.11/go.mod h1:HTf+CrKn2C3g5S8VImy6tdcUvCska2kB7j23XfzDpco= diff --git a/d-engine-server/src/api/embedded_test/embedded_env_test.rs b/d-engine-server/src/api/embedded_test/embedded_env_test.rs index 67173209..8c43d96e 100644 --- a/d-engine-server/src/api/embedded_test/embedded_env_test.rs +++ b/d-engine-server/src/api/embedded_test/embedded_env_test.rs @@ -31,15 +31,29 @@ mod start_data_dir_tests { } /// Opening an existing data directory is idempotent (data is preserved). + /// + /// Uses `start_with` + a timeout-only config so the client's write deadline + /// is not the 50ms `general_raft_timeout_duration_in_ms` default: since + /// #446 a `put` ack waits for a physical fdatasync, which under a loaded + /// test suite (many parallel RocksDB instances) can exceed 50ms. The + /// config's `data_dir` is still ignored β€” the explicit arg wins β€” so this + /// keeps testing exactly the reopen/data-preservation path. #[tokio::test] #[serial] async fn test_start_existing_directory_is_idempotent() { let temp_dir = tempfile::tempdir().expect("tempdir"); let data_dir = temp_dir.path().join("db"); + let config_path = temp_dir.path().join("d-engine.toml"); + std::fs::write( + &config_path, + "[raft]\ngeneral_raft_timeout_duration_in_ms = 3000\n", + ) + .expect("write config"); // First start: write a key { - let engine = EmbeddedEngine::start(&data_dir).await.expect("first start"); + let engine = + EmbeddedEngine::start_with(&data_dir, &config_path).await.expect("first start"); engine.wait_ready(std::time::Duration::from_secs(5)).await.expect("ready"); engine.client().put(b"k".to_vec(), b"v".to_vec()).await.expect("put"); tokio::time::sleep(std::time::Duration::from_millis(50)).await; @@ -48,7 +62,8 @@ mod start_data_dir_tests { // Second start: data must still be there { - let engine = EmbeddedEngine::start(&data_dir).await.expect("second start"); + let engine = + EmbeddedEngine::start_with(&data_dir, &config_path).await.expect("second start"); engine.wait_ready(std::time::Duration::from_secs(5)).await.expect("ready"); let val = engine.client().get_linearizable(b"k".to_vec()).await.expect("get"); assert_eq!(val.as_deref(), Some(b"v".as_ref()), "data must persist"); diff --git a/d-engine-server/src/membership/raft_membership_test.rs b/d-engine-server/src/membership/raft_membership_test.rs index 324ad826..addc1ea9 100644 --- a/d-engine-server/src/membership/raft_membership_test.rs +++ b/d-engine-server/src/membership/raft_membership_test.rs @@ -1192,9 +1192,22 @@ async fn test_health_monitoring_integration() { let (membership, mut zombie_rx) = RaftMembership::::new(1, vec![], config); + // Bind a port and immediately drop the listener β€” the port is now guaranteed + // closed, so the connection attempts below fail deterministically, independent + // of the host's DNS resolver (some resolvers redirect nonexistent hostnames to + // a sinkhole IP instead of failing lookup, which broke this test on such hosts). + let closed_port = { + let listener = std::net::TcpListener::bind("127.0.0.1:0").unwrap(); + listener.local_addr().unwrap().port() + }; + // Add test node membership - .add_learner(100, "invalid.address".to_string(), NodeStatus::Promotable) + .add_learner( + 100, + format!("127.0.0.1:{closed_port}"), + NodeStatus::Promotable, + ) .await .unwrap(); diff --git a/d-engine-server/src/network/grpc/grpc_raft_service.rs b/d-engine-server/src/network/grpc/grpc_raft_service.rs index 15940d48..9993f4dc 100644 --- a/d-engine-server/src/network/grpc/grpc_raft_service.rs +++ b/d-engine-server/src/network/grpc/grpc_raft_service.rs @@ -6,7 +6,6 @@ use crate::Node; use crate::proto_convert; use d_engine_core::InboundEvent; use d_engine_core::MaybeCloneOneshot; -use d_engine_core::MaybeCloneOneshotReceiver; use d_engine_core::RaftOneshot; use d_engine_core::TypeConfig; #[cfg(feature = "watch")] @@ -136,12 +135,24 @@ where Pin> + Send>>; /// Processes a persistent bidirectional AppendEntries stream from the cluster leader. + /// #446: responses are forwarded as soon as each one is ready, not in strict arrival + /// order β€” leader-side match_index/next_index updates are already designed to + /// tolerate out-of-order pipeline responses (`leader_state.rs`, "only advance, + /// never retreat"), so nothing downstream needs strict ordering. Strict FIFO would + /// let one response still waiting on this node's own durable_index (RPO=0) block + /// every later, already-ready response on the same connection β€” including + /// unrelated ones like heartbeats. /// - /// Decouples request ingestion from response emission: - /// - recv task: reads batches from the stream, dispatches each as a `InboundEvent::AppendEntries` - /// (non-blocking between batches) - /// - forwarder task: drains ordered response handles sequentially; ordering is guaranteed - /// by the Raft single-threaded event loop + /// Single task, bounded concurrency: reads a new request only while fewer than + /// `max_pending_append_responses` requests are still in flight, so a stalled fsync + /// bounds memory/task growth instead of growing without limit. + /// + /// Known tradeoff, not an oversight: this is a single task, so a slow/stuck + /// network write (`out_tx.send().await` blocking because the peer isn't reading) + /// also delays reading new requests AND processing the shutdown signal, until the + /// write unblocks or the connection dies. Accepted deliberately β€” if the peer + /// isn't reading responses, there's no useful work to do by reading more requests + /// either; this is legitimate backpressure, not a bug. async fn stream_append_entries( &self, request: tonic::Request>, @@ -157,30 +168,31 @@ where let mut in_stream = request.into_inner(); let event_tx = self.event_tx.clone(); - let ordered_channel_capacity = self.node_config.raft.ordered_channel_capacity; + let max_pending = self.node_config.raft.max_pending_append_responses; let mut shutdown = self.shutdown_signal.clone(); + let node_id = self.node_id; - // Output: ordered ACKs sent back to the leader over the bidi stream - let (out_tx, out_rx) = mpsc::channel::>(128); - - // Ordered queue: response oneshot receivers in FIFO arrival order - let (ordered_tx, mut ordered_rx) = mpsc::channel::< - MaybeCloneOneshotReceiver>, - >(ordered_channel_capacity); + // Output: ACKs sent back to the leader over the bidi stream, in completion order. + // Capacity matches max_pending β€” completed responses can never outnumber + // in-flight requests, so there's no separate number to reason about here. + let (out_tx, out_rx) = mpsc::channel::>(max_pending); - // Recv task: read batches, dispatch to Raft loop without waiting for each ACK. - // Selects on shutdown signal so the task exits immediately on node stop, rather - // than waiting for the next message from the leader. This unblocks serve_with_shutdown - // and allows Arc (and Arc) to be released promptly after stop(). + // Single task: read requests, dispatch to the Raft loop, and forward whichever + // response becomes ready first β€” bounded by `max_pending` in-flight responses. tokio::spawn(async move { use futures::StreamExt; + use futures::stream::FuturesUnordered; + + let mut pending = FuturesUnordered::new(); + let mut inbound_open = true; + loop { tokio::select! { biased; _ = shutdown.changed() => { break; } - result = in_stream.next() => { + result = in_stream.next(), if inbound_open && pending.len() < max_pending => { match result { Some(Ok(req)) => { let (resp_tx, resp_rx) = MaybeCloneOneshot::new(); @@ -188,31 +200,53 @@ where debug!("[stream_append_entries|recv] event_tx closed"); break; } - if ordered_tx.send(resp_rx).await.is_err() { - break; - } + pending.push(async move { + match resp_rx.await { + Ok(Ok(resp)) => Ok(resp), + Ok(Err(status)) => Err(status), + Err(_) => Err(Status::internal("Response channel closed")), + } + }); } Some(Err(e)) => { // Debug: expected when the peer goes away (crash/restart/shutdown), self-heals. debug!("[stream_append_entries|recv] stream error: {:?}", e); - break; + inbound_open = false; + } + None => inbound_open = false, + } + } + Some(result) = pending.next(), if !pending.is_empty() => { + // Observability only β€” behavior doesn't change, the send still + // runs to completion normally. If the peer isn't reading (network + // stall, dead connection with the TCP timeout not yet fired), this + // surfaces it instead of silently blocking with zero signal. + let mut send_fut = std::pin::pin!(out_tx.send(result)); + let mut stuck_logged = false; + let closed = loop { + tokio::select! { + res = &mut send_fut => break res.is_err(), + _ = tokio::time::sleep(Duration::from_secs(5)), if !stuck_logged => { + stuck_logged = true; + error!( + node_id, + "stream_append_entries forwarder stuck sending a \ + response for >5s β€” peer may not be reading \ + (network stall or dead connection)" + ); + metrics::counter!( + "server.grpc.stream_append_entries.forwarder_stuck" + ) + .increment(1); + } } - None => break, + }; + if closed { + break; } } } - } - }); - - // Forwarder task: drain ordered queue sequentially (FIFO guaranteed by Raft loop) - tokio::spawn(async move { - while let Some(resp_rx) = ordered_rx.recv().await { - let result = match resp_rx.await { - Ok(Ok(resp)) => Ok(resp), - Ok(Err(status)) => Err(status), - Err(_) => Err(Status::internal("Response channel closed")), - }; - if out_tx.send(result).await.is_err() { + if !inbound_open && pending.is_empty() { break; } } diff --git a/d-engine-server/src/network/grpc/grpc_raft_service_test.rs b/d-engine-server/src/network/grpc/grpc_raft_service_test.rs index 97d5583f..cc5fa4c5 100644 --- a/d-engine-server/src/network/grpc/grpc_raft_service_test.rs +++ b/d-engine-server/src/network/grpc/grpc_raft_service_test.rs @@ -3,12 +3,15 @@ use std::time::Duration; use crate::ApplyResult; use d_engine_core::AppendResponseWithUpdates; use d_engine_core::InternalEvent; +use d_engine_core::MaybeCloneOneshot; +use d_engine_core::MaybeCloneOneshotReceiver; use d_engine_core::MockElectionCore; use d_engine_core::MockMembership; use d_engine_core::MockRaftLog; use d_engine_core::MockReplicationCore; use d_engine_core::MockTypeConfig; use d_engine_core::RaftNodeConfig; +use d_engine_core::RaftOneshot; use d_engine_core::convert::safe_kv_bytes; use d_engine_proto::client::ClientReadRequest; use d_engine_proto::client::ClientWriteRequest; @@ -237,12 +240,12 @@ async fn test_handle_rpc_services_successfully() { commit_index_update: Some(1), }) }); - replication_handler - .expect_prepare_batch_requests() - .returning(move |payloads, _, _, _, _| { + replication_handler.expect_prepare_batch_requests().returning( + move |payloads, _, _, _, _, _| { li_prepare.fetch_add(payloads.len() as u64, Ordering::Relaxed); Ok(d_engine_core::PrepareResult::default()) - }); + }, + ); let mut election_handler = MockElectionCore::::new(); election_handler .expect_broadcast_vote_requests() @@ -257,7 +260,7 @@ async fn test_handle_rpc_services_successfully() { // Initializing Shutdown Signal let (_graceful_tx, graceful_rx) = watch::channel(()); // Create role channel so the test can inject LogFlushed events directly into the raft loop. - // Commit in single-voter mode is now async (driven by LogFlushed from BufferedRaftLog); + // Commit in single-voter mode is now async (driven by LogFlushed from RaftLogCore); // since this test uses MockRaftLog (no real batch_processor), we send LogFlushed manually. let (internal_event_tx, internal_event_rx) = mpsc::unbounded_channel::(); let test_internal_event_tx = internal_event_tx.clone(); @@ -605,3 +608,268 @@ async fn test_handle_client_scan_not_leader_carries_leader_hint_in_metadata() { Some("http://127.0.0.1:9082") ); } + +/// Historical record, not a regression guard: this is the strict-FIFO forwarder +/// pattern `stream_append_entries` used *before* #446 (one withheld response blocked +/// every later response on the same connection). It reconstructs the old primitives +/// rather than calling production code, because that code no longer exists β€” +/// `stream_append_entries` was rewritten to a bounded, order-tolerant forwarder (see +/// `test_stream_append_entries_does_not_block_ready_response_behind_pending_one` for +/// the real, current behavior). Kept only so a future reader can see what the old +/// failure mode looked like; do not treat this as coverage of current code. +#[tokio::test] +async fn test_ordered_forwarder_head_of_line_blocking() { + let (out_tx, mut out_rx) = mpsc::channel::>(128); + let (ordered_tx, mut ordered_rx) = mpsc::channel::< + MaybeCloneOneshotReceiver>, + >(128); + + // Mirrors grpc_raft_service.rs's forwarder loop. + tokio::spawn(async move { + while let Some(resp_rx) = ordered_rx.recv().await { + let result = match resp_rx.await { + Ok(Ok(resp)) => Ok(resp), + Ok(Err(status)) => Err(status), + Err(_) => Err(tonic::Status::internal("Response channel closed")), + }; + if out_tx.send(result).await.is_err() { + break; + } + } + }); + + // First item: never resolved β€” stands in for a response withheld pending durable_index. + let (_stuck_tx, stuck_rx) = MaybeCloneOneshot::new(); + ordered_tx.send(stuck_rx).await.unwrap(); + + // Second item: already resolved β€” stands in for an unrelated, ready-to-send response + // (e.g. a heartbeat) that arrived right after. + let (ready_tx, ready_rx) = MaybeCloneOneshot::new(); + ordered_tx.send(ready_rx).await.unwrap(); + ready_tx.send(Ok(AppendEntriesResponse::success(1, 1, None))).unwrap(); + + // The second, already-ready response must not be observable yet β€” it's stuck + // behind the first, unresolved one in strict FIFO order. + let blocked = time::timeout(Duration::from_millis(50), out_rx.recv()).await; + assert!( + blocked.is_err(), + "an already-ready response was blocked behind an earlier unresolved one β€” \ + confirms the forwarder is strict FIFO" + ); +} + +/// #446: `stream_append_entries` must not let a response still withheld (durable_index +/// hasn't caught up to what it claims) block a later, unrelated response that's already +/// answerable. Drives the real production method end-to-end β€” not a reconstruction β€” +/// via a synthetic 2-item input stream. +/// +/// request 1 claims index 10 while `durable_index()` is fixed at 5 β€” withheld, +/// queued in `pending_append_acks`, never released in this test. +/// request 2 claims index 3, which is `<= durable_index` β€” answerable immediately. +/// +/// If the forwarder is still strict FIFO, the first item out of the response stream +/// would have to be request 1's (never arrives) β€” this test would time out. If it's +/// the new bounded/order-tolerant forwarder, request 2's response (identifiable by its +/// distinct `last_match.term` marker) comes out first. +#[tokio::test] +async fn test_stream_append_entries_does_not_block_ready_response_behind_pending_one() { + tokio::time::pause(); + let settings = RaftNodeConfig::new().expect("Should succeed to init RaftNodeConfig."); + let mut settings = settings.validate().expect("Validate RaftNodeConfig successfully"); + settings.raft.general_raft_timeout_duration_in_ms = 200; + settings.raft.batching.max_batch_size = 1; + + let mut membership = MockMembership::::new(); + membership.expect_voters().returning(Vec::new); + membership.expect_members().returning(Vec::new); + membership.expect_replication_peers().returning(Vec::new); + membership.expect_get_peers_id_with_condition().returning(|_| vec![]); + + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 0); + raft_log.expect_flush().returning(|| Ok(())); + raft_log.expect_load_hard_state().returning(|| Ok(None)); + raft_log.expect_save_hard_state().returning(|_| Ok(())); + raft_log.expect_last_log_id().returning(|| None); + // Fixed durable frontier: only request 2's claimed index (3) clears it. + raft_log.expect_durable_index().returning(|| 5); + + let call_count = std::sync::Arc::new(std::sync::atomic::AtomicU64::new(0)); + let call_count_clone = call_count.clone(); + let mut replication_handler = MockReplicationCore::::new(); + replication_handler + .expect_check_append_entries_request_is_legal() + .returning(|my_term, _, _| AppendEntriesResponse::success(1, my_term, None)); + replication_handler.expect_handle_append_entries().returning(move |_, _, _| { + let is_first = call_count_clone.fetch_add(1, std::sync::atomic::Ordering::SeqCst) == 0; + let (claimed_index, term_marker) = if is_first { (10, 111) } else { (3, 222) }; + Ok(AppendResponseWithUpdates { + response: AppendEntriesResponse::success( + 1, + term_marker, + Some(LogId { + term: term_marker, + index: claimed_index, + }), + ), + commit_index_update: None, + }) + }); + + let (_graceful_tx, graceful_rx) = watch::channel(()); + let builder = MockBuilder::new(graceful_rx); + let node = builder + .with_raft_log(raft_log) + .with_membership(membership) + .with_replication_handler(replication_handler) + .with_node_config(settings) + .build_node(); + node.set_rpc_ready(true); + + let raft_lock = node.raft_core.clone(); + let _raft_handle = tokio::spawn(async move { + let mut raft = raft_lock.lock().await; + let _ = time::timeout(Duration::from_secs(5), raft.run()).await; + }); + + tokio::time::advance(Duration::from_millis(2)).await; + tokio::time::sleep(Duration::from_millis(2)).await; + + // request 1: prev_log_index=0. request 2: prev_log_index=99 β€” deliberately different + // from request 1's, so merge_append_entries (which only merges contiguous requests) + // can never combine them into a single handle_append_entries call. + let req1 = AppendEntriesRequest { + term: 1, + leader_id: 1, + prev_log_index: 0, + prev_log_term: 0, + entries: vec![], + leader_commit_index: 0, + }; + let req2 = AppendEntriesRequest { + prev_log_index: 99, + ..req1.clone() + }; + let stream = crate::test_utils::create_test_snapshot_stream(vec![req1, req2]); + + let response = node + .stream_append_entries(Request::new(stream)) + .await + .expect("stream_append_entries must accept the request"); + use futures::StreamExt; + let mut out_stream = response.into_inner(); + + let first_out = time::timeout(Duration::from_secs(2), out_stream.next()) + .await + .expect( + "the response for request 2 (already durable) must arrive without waiting for \ + request 1 (withheld) β€” if this times out, the forwarder is still strict FIFO", + ) + .expect("stream must yield an item") + .expect("must be Ok, not a transport error"); + + let last_match = match first_out.result { + Some(d_engine_proto::server::replication::append_entries_response::Result::Success( + success, + )) => success.last_match.expect("success response must carry last_match"), + other => panic!("expected a success response, got {other:?}"), + }; + assert_eq!( + last_match.term, 222, + "the first response observed must be request 2's (marker term=222) β€” request 1 \ + (marker term=111) is still withheld and must not be observed yet, nor block this one" + ); +} + +/// `stream_append_entries` must end the response stream once the inbound stream has +/// closed and every pending response has been delivered. Otherwise the forwarder task +/// (which owns `out_tx`) waits on `shutdown` alone, leaking one task + one channel per +/// closed connection. +/// +/// One request whose claimed index (3) is already `<= durable_index` (5) is answered +/// immediately; the synthetic input stream then ends. After the single response, the +/// output stream must yield `None`. If the loop has no exit for "inbound closed and +/// nothing pending", the second `next()` never completes and the timeout fires. +#[tokio::test] +async fn test_stream_append_entries_closes_response_stream_after_inbound_ends() { + tokio::time::pause(); + let settings = RaftNodeConfig::new().expect("Should succeed to init RaftNodeConfig."); + let mut settings = settings.validate().expect("Validate RaftNodeConfig successfully"); + settings.raft.general_raft_timeout_duration_in_ms = 200; + settings.raft.batching.max_batch_size = 1; + + let mut membership = MockMembership::::new(); + membership.expect_voters().returning(Vec::new); + membership.expect_members().returning(Vec::new); + membership.expect_replication_peers().returning(Vec::new); + membership.expect_get_peers_id_with_condition().returning(|_| vec![]); + + let mut raft_log = MockRaftLog::new(); + raft_log.expect_last_entry_id().returning(|| 0); + raft_log.expect_flush().returning(|| Ok(())); + raft_log.expect_load_hard_state().returning(|| Ok(None)); + raft_log.expect_save_hard_state().returning(|_| Ok(())); + raft_log.expect_last_log_id().returning(|| None); + raft_log.expect_durable_index().returning(|| 5); + + let mut replication_handler = MockReplicationCore::::new(); + replication_handler + .expect_check_append_entries_request_is_legal() + .returning(|my_term, _, _| AppendEntriesResponse::success(1, my_term, None)); + replication_handler.expect_handle_append_entries().returning(|_, _, _| { + Ok(AppendResponseWithUpdates { + response: AppendEntriesResponse::success(1, 1, Some(LogId { term: 1, index: 3 })), + commit_index_update: None, + }) + }); + + let (_graceful_tx, graceful_rx) = watch::channel(()); + let node = MockBuilder::new(graceful_rx) + .with_raft_log(raft_log) + .with_membership(membership) + .with_replication_handler(replication_handler) + .with_node_config(settings) + .build_node(); + node.set_rpc_ready(true); + + let raft_lock = node.raft_core.clone(); + let _raft_handle = tokio::spawn(async move { + let mut raft = raft_lock.lock().await; + let _ = time::timeout(Duration::from_secs(5), raft.run()).await; + }); + + tokio::time::advance(Duration::from_millis(2)).await; + tokio::time::sleep(Duration::from_millis(2)).await; + + let req = AppendEntriesRequest { + term: 1, + leader_id: 1, + prev_log_index: 0, + prev_log_term: 0, + entries: vec![], + leader_commit_index: 0, + }; + let stream = crate::test_utils::create_test_snapshot_stream(vec![req]); + + let response = node + .stream_append_entries(Request::new(stream)) + .await + .expect("stream_append_entries must accept the request"); + use futures::StreamExt; + let mut out_stream = response.into_inner(); + + time::timeout(Duration::from_secs(2), out_stream.next()) + .await + .expect("the single ready response must arrive") + .expect("stream must yield the response") + .expect("must be Ok, not a transport error"); + + let end = time::timeout(Duration::from_secs(2), out_stream.next()).await.expect( + "response stream must end once inbound closed and nothing is pending β€” if this \ + times out, the forwarder task is leaked waiting on shutdown", + ); + assert!( + end.is_none(), + "no further items expected after the only response" + ); +} diff --git a/d-engine-server/src/network/grpc/grpc_transport.rs b/d-engine-server/src/network/grpc/grpc_transport.rs index a8d4a0ef..a633b37f 100644 --- a/d-engine-server/src/network/grpc/grpc_transport.rs +++ b/d-engine-server/src/network/grpc/grpc_transport.rs @@ -308,6 +308,7 @@ where peer_id: u32, membership: Arc>, compress: bool, + send_queue_capacity: usize, ) -> Result { debug!(%peer_id, "Opening persistent bidi replication stream"); @@ -316,8 +317,8 @@ where .await .ok_or(NetworkError::PeerConnectionNotFound(peer_id))?; - // Bounded send channel (capacity 128) provides natural backpressure to the Raft loop. - let (req_tx, req_rx) = mpsc::channel::(128); + // Bounded send channel provides natural backpressure to the Raft loop. + let (req_tx, req_rx) = mpsc::channel::(send_queue_capacity); let req_stream = ReceiverStream::new(req_rx); let mut client = RaftReplicationServiceClient::new(channel); diff --git a/d-engine-server/src/node/builder.rs b/d-engine-server/src/node/builder.rs index 7d66360c..9edadd05 100644 --- a/d-engine-server/src/node/builder.rs +++ b/d-engine-server/src/node/builder.rs @@ -40,6 +40,7 @@ use d_engine_core::NewCommitData; use d_engine_core::Raft; use d_engine_core::RaftCoreHandlers; use d_engine_core::RaftLog; +use d_engine_core::RaftLogCore; use d_engine_core::RaftNodeConfig; use d_engine_core::RaftRole; use d_engine_core::RaftStorageHandles; @@ -78,7 +79,6 @@ use crate::Node; use crate::membership::RaftMembership; use crate::network::grpc; use crate::network::grpc::grpc_transport::GrpcTransport; -use crate::storage::BufferedRaftLog; type StateMachineHandler = ( Arc>>, @@ -348,15 +348,12 @@ where let (internal_event_tx, internal_event_rx) = mpsc::unbounded_channel(); let raft_log = { - let (log, receiver) = BufferedRaftLog::new( + RaftLogCore::new( node_id, - node_config.raft.persistence.clone(), storage_engine.clone(), - ); - - // Start processor and get Arc-wrapped instance. - // Pass internal_event_tx so batch_processor sends InternalEvent::LogFlushed after each fsync. - log.start(receiver, Some(internal_event_tx.clone())) + Some(internal_event_tx.clone()), + node_config.raft.persistence.shutdown_timeout_ms, + ) }; // Peer health channels: transport fires try_send(peer_id) on stream failure/success. @@ -367,9 +364,38 @@ where GrpcTransport::new_with_channels(node_id, peer_failure_tx, peer_success_tx) }); - let snapshot_policy = self.snapshot_policy.take().unwrap_or(LogSizePolicy::new( - node_config.raft.snapshot.max_log_entries_before_snapshot, - )); + let max_log_entries = node_config.raft.snapshot.max_log_entries_before_snapshot; + let retained_log_entries = node_config.raft.snapshot.retained_log_entries; + + // Startup memory-budget line β€” visible on stdout regardless of log setup. + { + const EST_LOG_ENTRY_BYTES: u64 = 512; // small/medium KV write + proto + SkipMap node overhead + let est_mb = max_log_entries + .saturating_add(retained_log_entries) + .saturating_mul(EST_LOG_ENTRY_BYTES) + / (1024 * 1024); + tracing::info!( + node_id, + max_log_entries, + retained_log_entries, + est_log_ram_mb = est_mb, + "Raft log memory estimate: ~{est_mb} MB \ + (({max_log_entries} snapshot-trigger + {retained_log_entries} retained entries) \ + Γ— ~512 B/entry). Excludes snapshot backlog" + ); + if est_mb > 100 { + tracing::warn!( + node_id, + est_log_ram_mb = est_mb, + "in-memory Raft log estimate > 100 MB β€” lower \ + raft.snapshot.max_log_entries_before_snapshot or \ + raft.snapshot.retained_log_entries if RAM-constrained" + ); + } + } + + let snapshot_policy = + self.snapshot_policy.take().unwrap_or(LogSizePolicy::new(max_log_entries)); let shutdown_signal = self.shutdown_signal.clone(); diff --git a/d-engine-server/src/node/builder_test.rs b/d-engine-server/src/node/builder_test.rs index 6c3f85b5..51e0b236 100644 --- a/d-engine-server/src/node/builder_test.rs +++ b/d-engine-server/src/node/builder_test.rs @@ -2,12 +2,10 @@ use std::path::PathBuf; use std::sync::Arc; use d_engine_core::Error; -use d_engine_core::FlushPolicy; use d_engine_core::LogStore; use d_engine_core::MockStateMachine; use d_engine_core::MockStorageEngine; -use d_engine_core::PersistenceConfig; -use d_engine_core::PersistenceStrategy; +use d_engine_core::RaftLogCore; use d_engine_core::RaftNodeConfig; use d_engine_core::StateMachine; use d_engine_core::StorageEngine; @@ -21,7 +19,6 @@ use crate::FileStateMachine; use crate::FileStorageEngine; use crate::node::NodeBuilder; use crate::node::RaftTypeConfig; -use crate::storage::BufferedRaftLog; use crate::test_utils::insert_raft_log; use crate::test_utils::insert_state_machine; @@ -54,20 +51,12 @@ async fn test_set_raft_log_replaces_default() { let mock_storage_engine = Arc::new(FileStorageEngine::new(temp_dir.path().join("storage_engine")).unwrap()); - let (buffered_raft_log, receiver) = - BufferedRaftLog::>::new( - id, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - max_buffered_entries: 1000, - shutdown_timeout_ms: 5000, - }, - mock_storage_engine.clone(), - ); - let buffered_raft_log = buffered_raft_log.start(receiver, None); + let buffered_raft_log = RaftLogCore::>::new( + id, + mock_storage_engine.clone(), + None, + 5000, + ); // diff customization raft_log with orgional one let expected_raft_log_ids = vec![1, 2]; diff --git a/d-engine-server/src/node/type_config/raft_type_config.rs b/d-engine-server/src/node/type_config/raft_type_config.rs index d5271d6c..60c6f785 100644 --- a/d-engine-server/src/node/type_config/raft_type_config.rs +++ b/d-engine-server/src/node/type_config/raft_type_config.rs @@ -6,6 +6,7 @@ use d_engine_core::DefaultStateMachineHandler; use d_engine_core::DefaultStateMachineWriter; use d_engine_core::ElectionHandler; use d_engine_core::LogSizePolicy; +use d_engine_core::RaftLogCore; use d_engine_core::ReplicationHandler; use d_engine_core::StateMachine; use d_engine_core::StorageEngine; @@ -13,7 +14,6 @@ use d_engine_core::TypeConfig; use crate::membership::RaftMembership; use crate::network::grpc::grpc_transport::GrpcTransport; -use crate::storage::BufferedRaftLog; #[derive(Debug)] pub struct RaftTypeConfig @@ -33,7 +33,7 @@ where type SM = SM; - type R = BufferedRaftLog; + type R = RaftLogCore; type TR = GrpcTransport; diff --git a/d-engine-server/src/storage/adaptors/file/file_storage_engine.rs b/d-engine-server/src/storage/adaptors/file/file_storage_engine.rs index 4aec4eba..db63c514 100644 --- a/d-engine-server/src/storage/adaptors/file/file_storage_engine.rs +++ b/d-engine-server/src/storage/adaptors/file/file_storage_engine.rs @@ -350,7 +350,7 @@ impl LogStore for FileLogStore { &self, from_index: u64, new_entries: Vec, - ) -> Result<(), Error> { + ) -> Result { let encoded: Vec> = new_entries.iter().map(|e| e.encode_to_vec()).collect(); let new_last = { @@ -378,7 +378,7 @@ impl LogStore for FileLogStore { }; self.last_index.store(new_last, Ordering::SeqCst); - Ok(()) + Ok(new_last) } fn is_write_durable(&self) -> bool { diff --git a/d-engine-server/src/storage/adaptors/rocksdb/mod.rs b/d-engine-server/src/storage/adaptors/rocksdb/mod.rs index da028955..81ac2866 100644 --- a/d-engine-server/src/storage/adaptors/rocksdb/mod.rs +++ b/d-engine-server/src/storage/adaptors/rocksdb/mod.rs @@ -41,7 +41,8 @@ pub(super) fn base_db_options() -> Options { // This is NOT fdatasync β€” it spreads dirty page writeback to avoid IO spikes, // but provides no durability guarantee. Pairs well with Level 3 (flush_wal) if added. opts.set_wal_bytes_per_sync(1024 * 1024); - opts.set_max_background_jobs(4); + opts.set_manual_wal_flush(true); + opts.set_max_background_jobs(2); opts.set_max_open_files(5000); opts.set_use_direct_io_for_flush_and_compaction(true); opts.set_use_direct_reads(true); diff --git a/d-engine-server/src/storage/adaptors/rocksdb/rocksdb_storage_engine.rs b/d-engine-server/src/storage/adaptors/rocksdb/rocksdb_storage_engine.rs index 624c901e..e16b16ef 100644 --- a/d-engine-server/src/storage/adaptors/rocksdb/rocksdb_storage_engine.rs +++ b/d-engine-server/src/storage/adaptors/rocksdb/rocksdb_storage_engine.rs @@ -185,7 +185,6 @@ impl RocksDBLogStore { #[async_trait] impl LogStore for RocksDBLogStore { - #[instrument(skip(self, entries))] async fn persist_entries( &self, entries: Vec, @@ -197,11 +196,15 @@ impl LogStore for RocksDBLogStore { let mut batch = WriteBatch::default(); let mut max_index = 0; + let mut value_buf = Vec::new(); for entry in entries { let key = Self::index_to_key(entry.index); - let value = entry.encode_to_vec(); - batch.put_cf(&cf, key, value); + value_buf.clear(); + entry + .encode(&mut value_buf) + .map_err(|e| StorageError::SerializationError(e.to_string()))?; + batch.put_cf(&cf, key, &value_buf); max_index = max_index.max(entry.index); } @@ -313,7 +316,7 @@ impl LogStore for RocksDBLogStore { self.db.write(&batch).map_err(|e| StorageError::DbError(e.to_string()))?; // Persist purge boundary to META_CF for crash recovery. - // BufferedRaftLog::new() reads this on restart to restore last_purged_index/term + // RaftLogCore::new() reads this on restart to restore last_purged_index/term // so that entry_term(last_purged_index) returns the correct term after restart. if let Some(cf_meta) = self.db.cf_handle(META_CF) { let encoded = cutoff_index.encode_to_vec(); @@ -383,7 +386,7 @@ impl LogStore for RocksDBLogStore { &self, from_index: u64, new_entries: Vec, - ) -> Result<(), Error> { + ) -> Result { let cf = self .db .cf_handle(LOG_CF) @@ -404,7 +407,7 @@ impl LogStore for RocksDBLogStore { self.db.write(&batch).map_err(|e| StorageError::DbError(e.to_string()))?; self.last_index.store(new_last_index, Ordering::SeqCst); - Ok(()) + Ok(new_last_index) } fn is_write_durable(&self) -> bool { diff --git a/d-engine-server/src/storage/buffered/mod.rs b/d-engine-server/src/storage/buffered/mod.rs deleted file mode 100644 index d4a220ec..00000000 --- a/d-engine-server/src/storage/buffered/mod.rs +++ /dev/null @@ -1,2 +0,0 @@ -// Re-export BufferedRaftLog from core (now lives in d-engine-core) -pub use d_engine_core::BufferedRaftLog; diff --git a/d-engine-server/src/storage/mod.rs b/d-engine-server/src/storage/mod.rs index 432eb9e1..66e929a7 100644 --- a/d-engine-server/src/storage/mod.rs +++ b/d-engine-server/src/storage/mod.rs @@ -19,11 +19,9 @@ /// This module contains pluggable storage backends including file-based /// and RocksDB-based state machines that implement the `StateMachine` trait. pub mod adaptors; -mod buffered; mod lease; pub use adaptors::*; -pub use buffered::*; // Re-export Lease trait from core for convenience pub use d_engine_core::Lease; pub use lease::TtlLease; diff --git a/d-engine-server/src/test_utils/integration/mod.rs b/d-engine-server/src/test_utils/integration/mod.rs index 838f4c9c..f7c2da0c 100644 --- a/d-engine-server/src/test_utils/integration/mod.rs +++ b/d-engine-server/src/test_utils/integration/mod.rs @@ -45,12 +45,10 @@ use std::sync::Arc; use bytes::Bytes; use d_engine_core::DefaultStateMachineHandler; use d_engine_core::ElectionHandler; -use d_engine_core::FlushPolicy; use d_engine_core::LogSizePolicy; use d_engine_core::MockStateMachine; -use d_engine_core::PersistenceConfig; -use d_engine_core::PersistenceStrategy; use d_engine_core::RaftLog; +use d_engine_core::RaftLogCore; use d_engine_core::RaftNodeConfig; use d_engine_core::ReplicationHandler; use d_engine_core::StateMachine; @@ -77,7 +75,6 @@ use crate::FileStorageEngine; use crate::membership::RaftMembership; use crate::network::grpc::grpc_transport::GrpcTransport; use crate::node::RaftTypeConfig; -use crate::storage::BufferedRaftLog; /// Complete testing environment for Raft consensus algorithm integration tests. /// @@ -178,19 +175,7 @@ pub fn setup_raft_components( storage_engine.log_store().reset_sync().unwrap(); } - let (buffered_raft_log, receiver) = BufferedRaftLog::new( - id, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - max_buffered_entries: 10000, - shutdown_timeout_ms: 5000, - }, - storage_engine.clone(), - ); - let buffered_raft_log = buffered_raft_log.start(receiver, None); + let buffered_raft_log = RaftLogCore::new(id, storage_engine.clone(), None, 5000); let mock_state_machine = mock_state_machine(); let last_applied_pair = mock_state_machine.last_applied(); diff --git a/d-engine-server/tests/cas_operations/leader_failover_cas_standalone.rs b/d-engine-server/tests/cas_operations/leader_failover_cas_standalone.rs index 85f75c28..d832a514 100644 --- a/d-engine-server/tests/cas_operations/leader_failover_cas_standalone.rs +++ b/d-engine-server/tests/cas_operations/leader_failover_cas_standalone.rs @@ -15,6 +15,7 @@ use crate::common::create_node_config; use crate::common::get_available_ports; use crate::common::node_config; use crate::common::start_node; +use crate::common::wait_for_stable_leader; /// Test CAS operation behavior during leader failover (Standalone/gRPC mode) /// @@ -54,22 +55,25 @@ async fn test_leader_failover_cas_standalone() -> Result<(), ClientApiError> { info!("Starting 3-node cluster for CAS failover test (gRPC mode)"); for (i, port) in ports.iter().enumerate() { let node_data_dir = temp_dir.path().join(format!("node{}", i + 1)); - let (graceful_tx, node_handle) = start_node( - &node_data_dir, - node_config( - &create_node_config( - (i + 1) as u64, - *port, - ports, - &node_data_dir.to_string_lossy(), - &log_dir, - ) - .await, - ), - None, - None, - ) - .await?; + let mut node_cfg = node_config( + &create_node_config( + (i + 1) as u64, + *port, + ports, + &node_data_dir.to_string_lossy(), + &log_dir, + ) + .await, + ); + // TODO(#428): widen the election timeout for this test only. After the leader is + // killed, the 2 surviving nodes re-elect; with the default 300ms min, slow CI can + // livelock (the new leader's 100ms heartbeat slips past the follower's 300ms + // timeout, so the follower votes it out and split-vote cascades). 3000/6000 gives + // the leader time to establish β€” same rationale as `create_rejoin_node_config`. + // Revert once #428 (leader lease) lands. + node_cfg.raft.election.election_timeout_min = 3000; + node_cfg.raft.election.election_timeout_max = 6000; + let (graceful_tx, node_handle) = start_node(&node_data_dir, node_cfg, None, None).await?; ctx.graceful_txs.push(graceful_tx); ctx.node_handles.push(node_handle); } @@ -118,100 +122,15 @@ async fn test_leader_failover_cas_standalone() -> Result<(), ClientApiError> { info!("CAS result during leader stop: {:?}", cas_result); // Expected: timeout, NOT_LEADER, or UNAVAILABLE error - // Phase 2 + 3: Wait for a stable leader AND verify lock state consistency. - // - // Why these two phases are merged into a single retry loop: - // - // After node 3 crashes, the surviving nodes may undergo cascading elections - // before settling on a stable leader. A pattern observed in both CI and local: - // - // Node 1 & 2 both start elections (split vote, multiple rounds) - // β†’ Node A wins term N and becomes leader ← refresh() sees this and returns - // β†’ Node B's election timer fires, starts term N+1 election - // β†’ Node A receives term N+1, MUST step down (Raft protocol) - // β†’ Phase 3 read hits node A (now a follower) β†’ "Not leader" + // Phase 2 + 3: Wait for a stable leader, then verify lock state consistency. // - // The key insight: checking leader_id consistency across two refresh() calls - // does NOT reliably detect this. Cluster metadata (current_leader_id) reflects - // what each node *last committed as leader*, not the live Raft state. A node - // that just stepped down may still appear as "leader" in other nodes' metadata - // until the new leader's noop is replicated. - // - // The only authoritative check is: can the leader actually serve a read right now? - // - // Fix: merge Phase 2 and Phase 3 into a refreshβ†’read loop. refresh() discovers - // the best-known leader; the subsequent get() is the live proof that the leader - // is active. On StaleOperation the loop retries immediately. This converges once - // the cluster elects a stable leader that can serve requests end-to-end. - // refresh() internally handles cluster_ready_timeout, so the loop is bounded. + // wait_for_stable_leader() is the authoritative check for "can the leader + // actually serve a read right now?" β€” see its own doc comment for why + // cheaper checks (e.g. leader_id consistency across refresh() calls) don't + // reliably detect a cascading election still settling after node 3 crashes. info!("Phase 2+3: Waiting for stable leader and verifying lock state consistency"); - let lock_value = loop { - client.refresh(None).await?; - let new_leader_id = client - .get_leader_id() - .await? - .expect("Leader must be known after successful refresh"); - info!("Candidate leader: node {}", new_leader_id); - - match client.get(lock_key).await { - Ok(value) => { - // Read succeeded β€” this leader is actively serving requests. - info!("Stable leader confirmed: node {}", new_leader_id); - break value; - } - Err(ClientApiError::Business { - code: ErrorCode::StaleOperation, - .. - }) => { - // The leader changed between refresh() and the read RPC. - // A cascading election is still in progress β€” refresh and retry. - info!( - "Node {} is no longer leader (cascading election in progress), retrying", - new_leader_id - ); - continue; - } - Err(ClientApiError::Network { - code: ErrorCode::ConnectionTimeout, - .. - }) => { - // Node 1 just won the election but node 2 immediately started its own - // candidacy (cascading elections). A linearizable read requires a - // heartbeat-quorum ACK from node 2; while node 2 is a candidate it - // ignores node 1's heartbeats. The pending read sits on the server - // until node 2 steps down, which can exceed the client's default - // request_timeout (3 s), causing Code::Cancelled / "Timeout expired". - // - // The cluster will stabilise once node 2 receives node 1's heartbeat - // with the winning term and steps down. Retry refresh() + get() so - // we wait for a leader that can actually serve requests end-to-end. - info!( - "get() timed out waiting for leader {} to achieve read quorum \ - (cascading election still settling), retrying", - new_leader_id - ); - tokio::time::sleep(Duration::from_millis(100)).await; - continue; - } - Err(ClientApiError::Network { - code: ErrorCode::NotLeader, - .. - }) => { - // The node `refresh()` just picked as leader had already stepped down - // by the time this read reached it (same cascading-election window as - // above, one term later). Transient β€” retry; the next refresh() will - // pick up the new leader. - info!( - "Node {} is no longer leader (stepped down before serving read), retrying", - new_leader_id - ); - continue; - } - Err(e) => { - return Err(e); - } - } - }; + wait_for_stable_leader(&client).await?; + let lock_value = client.get(lock_key).await?; match lock_value { None => { diff --git a/d-engine-server/tests/common/mod.rs b/d-engine-server/tests/common/mod.rs index 2d10210a..f82e8cd9 100644 --- a/d-engine-server/tests/common/mod.rs +++ b/d-engine-server/tests/common/mod.rs @@ -11,9 +11,7 @@ use d_engine_core::alias::SOF; use d_engine_core::client::ErrorCode; use d_engine_core::config::BackoffPolicy; use d_engine_core::config::ElectionConfig; -use d_engine_core::config::FlushPolicy; use d_engine_core::config::PersistenceConfig; -use d_engine_core::config::PersistenceStrategy; use d_engine_core::config::RaftConfig; use d_engine_core::config::RaftNodeConfig; use d_engine_core::config::SnapshotConfig; @@ -123,9 +121,8 @@ pub async fn create_node_config( {initial_cluster_entries} ] - [raft.persistence] - strategy = "MemFirst" - flush_policy = {{ Batch = {{ threshold = 100, idle_flush_interval_ms = 1 }} }} + [raft] + general_raft_timeout_duration_in_ms = 5000 [raft.election] election_timeout_min = 300 @@ -173,10 +170,6 @@ pub async fn create_node_config_with_role( {initial_cluster_entries} ] - [raft.persistence] - strategy = "MemFirst" - flush_policy = {{ Batch = {{ threshold = 1, idle_flush_interval_ms = 1 }} }} - [raft.election] election_timeout_min = 300 election_timeout_max = 3000 @@ -218,10 +211,6 @@ pub fn node_config(cluster_toml: &str) -> RaftNodeConfig { ..Default::default() }, persistence: PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, ..Default::default() }, election: ElectionConfig { @@ -560,12 +549,15 @@ election_timeout_max = 6000 /// /// Equivalent to the Phase 2+3 retry loop in `leader_failover_cas_standalone`. /// -/// Bounded to 30s: if elections never stabilize (a real regression, not just a slow +/// Bounded to 45s: if elections never stabilize (a real regression, not just a slow /// CI box), this returns the last transient error instead of hanging the test forever /// with no diagnostic β€” an unbounded loop here turns "election liveness broke" into a /// bare test-runner timeout with no indication of what was actually still failing. +/// +/// TODO(#428): 45s (was 30s) gives the 2-node re-election room to converge with the +/// widened 3000/6000 election timeout in `node_config`. Tighten once leader lease lands. pub async fn wait_for_stable_leader(client: &Client) -> Result<(), ClientApiError> { - const DEADLINE: Duration = Duration::from_secs(30); + const DEADLINE: Duration = Duration::from_secs(45); let deadline = tokio::time::Instant::now() + DEADLINE; let mut last_err: Option = None; @@ -613,6 +605,20 @@ pub async fn wait_for_stable_leader(client: &Client) -> Result<(), ClientApiErro tokio::time::sleep(Duration::from_millis(100)).await; continue; } + // New leader hasn't committed its own term's noop entry yet (Raft + // safety requirement before serving reads) β€” same cascading-election + // window as the arms above, just a different stage of it. Transient. + Err( + ref e @ ClientApiError::Business { + code: ErrorCode::ClusterUnavailable, + ref message, + .. + }, + ) if message.contains("noop not committed") => { + last_err = Some(e.clone()); + tokio::time::sleep(Duration::from_millis(100)).await; + continue; + } Err(e) => return Err(e), } } diff --git a/d-engine-server/tests/integration_test.rs b/d-engine-server/tests/integration_test.rs index 2432e2c6..af32b736 100644 --- a/d-engine-server/tests/integration_test.rs +++ b/d-engine-server/tests/integration_test.rs @@ -22,7 +22,9 @@ mod leader_election; mod replication_and_sync; // Storage layer integration tests -mod storage_buffered_raft_log; +// mod storage_buffered_raft_log; + +mod storage_raft_log; #[cfg(feature = "rocksdb")] mod readonly_and_learner_mode; diff --git a/d-engine-server/tests/leader_election/leader_election_log_term_index_standalone.rs b/d-engine-server/tests/leader_election/leader_election_log_term_index_standalone.rs index 9ffa1d67..bfc32d5b 100644 --- a/d-engine-server/tests/leader_election/leader_election_log_term_index_standalone.rs +++ b/d-engine-server/tests/leader_election/leader_election_log_term_index_standalone.rs @@ -1,7 +1,7 @@ //! Case 1: Verify that the Raft leader is elected based on the highest log Term and Index, not -//! merely the number of log entries β€” and that node 2 (the eventual leader) explicitly rejects -//! at least one vote request from a worse-log peer along the way, rather than just happening to -//! win the timeout race. +//! merely the number of log entries β€” and that node 2 (the eventual leader) rejects a vote +//! request from a worse-log peer, proving Raft's election-safety guarantee (Β§5.4.1) is enforced +//! *on the wire*, not just that node 2 happened to win the timeout race. //! //! Scenario: //! @@ -10,29 +10,41 @@ //! 3. Node B appends 8 log entries with Term=3 (higher term). //! 4. Node C is a new node with no logs. //! 5. Trigger a leader election (all three nodes use the same default randomized timeout β€” -//! see the note on the rejection assertion below for why no timing tricks are needed). +//! the rejection is no longer sourced from this race; see below). //! //! Expected Result: //! //! - Node B becomes the leader because its logs have the highest Term (Term=3), even though it has //! fewer entries than Node A. //! - Nodes A and C recognize B as the leader. -//! - Node B must have sent at least one `VoteResponse { vote_granted: false, .. }` before winning β€” -//! proving Raft's election-safety guarantee (Β§5.4.1) is enforced *on the wire*, not just that -//! node B happened to win the randomized timeout race. `election_handler::check_vote_request_is_legal` -//! (the function that decides this) already has thorough pure-logic unit test coverage β€” see -//! `election_handler_test.rs` β€” but until this assertion, nothing exercised it end-to-end over -//! real gRPC in a live multi-node cluster. +//! - Node B rejects a synthetic vote request built with node A's real log profile (more entries, +//! lower term) β€” proving the comparison is term-then-index, not entry count. //! -//! Note: earlier versions of this test tried to force node B into a passive role by giving it a -//! much longer election timeout than A/C, so a rejection would be guaranteed to happen before B's -//! own timer ever fired. That backfired: A and C (whose logs disagree with each other too β€” A has -//! no log entries in common with C's empty log, different terms) ended up in a prolonged -//! split-vote livelock (term counters observed climbing past 20) with neither able to reach a -//! majority without B. The uniform default timeout (used below) avoids that: whichever node times -//! out first almost immediately triggers *some* rejection (the exact reason β€” stale term vs. stale -//! log β€” varies run to run, which is why the assertion below checks for the rejection outcome, -//! not a specific internal reason). +//! ## Why the rejection is injected directly, not observed from the natural election +//! +//! Earlier versions of this test relied on the natural 3-node election race to produce the +//! rejection organically: whichever of A/C's randomized timers fired first would ask node B for +//! a vote and (having a worse log) get rejected. Two problems with that: +//! +//! 1. It's not guaranteed. If node B's own timer happens to fire first, it becomes candidate +//! itself and never receives a vote request to reject β€” the assertion would then have nothing +//! to observe, purely due to timer luck (observed in CI: ~1/9 runs). +//! 2. An earlier attempt to fix (1) by giving node B a much longer timeout than A/C β€” forcing A/C +//! to go first β€” backfired: A and C's logs disagree with each other too (A has no entries in +//! common with C's empty log, different terms), and they ended up in a prolonged split-vote +//! livelock (term counters observed climbing past 20) with neither able to reach a majority +//! without B. +//! +//! Fix: stop depending on the natural race for the rejection half of the proof. The test itself +//! opens a raw `RaftElectionServiceClient` to node B's real gRPC endpoint (the same generated +//! client real peers use β€” nothing mocked) and sends a `VoteRequest` carrying node A's exact log +//! signature (last_log_index=10, last_log_term=2) under a fake `candidate_id`, at `term=10` β€” +//! comfortably above any term the natural 3-node election reaches in this test's runtime +//! (observed: single digits) β€” so the rejection is unambiguously the log check, not term +//! staleness. This is sent immediately after the nodes start, before node B could plausibly have +//! become leader itself, so there is no leader-step-down side effect from the injected term bump. +//! The natural election (who the cluster actually elects) is untouched and still runs exactly as +//! before. use crate::client_manager::ClientManager; use crate::common::TestContext; @@ -48,8 +60,8 @@ use crate::common::prepare_storage_engine; use crate::common::reset; use crate::common::start_node; use d_engine_core::ClientApiError; -use d_engine_core::capture_logs_globally_filtered; -use d_engine_core::logs_contain_globally; +use d_engine_proto::server::election::VoteRequest; +use d_engine_proto::server::election::raft_election_service_client::RaftElectionServiceClient; use std::time::Duration; use tracing::debug; @@ -60,14 +72,6 @@ const ELECTION_CASE1_DIR: &str = "election/case1"; async fn test_leader_election_based_on_log_term_and_index() -> Result<(), ClientApiError> { // enable_logger(); - // Captures `election_handler`'s debug logs across every node's tokio task - // (not just this test's own thread) so we can assert, below, that a stale - // vote request was actually rejected on the wire β€” not just that node 2 - // eventually won. - let logs = capture_logs_globally_filtered( - "info,d_engine_core=debug,d_engine_server=debug,h2=off,tonic=warn,hyper=warn", - ); - debug!("...test_leader_election_based_on_log_term_and_index..."); reset(ELECTION_CASE1_DIR).await?; @@ -120,6 +124,52 @@ async fn test_leader_election_based_on_log_term_and_index() -> Result<(), Client ctx.node_handles.push(node_handle); } + // Prove node 2 rejects a worse-log candidate on the wire (Β§5.4.1) β€” see the + // module doc for why this is injected directly instead of relied on from + // the natural election race. Sent now, immediately after node startup and + // before the readiness sleep below, so node 2 cannot yet be leader (no + // step-down side effect from the term bump this induces). + let node2_addr = format!("http://127.0.0.1:{}", ports[1]); + let inject_deadline = std::time::Instant::now() + Duration::from_secs(10); + let vote_response = loop { + let attempt = async { + let mut client = RaftElectionServiceClient::connect(node2_addr.clone()) + .await + .map_err(|e| tonic::Status::unavailable(e.to_string()))?; + client + .request_vote(tonic::Request::new(VoteRequest { + term: 10, // comfortably above any term the natural election reaches here + candidate_id: 99, // synthetic candidate β€” not a real cluster member + last_log_index: 10, // node A's real log profile: more entries... + last_log_term: 2, // ...but a lower term than node 2's (3) + })) + .await + } + .await; + + match attempt { + Ok(resp) => break resp.into_inner(), + Err(e) if std::time::Instant::now() < inject_deadline => { + debug!("synthetic vote request to node 2 not ready yet, retrying: {e:?}"); + tokio::time::sleep(Duration::from_millis(20)).await; + } + Err(e) => panic!("synthetic vote request to node 2 never succeeded: {e:?}"), + } + }; + assert!( + !vote_response.vote_granted, + "node 2 must reject a candidate with a lower log term (2) even though \ + that candidate has more log entries (10) than node 2's own log \ + (index=8, term=3) β€” Raft's election-safety guarantee (Β§5.4.1) compares \ + term first, then index; entry count never matters" + ); + assert_eq!( + (vote_response.last_log_index, vote_response.last_log_term), + (8, 3), + "rejection response must carry node 2's own real log signature, \ + confirming node 2 itself (not some other node) is the responder" + ); + tokio::time::sleep(Duration::from_secs(WAIT_FOR_NODE_READY_IN_SEC)).await; // Verify cluster is ready @@ -151,27 +201,6 @@ async fn test_leader_election_based_on_log_term_and_index() -> Result<(), Client let leader_id = client_manager.list_leader_id().await.unwrap(); assert_eq!(leader_id, Some(2)); - // Node 2's log signature (index 8, term 3, seeded above) uniquely identifies - // it as the responder. Matching on `vote_granted: false` together with that - // signature proves node 2 itself explicitly rejected a vote request at some - // point β€” not just that it eventually won. This deliberately does not pin - // down *which* of check_vote_request_is_legal's checks (stale term vs. - // stale log) caused the rejection: both are valid, and which one fires - // first depends on exactly how far node 1/3's term has climbed by the time - // their request reaches node 2, which varies run to run. - assert!( - logs_contain_globally( - &logs, - "vote_granted: false, last_log_index: 8, last_log_term: 3" - ), - "expected node 2 to have explicitly rejected at least one peer's vote \ - request (VoteResponse{{vote_granted: false}}) before being elected β€” \ - this is the Raft election-safety guarantee (Β§5.4.1) that stops a node \ - with an incomplete log from ever becoming leader; seeing this only via \ - `leader_id == Some(2)` above would also be consistent with node 2 \ - simply winning the timeout race without the rejection path ever firing" - ); - // Clean up ctx.shutdown().await } diff --git a/d-engine-server/tests/snapshot_and_recovery/snapshot_generation_standalone.rs b/d-engine-server/tests/snapshot_and_recovery/snapshot_generation_standalone.rs index afd51efe..84f3b8db 100644 --- a/d-engine-server/tests/snapshot_and_recovery/snapshot_generation_standalone.rs +++ b/d-engine-server/tests/snapshot_and_recovery/snapshot_generation_standalone.rs @@ -114,7 +114,18 @@ async fn test_snapshot_scenario() -> Result<(), ClientApiError> { _ => None, }; - let node_config = node_config(&config); + let mut node_config = node_config(&config); + // `node_config()`'s default election_timeout_min (300ms) is tuned for fast + // *first*-election tests, not this one: this test hard-codes which node + // wins (node 3, pre-seeded with the longest log) and everything after + // that depends on node 3 staying leader. Under CI CPU contention a + // heartbeat can land >300ms late even after the cluster reports ready, + // triggering a spurious re-election that hands leadership to a + // different node β€” the snapshot then never appears where this test + // looks for it (root cause of the observed CI flake). Widen the window + // so a late heartbeat doesn't cost us the election. + node_config.raft.election.election_timeout_min = 1500; + node_config.raft.election.election_timeout_max = 6000; let (graceful_tx, node_handle) = start_node(&node_data_dir, node_config, Some(state_machine), raft_log).await?; @@ -132,11 +143,22 @@ async fn test_snapshot_scenario() -> Result<(), ClientApiError> { println!("[test_snapshot_scenario] Cluster started. Running tests..."); - sleep(Duration::from_secs(3)).await; - - // Verify snapshot file exists on leader (node 3) + // Poll instead of a fixed sleep + single assert: a fixed window races CI + // scheduling jitter (see the election_timeout widening above for why that + // jitter matters here) β€” poll until the snapshot appears or we genuinely + // time out, rather than guessing a duration that "usually" works. let snapshot_path = format!("{SNAPSHOT_CASE1_DATA_DIR}/cs/3/snapshots"); - assert!(check_path_contents(&snapshot_path).unwrap_or(false)); + let snapshot_deadline = tokio::time::Instant::now() + Duration::from_secs(15); + loop { + if check_path_contents(&snapshot_path).unwrap_or(false) { + break; + } + assert!( + tokio::time::Instant::now() < snapshot_deadline, + "snapshot did not appear at {snapshot_path} within 15s" + ); + sleep(Duration::from_millis(200)).await; + } // Verify state machine data via client API (snapshot has been applied to leader) let mut client_manager = ClientManager::new(&create_bootstrap_urls(ports)).await?; diff --git a/d-engine-server/tests/snapshot_and_recovery/snapshot_transfer_does_not_block_apply_embedded.rs b/d-engine-server/tests/snapshot_and_recovery/snapshot_transfer_does_not_block_apply_embedded.rs index b54a1080..f5117e27 100644 --- a/d-engine-server/tests/snapshot_and_recovery/snapshot_transfer_does_not_block_apply_embedded.rs +++ b/d-engine-server/tests/snapshot_and_recovery/snapshot_transfer_does_not_block_apply_embedded.rs @@ -91,9 +91,20 @@ async fn test_snapshot_transfer_does_not_block_apply() -> Result<(), Box RETAINED_LOGS=8 (checked below), no node's log can cross a - // purge boundary before these writes land, so nothing earlier in the buffer can - // satisfy the match below β€” no extra gate is needed to make `since` safe. - // // This used to gate `since` on a `wait_for_snapshot` directory scan for the leader's // `.gz` file first, on the theory that "file exists" proves the snapshot is built. // It doesn't: `compress_directory` (default_state_machine_handler.rs) calls @@ -196,7 +202,6 @@ push_queue_size = 1 // made this test flake under CI load: RocksDB checkpoint export + tar/gzip // compression + metadata persist is genuinely sequential disk+CPU work that slows // down under contention. - let since = logs.lock().unwrap().len(); // The retained-log purge boundary β€” the actual signal this test needs, not just "a // snapshot file exists" (see comment above for why those differ). Emitted by @@ -208,14 +213,32 @@ push_queue_size = 1 // SNAPSHOT_THRESHOLD=64 > RETAINED_LOGS=8, the earliest possible snapshot on this // cluster already has last_included.index >= 64, so purge_upto_index is always > 0 // by the time this log line can appear at all. + // 60 x 500ms = 30s, not 15s: this test lives in the `multi-node-cluster-local` + // nextest group (throttled but not serialized, see .config/nextest.toml), and the + // log line polled below shares a process-global Mutex> with every other + // concurrently-running test in this binary (see log_capture.rs). Under full-suite + // load, RocksDB checkpoint export + tar/gzip (genuinely sequential CPU+disk work, + // see comment above) plus that shared-mutex contention can push real completion past + // 15s even though nothing is actually wrong β€” same root cause already documented in + // stress_test.rs's 30s bound. + // Scoped to the leader's node id: the capture buffer is process-global, so a follower's + // (or another test's) purge line must not satisfy this wait. + let leader_purge_needle = format!("node_id={} purge_upto_index=", leader_info.leader_id); let mut purged = false; - for _ in 0..30 { - if logs_contain_globally_since(&logs, since, "purge_upto_index=") { + for _ in 0..60 { + if logs_contain_globally_since(&logs, since, &leader_purge_needle) { purged = true; break; } tokio::time::sleep(Duration::from_millis(500)).await; } + if !purged { + eprintln!("=== DEBUG: captured logs since baseline writes ==="); + for line in logs.lock().unwrap()[since..].iter() { + eprintln!("{line}"); + } + eprintln!("=== END DEBUG ==="); + } assert!( purged, "Leader never logged a completed log purge β€” node 4 joining now would prove \ diff --git a/d-engine-server/tests/storage_buffered_raft_log/crash_recovery_test.rs b/d-engine-server/tests/storage_raft_log/crash_recovery_test.rs similarity index 52% rename from d-engine-server/tests/storage_buffered_raft_log/crash_recovery_test.rs rename to d-engine-server/tests/storage_raft_log/crash_recovery_test.rs index de85eebd..1b955773 100644 --- a/d-engine-server/tests/storage_buffered_raft_log/crash_recovery_test.rs +++ b/d-engine-server/tests/storage_raft_log/crash_recovery_test.rs @@ -1,6 +1,6 @@ -//! Crash recovery integration tests for BufferedRaftLog +//! Crash recovery integration tests for RaftLogCore //! -//! These tests verify BufferedRaftLog behavior with real disk persistence +//! These tests verify RaftLogCore behavior with real disk persistence //! and crash recovery semantics using FileStorageEngine. use std::path::PathBuf; @@ -8,9 +8,7 @@ use std::sync::Arc; use std::time::Duration; use bytes::Bytes; -use d_engine_core::{ - BufferedRaftLog, FlushPolicy, PersistenceConfig, PersistenceStrategy, RaftLog, -}; +use d_engine_core::{RaftLog, RaftLogCore}; use d_engine_proto::common::{Entry, EntryPayload}; use d_engine_server::{FileStateMachine, FileStorageEngine, node::RaftTypeConfig}; use tokio::time::sleep; @@ -20,13 +18,7 @@ use super::TestContext; #[tokio::test] async fn test_crash_recovery() { // Create and populate storage - let original_ctx = TestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_crash_recovery", - ); + let original_ctx = TestContext::new("test_crash_recovery"); // Append an entry original_ctx @@ -39,19 +31,19 @@ async fn test_crash_recovery() { .await .unwrap(); - // Ensure the entry is persisted for DiskFirst strategy + // Ensure the entry is persisted original_ctx.raft_log.flush().await.unwrap(); // Recover from the same storage (simulating restart) let recovered_ctx = original_ctx.recover_from_crash(); - // Graceful shutdown: close() joins the IO thread before returning, + // Graceful shutdown: close() flushes remaining data before returning, // preventing Tokio runtime shutdown panics. original_ctx.close().await; sleep(Duration::from_millis(50)).await; // Allow recovery - // Verify recovery - for DiskFirst, entries should be immediately durable + // Verify recovery - entries should be immediately durable assert_eq!(recovered_ctx.raft_log.durable_index(), 1); // The entry should be available @@ -64,13 +56,7 @@ async fn test_crash_recovery() { #[tokio::test] async fn test_crash_recovery_with_multiple_entries() { // Create and populate storage - let original_ctx = TestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_crash_recovery_with_multiple_entries", - ); + let mut original_ctx = TestContext::new("test_crash_recovery_with_multiple_entries"); // Append multiple entries for i in 1..=5 { @@ -87,8 +73,9 @@ async fn test_crash_recovery_with_multiple_entries() { .unwrap(); } - // Ensure all entries are persisted for DiskFirst strategy + // Ensure all entries are persisted original_ctx.raft_log.flush().await.unwrap(); + original_ctx.drain_fsync_completions(); // Verify all entries are in memory and durable assert_eq!(original_ctx.raft_log.durable_index(), 5); @@ -97,7 +84,7 @@ async fn test_crash_recovery_with_multiple_entries() { // Recover from the same storage (simulating restart) let recovered_ctx = original_ctx.recover_from_crash(); - // Graceful shutdown: close() joins the IO thread before returning, + // Graceful shutdown: close() flushes remaining data before returning, // preventing Tokio runtime shutdown panics. original_ctx.close().await; @@ -124,20 +111,9 @@ async fn test_partial_flush_with_graceful_shutdown() { { let storage = Arc::new(FileStorageEngine::new(storage_path.clone()).unwrap()); - let (raft_log, receiver) = - BufferedRaftLog::>::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 100, - }, - max_buffered_entries: 10000, - shutdown_timeout_ms: 5000, - }, - storage, - ); - let raft_log = raft_log.start(receiver, None); + let raft_log = RaftLogCore::>::new( + 1, storage, None, 5000, + ); // Add 75 entries (1.5 batches) for i in 1..=75 { @@ -161,20 +137,9 @@ async fn test_partial_flush_with_graceful_shutdown() { // Recover from disk let storage = Arc::new(FileStorageEngine::new(storage_path).unwrap()); - let (raft_log, receiver) = - BufferedRaftLog::>::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - max_buffered_entries: 10000, - shutdown_timeout_ms: 5000, - }, - storage, - ); - let raft_log = raft_log.start(receiver, None); + let raft_log = RaftLogCore::>::new( + 1, storage, None, 5000, + ); tokio::time::sleep(Duration::from_millis(50)).await; @@ -184,14 +149,13 @@ async fn test_partial_flush_with_graceful_shutdown() { raft_log.close().await; } -/// MemFirst crash semantics with notify-then-fsync architecture. +/// Crash semantics with inline-persist + FsyncWorker architecture. /// -/// `append_entries` calls `write_notify.notify_one()` on every write, which eagerly -/// wakes the IO thread regardless of `idle_flush_interval_ms`. The idle timer is a -/// safety-net only β€” the normal path persists data almost immediately after each append. +/// `append_entries` persists the new tail and submits it to FsyncWorker on every +/// write. There is no idle timer β€” the normal path fsyncs immediately after each append. /// -/// True crash (kill -9 / power loss) means data MAY survive if the IO thread had time -/// to run before the crash. This test verifies: +/// True crash (kill -9 / power loss) means data MAY survive if the fsync had time +/// to complete before the crash. This test verifies: /// - Entries flushed before the crash (first batch, waited 150ms) always survive. /// - Entries written immediately before crash (second batch, no wait) may or may not /// survive depending on IO thread scheduling at crash time. @@ -206,20 +170,9 @@ async fn test_partial_flush_after_crash() { { let storage = Arc::new(FileStorageEngine::new(storage_path.clone()).unwrap()); - let (raft_log, receiver) = - BufferedRaftLog::>::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 100, - }, - max_buffered_entries: 10000, - shutdown_timeout_ms: 5000, - }, - storage, - ); - let raft_log = raft_log.start(receiver, None); + let raft_log = RaftLogCore::>::new( + 1, storage, None, 5000, + ); // Add first batch (50 entries) for i in 1..=50 { @@ -248,33 +201,22 @@ async fn test_partial_flush_after_crash() { raft_log.append_entries(second_batch).await.unwrap(); // Simulate crash immediately: Skip Drop with mem::forget (like kill -9 or power loss) - // This prevents the processor from flushing the remaining 25 entries + // This prevents the remaining 25 entries from being flushed. // Storage is moved into raft_log, so forgetting raft_log is enough std::mem::forget(raft_log); } // Recover from disk let storage = Arc::new(FileStorageEngine::new(storage_path).unwrap()); - let (raft_log, receiver) = - BufferedRaftLog::>::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - max_buffered_entries: 10000, - shutdown_timeout_ms: 5000, - }, - storage, - ); - let raft_log = raft_log.start(receiver, None); + let raft_log = RaftLogCore::>::new( + 1, storage, None, 5000, + ); tokio::time::sleep(Duration::from_millis(50)).await; - // With notify-then-fsync, the IO thread processes writes eagerly. - // At minimum, the first batch (flushed by safety timer) must survive. - // The second batch may also survive depending on IO thread scheduling. + // With inline-persist + FsyncWorker, writes are fsynced eagerly. + // At minimum, the first batch (fsynced before the crash) must survive. + // The second batch may also survive depending on fsync timing. let recovered = raft_log.len(); assert!( recovered >= batch_size, @@ -294,67 +236,41 @@ async fn test_partial_flush_after_crash() { #[tokio::test] async fn test_recovery_under_different_scenarios() { - // Test various recovery scenarios - // Method C (drain-then-fsync): all writes are fsynced by the IO thread after each - // drain cycle, so all 100 entries are always durable after explicit flush(). - let scenarios = vec![ - ( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - 100usize, - ), - ( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 10, - }, - 100, - ), - ( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1000, - }, - 100, - ), - ]; - - for (strategy, flush_policy, expected_recovery) in scenarios { - let instance_id = format!("recovery_test_{strategy:?}_{flush_policy:?}"); - let original_ctx = TestContext::new(strategy.clone(), flush_policy.clone(), &instance_id); - - // Add test data - for i in 1..=100 { - original_ctx - .raft_log - .append_entries(vec![Entry { - index: i, - term: 1, - payload: Some(EntryPayload::command(Bytes::from( - format!("data{i}").into_bytes(), - ))), - }]) - .await - .unwrap(); - } - - // Explicit flush ensures all entries are durable before crash simulation. - original_ctx.raft_log.flush().await.unwrap(); + // With inline-persist + FsyncWorker, all writes are fsynced after explicit + // flush(), so all 100 entries are always durable. + let expected_recovery = 100usize; - // Simulate crash and recovery - let recovered_ctx = original_ctx.recover_from_crash(); - original_ctx.close().await; + let original_ctx = TestContext::new("test_recovery_under_different_scenarios"); - // Verify recovery based on expected behavior - assert_eq!( - recovered_ctx.raft_log.len(), - expected_recovery, - "Recovery mismatch for strategy {strategy:?} policy {flush_policy:?}" - ); - recovered_ctx.close().await; + // Add test data + for i in 1..=100 { + original_ctx + .raft_log + .append_entries(vec![Entry { + index: i, + term: 1, + payload: Some(EntryPayload::command(Bytes::from( + format!("data{i}").into_bytes(), + ))), + }]) + .await + .unwrap(); } + + // Explicit flush ensures all entries are durable before crash simulation. + original_ctx.raft_log.flush().await.unwrap(); + + // Simulate crash and recovery + let recovered_ctx = original_ctx.recover_from_crash(); + original_ctx.close().await; + + // Verify recovery + assert_eq!( + recovered_ctx.raft_log.len(), + expected_recovery, + "Recovery mismatch: expected {expected_recovery}" + ); + recovered_ctx.close().await; } #[tokio::test] @@ -362,13 +278,7 @@ async fn test_memfirst_crash_recovery_durability() { let instance_id = "test_memfirst_durability"; let recovered_path = { - let ctx = TestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 10000, - }, - instance_id, - ); + let ctx = TestContext::new(instance_id); ctx.append_entries(1, 100, 1).await; @@ -386,20 +296,9 @@ async fn test_memfirst_crash_recovery_durability() { // Recovery let storage = Arc::new(FileStorageEngine::new(PathBuf::from(&recovered_path)).unwrap()); - let (raft_log, receiver) = - BufferedRaftLog::>::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - max_buffered_entries: 10000, - shutdown_timeout_ms: 5000, - }, - storage, - ); - let raft_log = raft_log.start(receiver, None); + let raft_log = RaftLogCore::>::new( + 1, storage, None, 5000, + ); tokio::time::sleep(Duration::from_millis(50)).await; @@ -420,20 +319,9 @@ async fn test_diskfirst_crash_recovery_durability() { let ctx1 = { let storage = Arc::new(FileStorageEngine::new(storage_path.clone()).unwrap()); - let (raft_log, receiver) = - BufferedRaftLog::>::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - max_buffered_entries: 10000, - shutdown_timeout_ms: 5000, - }, - storage, - ); - let raft_log = raft_log.start(receiver, None); + let raft_log = RaftLogCore::>::new( + 1, storage, None, 5000, + ); let entries: Vec<_> = (1..=100) .map(|index| Entry { @@ -454,20 +342,9 @@ async fn test_diskfirst_crash_recovery_durability() { // Phase 2: Recovery let storage = Arc::new(FileStorageEngine::new(storage_path).unwrap()); - let (raft_log, receiver) = - BufferedRaftLog::>::new( - 1, - PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - max_buffered_entries: 10000, - shutdown_timeout_ms: 5000, - }, - storage, - ); - let raft_log = raft_log.start(receiver, None); + let raft_log = RaftLogCore::>::new( + 1, storage, None, 5000, + ); tokio::time::sleep(Duration::from_millis(50)).await; diff --git a/d-engine-server/tests/storage_buffered_raft_log/mod.rs b/d-engine-server/tests/storage_raft_log/mod.rs similarity index 55% rename from d-engine-server/tests/storage_buffered_raft_log/mod.rs rename to d-engine-server/tests/storage_raft_log/mod.rs index ad11cddb..ad40a14b 100644 --- a/d-engine-server/tests/storage_buffered_raft_log/mod.rs +++ b/d-engine-server/tests/storage_raft_log/mod.rs @@ -1,6 +1,6 @@ -//! Integration tests for BufferedRaftLog with real FileStorageEngine +//! Integration tests for RaftLogCore with real FileStorageEngine //! -//! These tests verify BufferedRaftLog behavior with actual disk I/O, +//! These tests verify RaftLogCore behavior with actual disk I/O, //! crash recovery semantics, and performance characteristics. //! //! ## Test Modules @@ -12,18 +12,16 @@ use std::path::PathBuf; use std::sync::Arc; -use std::time::Duration; use bytes::Bytes; -use d_engine_core::{ - BufferedRaftLog, FlushPolicy, PersistenceConfig, PersistenceStrategy, RaftLog, alias::ROF, -}; +use d_engine_core::{RaftLog, RaftLogCore, alias::ROF}; use d_engine_proto::common::{Entry, EntryPayload}; use d_engine_server::{FileStateMachine, FileStorageEngine, node::RaftTypeConfig}; use tempfile::tempdir; mod crash_recovery_test; mod performance_test; +mod quorum_crash_recovery_test; mod storage_integration_test; mod stress_test; @@ -32,56 +30,52 @@ pub struct TestContext { pub raft_log: Arc>>, pub storage: Arc, pub _temp_dir: Option, - pub strategy: PersistenceStrategy, - pub flush_policy: FlushPolicy, pub path: String, + log_flush_rx: tokio::sync::mpsc::UnboundedReceiver, } impl TestContext { /// Create new test context with FileStorageEngine - pub fn new( - strategy: PersistenceStrategy, - flush_policy: FlushPolicy, - instance_id: &str, - ) -> Self { + pub fn new(instance_id: &str) -> Self { let temp_dir = tempdir().unwrap(); let path = temp_dir.path().to_path_buf().join(instance_id); let storage = Arc::new(FileStorageEngine::new(path.clone()).unwrap()); - let (raft_log, receiver) = BufferedRaftLog::new( - 1, - PersistenceConfig { - strategy: strategy.clone(), - flush_policy: flush_policy.clone(), - max_buffered_entries: 10000, - shutdown_timeout_ms: 5000, - }, - storage.clone(), - ); - let raft_log = raft_log.start(receiver, None); - - // Small delay to ensure processor is ready - std::thread::sleep(Duration::from_millis(10)); + let (log_flush_tx, log_flush_rx) = tokio::sync::mpsc::unbounded_channel(); + let raft_log = RaftLogCore::new(1, storage.clone(), Some(log_flush_tx), 5000); Self { path: path.to_str().unwrap().to_string(), raft_log, storage, - strategy, - flush_policy, _temp_dir: Some(temp_dir), + log_flush_rx, } } - /// Explicitly close the raft log IO thread. + /// Stands in for `raft.rs`'s `InternalEvent::FsyncCompleted` handler, + /// which isn't running in these `RaftLogCore`-only integration + /// tests. Since #446/#447, `durable_index` only advances when something + /// drains that event and calls `try_advance_durable_index` β€” call this + /// after any operation that should make `durable_index` advance and + /// before asserting on it. Not needed after `recover_from_crash()`: the + /// recovered context's `durable_index` is derived directly from on-disk + /// state at construction, not from this event. + pub fn drain_fsync_completions(&mut self) { + while let Ok(event) = self.log_flush_rx.try_recv() { + if let d_engine_core::InternalEvent::FsyncCompleted { mark, sent_at: _ } = event { + self.raft_log.try_advance_durable_index(mark); + } + } + } + + /// Explicitly close the raft log. /// - /// Must be called at the end of tests using graceful-shutdown semantics. - /// Unlike `drop()` which only sends Shutdown (fire-and-forget), `close()` - /// joins the IO thread before returning, preventing Tokio runtime - /// shutdown panics. + /// Must be called at the end of tests using graceful-shutdown semantics: + /// `close()` flushes remaining data (bounded by `shutdown_timeout_ms`) before + /// returning, preventing Tokio runtime shutdown panics. pub async fn close(self) { self.raft_log.close().await; - // _temp_dir drops here, after the IO thread has fully exited } /// Simulate crash recovery by creating new context from same storage path @@ -89,27 +83,15 @@ impl TestContext { let temp_dir = tempdir().unwrap(); let storage = Arc::new(FileStorageEngine::new(PathBuf::from(self.path.clone())).unwrap()); - let (raft_log, receiver) = BufferedRaftLog::new( - 1, - PersistenceConfig { - strategy: self.strategy.clone(), - flush_policy: self.flush_policy.clone(), - max_buffered_entries: 10000, - shutdown_timeout_ms: 5000, - }, - storage.clone(), - ); - let raft_log = raft_log.start(receiver, None); - - std::thread::sleep(Duration::from_millis(10)); + let (log_flush_tx, log_flush_rx) = tokio::sync::mpsc::unbounded_channel(); + let raft_log = RaftLogCore::new(1, storage.clone(), Some(log_flush_tx), 5000); Self { raft_log, storage, - strategy: self.strategy.clone(), - flush_policy: self.flush_policy.clone(), _temp_dir: Some(temp_dir), path: self.path.clone(), + log_flush_rx, } } diff --git a/d-engine-server/tests/storage_buffered_raft_log/performance_test.rs b/d-engine-server/tests/storage_raft_log/performance_test.rs similarity index 83% rename from d-engine-server/tests/storage_buffered_raft_log/performance_test.rs rename to d-engine-server/tests/storage_raft_log/performance_test.rs index 0368f99a..89997074 100644 --- a/d-engine-server/tests/storage_buffered_raft_log/performance_test.rs +++ b/d-engine-server/tests/storage_raft_log/performance_test.rs @@ -1,4 +1,4 @@ -//! Performance benchmark integration tests for BufferedRaftLog +//! Performance benchmark integration tests for RaftLogCore //! //! These tests measure real I/O performance with FileStorageEngine. //! Most tests are marked with #[ignore] and can be run explicitly or @@ -6,9 +6,7 @@ use super::TestContext; use bytes::Bytes; -use d_engine_core::{ - BufferedRaftLog, FlushPolicy, PersistenceConfig, PersistenceStrategy, RaftLog, -}; +use d_engine_core::{RaftLog, RaftLogCore}; use d_engine_proto::common::{Entry, EntryPayload}; use d_engine_server::{FileStateMachine, FileStorageEngine, node::RaftTypeConfig}; use std::collections::HashMap; @@ -31,25 +29,15 @@ mod filter_out_conflicts_and_append_performance_tests { ]; for (idle_flush_interval_ms, max_duration_ms) in test_cases { - // Create MemFirst storage with batch policy - let config = PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms, - }, - max_buffered_entries: 10000, - shutdown_timeout_ms: 5000, - }; - let temp_dir = tempdir().unwrap(); let path = temp_dir.path().to_path_buf(); - let (log, receiver) = BufferedRaftLog::< - RaftTypeConfig, - >::new( - 1, config, Arc::new(FileStorageEngine::new(path).unwrap()) + let log = RaftLogCore::>::new( + 1, + Arc::new(FileStorageEngine::new(path).unwrap()), + None, + 5000, ); - let log = log.start(receiver, None); // Populate with test data (1000 entries) let mut entries = vec![]; @@ -85,10 +73,17 @@ mod filter_out_conflicts_and_append_performance_tests { "Duration {duration}ms exceeds max {max_duration_ms}ms for {idle_flush_interval_ms}ms interval" ); - // Verify correctness - assert!(log.entry(500).unwrap().is_none()); + // Verify correctness: index=501 term=1 already exists (populated above), so this + // prev_log_index=0 resend is a pure duplicate β€” must be a no-op, not a reset. + // See 446-expert-q-probe-backpressure-fix-8020.md. + assert_eq!( + log.last_entry_id(), + 1000, + "duplicate resend must not touch the log" + ); + assert!(log.entry(500).unwrap().is_some()); assert!(log.entry(501).unwrap().is_some()); - assert!(log.entry(502).unwrap().is_none()); + assert!(log.entry(502).unwrap().is_some()); } } @@ -102,25 +97,15 @@ mod filter_out_conflicts_and_append_performance_tests { ]; for (idle_flush_interval_ms, max_duration_ms) in test_cases { - // Create MemFirst storage with batch policy - let config = PersistenceConfig { - strategy: PersistenceStrategy::MemFirst, - flush_policy: FlushPolicy::Batch { - idle_flush_interval_ms, - }, - max_buffered_entries: 10000, - shutdown_timeout_ms: 5000, - }; - let temp_dir = tempdir().unwrap(); let path = temp_dir.path().to_path_buf(); - let (log, receiver) = BufferedRaftLog::< - RaftTypeConfig, - >::new( - 1, config, Arc::new(FileStorageEngine::new(path).unwrap()) + let log = RaftLogCore::>::new( + 1, + Arc::new(FileStorageEngine::new(path).unwrap()), + None, + 5000, ); - let log = log.start(receiver, None); // Populate with test data (1000 entries) let mut entries = vec![]; @@ -170,13 +155,7 @@ mod filter_out_conflicts_and_append_performance_tests { #[tokio::test] async fn test_last_entry_id_performance() { // Set up test context - let test_context = TestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 360_000, - }, - "test_last_entry_id_performance", - ); + let test_context = TestContext::new("test_last_entry_id_performance"); // Create a large number of entries const ENTRY_COUNT: usize = 1_000_000; @@ -225,29 +204,33 @@ async fn test_performance_benchmarks() { // Adjust test parameters according to the environment let operations = if is_ci { // CI environment uses a more relaxed threshold + // + // append_entries now round-trips through a dedicated IO thread via + // oneshot (see #444: leader's own write must reach the storage + // engine before counting toward quorum) β€” this is an intentional + // correctness/speed tradeoff, not a regression. The old threshold + // (500) predates that fix. New floor leaves ~2x headroom below the + // observed ~318-328 ops/sec on a modern dev machine, keeping the + // 2:1 local:CI ratio from before. [ - ("append_entries", 500, 500.0), + ("append_entries", 500, 100.0), ("get_entries_range", 2500, 25000.0), ("entry_lookup", 5000, 100000.0), ("term_queries", 4000, 25000.0), ] } else { // Local environment uses a stricter threshold + // + // See CI-branch comment above β€” same #444 rationale. [ - ("append_entries", 1000, 1000.0), + ("append_entries", 1000, 200.0), ("get_entries_range", 5000, 50000.0), ("entry_lookup", 10000, 200000.0), ("term_queries", 8000, 50000.0), ] }; - let ctx = TestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 100, - }, - "performance_benchmark", - ); + let ctx = TestContext::new("performance_benchmark"); // Pre-populate with data let mut entries = Vec::new(); @@ -331,13 +314,7 @@ async fn test_read_performance_under_concurrent_write_load() { 10_000.0 }; - let ctx = TestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 100, - }, - "test_read_performance_under_concurrent_write_load", - ); + let ctx = TestContext::new("test_read_performance_under_concurrent_write_load"); // Pre-populate ctx.append_entries(1, 10000, 1).await; diff --git a/d-engine-server/tests/storage_raft_log/quorum_crash_recovery_test.rs b/d-engine-server/tests/storage_raft_log/quorum_crash_recovery_test.rs new file mode 100644 index 00000000..a307885f --- /dev/null +++ b/d-engine-server/tests/storage_raft_log/quorum_crash_recovery_test.rs @@ -0,0 +1,104 @@ +//! Quorum + real-disk crash recovery integration test (#446 gap 4). +//! +//! Composes two pieces that are each already covered in isolation elsewhere, but never +//! together: `calculate_majority_matched_index` (RPO=0 quorum arithmetic, unit-tested +//! against a gated mock in `raft_log_core_test/quorum_durability_test.rs`) and real +//! `FileStorageEngine` crash/reopen (unit-tested without any quorum math in +//! `crash_recovery_test.rs`). This file proves they actually compose: an index that the +//! quorum calculation says is safe to acknowledge to the client is still there after a +//! real crash + reopen from the same on-disk path. +//! +//! Followers are represented as reported match_index values, same as in +//! `quorum_durability_test.rs` β€” this file's job is the leader-side real-disk durability +//! boundary, not follower ACK withholding (covered by follower_state_test.rs / +//! learner_state_test.rs). +//! +//! Deliberately NOT attempted here, and now CONFIRMED impossible with this engine's +//! architecture (not just a flakiness risk β€” an actual dead end, verified by building and +//! deadlocking it): proving that an entry which never reached quorum-durable is genuinely +//! absent from a real crash + reopen. `RaftLogCore::append_entries` +//! (`d-engine-core/src/storage/raft_log_core.rs`) is documented and +//! implemented to block the caller until `persist_entries()` returns β€” "still blocks the +//! caller until truly persisted" β€” and `FileLogStore::persist_entries` +//! (`d-engine-server/src/storage/adaptors/file/file_storage_engine.rs:254`) already writes +//! the entry to the real OS-visible file as an unconditional part of its body, before it +//! can return. So by the time `append_entries().await` ever resolves at all, the entry is +//! already on the file β€” there is no window where it's "acknowledged as appended" yet +//! "recoverably absent." A gate on `persist_entries()` was built and tried here; it did +//! not create the intended window, it just deadlocked `append_entries()` forever (the +//! call this file's other test depends on to make progress at all). Reverted. +//! What IS real and already correctly tested (see below): the gap between +//! `last_entry_id` and `durable_index` β€” `persist_entries()` writes the bytes, but a +//! *separate* `flush()` call (`sync_all()`) is what advances `durable_index`, and that one +//! genuinely runs later/independently. What's NOT reachable by a same-process test is +//! observing that an un-`sync_all`'d write doesn't survive β€” on the same OS instance, +//! `write()` alone (which `persist_entries` already does) is enough for a freshly-opened +//! handle to see the bytes, real crash or not. Proving the un-fsynced case would need an +//! actual power-loss simulation (dropped page cache / real reboot) β€” which prior sessions +//! already found to be a poor fit for this class of bug (see mempalace notes on the +//! Jepsen/lazyfs work for #444). + +use super::TestContext; +use d_engine_core::RaftLog; + +#[tokio::test] +async fn test_quorum_acknowledged_index_survives_real_crash_and_reopen() { + let mut ctx = TestContext::new("test_quorum_ack_survives_crash"); + + // First 5 entries, explicitly flushed: genuinely durable, deterministic. + ctx.append_entries(1, 5, 1).await; + ctx.raft_log.flush().await.unwrap(); + ctx.drain_fsync_completions(); + assert_eq!(ctx.raft_log.durable_index(), 5); + + // 3-node cluster: both followers already report match_index=5 (post-Stage2 + // semantics β€” a follower only reports a match_index once its own durable_index + // reaches it). This is the index the leader would actually acknowledge to the + // client. + let commit = ctx.raft_log.calculate_majority_matched_index(1, 0, vec![5, 5]); + assert_eq!( + commit, + Some(5), + "index 5 is durable on the leader and acked by both followers" + ); + + // A follower report of 10 must not move commit past what the leader itself has + // fsynced β€” restates the Stage1 invariant as this test's own setup precondition + // rather than assuming it silently. + let would_be_wrong = ctx.raft_log.calculate_majority_matched_index(1, 0, vec![10, 5]); + assert_eq!( + would_be_wrong, + Some(5), + "leader's own un-fsynced tail must not leak into the client-visible commit index" + ); + + // Second batch, also explicitly flushed, so the whole log is durable before the + // simulated crash β€” keeps this test's crash/recovery assertions exact, not bounded. + ctx.append_entries(6, 5, 1).await; + ctx.raft_log.flush().await.unwrap(); + ctx.drain_fsync_completions(); + assert_eq!(ctx.raft_log.durable_index(), 10); + + let recovered = ctx.recover_from_crash(); + ctx.close().await; + tokio::time::sleep(std::time::Duration::from_millis(50)).await; + + // The index that was actually acknowledged to the client (5) β€” and everything else + // that was durably flushed (up to 10) β€” survives the real crash + reopen. + assert_eq!(recovered.raft_log.durable_index(), 10); + for i in 1..=10 { + assert!( + recovered.raft_log.entry(i).unwrap().is_some(), + "entry {i} must survive real crash + reopen" + ); + } + + // Re-running the same quorum calculation against the recovered log reaches the same + // conclusion β€” the leader's durability contribution to quorum is stable across a + // real restart, not just in the pre-crash in-memory view. + let commit_after_recovery = + recovered.raft_log.calculate_majority_matched_index(1, 0, vec![5, 5]); + assert_eq!(commit_after_recovery, Some(5)); + + recovered.close().await; +} diff --git a/d-engine-server/tests/storage_buffered_raft_log/storage_integration_test.rs b/d-engine-server/tests/storage_raft_log/storage_integration_test.rs similarity index 68% rename from d-engine-server/tests/storage_buffered_raft_log/storage_integration_test.rs rename to d-engine-server/tests/storage_raft_log/storage_integration_test.rs index 8cd4ab85..b160990b 100644 --- a/d-engine-server/tests/storage_buffered_raft_log/storage_integration_test.rs +++ b/d-engine-server/tests/storage_raft_log/storage_integration_test.rs @@ -1,9 +1,9 @@ -//! Storage-level integration tests for BufferedRaftLog +//! Storage-level integration tests for RaftLogCore //! -//! These tests verify BufferedRaftLog integration with FileStorageEngine +//! These tests verify RaftLogCore integration with FileStorageEngine //! at the storage layer, including compaction and storage-specific operations. -use d_engine_core::{FlushPolicy, PersistenceStrategy, RaftLog}; +use d_engine_core::RaftLog; use d_engine_proto::common::LogId; use super::TestContext; @@ -13,17 +13,12 @@ use super::TestContext; #[tokio::test] async fn test_log_compaction() { - let ctx = TestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_log_compaction", - ); + let mut ctx = TestContext::new("test_log_compaction"); ctx.append_entries(1, 100, 1).await; // With MemFirst, entries are buffered and flushed asynchronously. // Wait for all entries to become durable before checking durable_index. ctx.raft_log.flush().await.unwrap(); + ctx.drain_fsync_completions(); // Compact first 50 entries ctx.raft_log.purge_logs_up_to(LogId { index: 50, term: 1 }).await.unwrap(); diff --git a/d-engine-server/tests/storage_buffered_raft_log/stress_test.rs b/d-engine-server/tests/storage_raft_log/stress_test.rs similarity index 76% rename from d-engine-server/tests/storage_buffered_raft_log/stress_test.rs rename to d-engine-server/tests/storage_raft_log/stress_test.rs index f89ae55a..d3b079fc 100644 --- a/d-engine-server/tests/storage_buffered_raft_log/stress_test.rs +++ b/d-engine-server/tests/storage_raft_log/stress_test.rs @@ -1,12 +1,12 @@ -//! High concurrency stress tests for BufferedRaftLog +//! High concurrency stress tests for RaftLogCore //! -//! These tests verify BufferedRaftLog behavior under high load with +//! These tests verify RaftLogCore behavior under high load with //! real FileStorageEngine, testing race conditions and resource limits. use std::time::Duration; use bytes::Bytes; -use d_engine_core::{FlushPolicy, LogStore, PersistenceStrategy, RaftLog, StorageEngine}; +use d_engine_core::{LogStore, RaftLog, StorageEngine}; use d_engine_proto::common::{Entry, EntryPayload}; use futures::future::join_all; use tokio::time::Instant; @@ -22,13 +22,7 @@ use super::TestContext; #[tokio::test] async fn test_high_concurrency() { - let ctx = TestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_high_concurrency", - ); + let mut ctx = TestContext::new("test_high_concurrency"); let mut handles = vec![]; for i in 0..10 { @@ -53,6 +47,7 @@ async fn test_high_concurrency() { // With MemFirst, entries are buffered; wait for all to be durable before asserting. ctx.raft_log.flush().await.unwrap(); + ctx.drain_fsync_completions(); // Verify all entries persisted assert_eq!(ctx.raft_log.durable_index(), 1000); @@ -62,13 +57,7 @@ async fn test_high_concurrency() { #[tokio::test] #[traced_test] async fn test_high_concurrency_mixed_operations() { - let ctx = TestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 100, - }, - "test_high_concurrency_mixed_operations", - ); + let ctx = TestContext::new("test_high_concurrency_mixed_operations"); let mut handles = vec![]; let start_time = Instant::now(); @@ -123,8 +112,15 @@ async fn test_high_concurrency_mixed_operations() { // Verify data integrity assert_eq!(ctx.raft_log.len(), 10000); + // append_entries() now round-trips through a dedicated IO thread via + // oneshot (see #444: leader's own write must reach the storage engine + // before counting toward quorum) β€” an intentional correctness/speed + // tradeoff, not a regression. The old 10s bound predates that fix; + // observed wall-clock for this test's 10k concurrent writes is now + // 13-18s depending on machine load. New bound leaves real headroom + // above that range rather than chasing the exact number. assert!( - duration < Duration::from_secs(10), + duration < Duration::from_secs(30), "Operations took too long: {duration:?}" ); } @@ -134,13 +130,7 @@ mod mem_first_tests { #[tokio::test] async fn test_basic_write_before_persist() { - let ctx = TestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_basic_write_before_persist", - ); + let ctx = TestContext::new("test_basic_write_before_persist"); ctx.append_entries(1, 5, 1).await; // Verify in memory but not yet durable @@ -150,17 +140,12 @@ mod mem_first_tests { #[tokio::test] async fn test_async_persistence() { - let ctx = TestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_async_persistence", - ); + let mut ctx = TestContext::new("test_async_persistence"); ctx.append_entries(1, 100, 1).await; // Trigger flush ctx.raft_log.flush().await.unwrap(); + ctx.drain_fsync_completions(); // Verify persistence assert_eq!(ctx.raft_log.durable_index(), 100); @@ -169,13 +154,7 @@ mod mem_first_tests { #[tokio::test] async fn test_power_loss_data_loss() { - let ctx = TestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_power_loss_data_loss", - ); + let ctx = TestContext::new("test_power_loss_data_loss"); ctx.append_entries(1, 100, 1).await; // Simulate power loss before flush @@ -187,13 +166,7 @@ mod mem_first_tests { #[tokio::test] async fn test_high_concurrency_memory_only() { - let ctx = TestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_high_concurrency_memory_only", - ); + let ctx = TestContext::new("test_high_concurrency_memory_only"); let mut handles = vec![]; for i in 0..10 { @@ -224,13 +197,7 @@ mod mem_first_tests { #[tokio::test] async fn test_term_index_correctness_under_load() { - let ctx = TestContext::new( - PersistenceStrategy::MemFirst, - FlushPolicy::Batch { - idle_flush_interval_ms: 1, - }, - "test_term_index_under_load", - ); + let ctx = TestContext::new("test_term_index_under_load"); // Concurrent writes with different terms let mut handles = vec![]; diff --git a/d-engine-server/tests/watch_and_subscriptions/watch_membership_embedded.rs b/d-engine-server/tests/watch_and_subscriptions/watch_membership_embedded.rs index b5edc3e4..051bd020 100644 --- a/d-engine-server/tests/watch_and_subscriptions/watch_membership_embedded.rs +++ b/d-engine-server/tests/watch_and_subscriptions/watch_membership_embedded.rs @@ -125,8 +125,6 @@ learner_check_throttle_ms = 100 election_timeout_min = 300 election_timeout_max = 3000 -[raft.persistence] -strategy = "MemFirst" [retry.election] max_retries = 5 diff --git a/d-engine/src/docs/examples/three-nodes-standalone.md b/d-engine/src/docs/examples/three-nodes-standalone.md index 96cbe01a..1f29881a 100644 --- a/d-engine/src/docs/examples/three-nodes-standalone.md +++ b/d-engine/src/docs/examples/three-nodes-standalone.md @@ -57,11 +57,6 @@ initial_cluster = [ [raft.read_consistency] default_policy = "LeaseRead" lease_duration_ms = 500 - -[raft.persistence] -strategy = "MemFirst" # Only strategy in v0.2.4+ (DiskFirst removed) -flush_policy = { Batch = { idle_flush_interval_ms = 1000 } } -max_buffered_entries = 10000 ``` **Key differences from single-node expansion:** @@ -123,7 +118,7 @@ All performance reports in `/benches/standalone-bench/reports` use this exact co **Raft settings:** -- Persistence: `MemFirst` (only strategy in v0.2.4+) with 1000ms idle flush interval +- Persistence: inline fsync (Level 3, fdatasync) - Read consistency: `LeaseRead` (500ms lease duration) - Replication: Batched append entries (5000 threshold, 0ms delay) - Network: Tuned for high throughput (see `config/n1.toml` for details) diff --git a/d-engine/src/docs/performance/metrics-reference.md b/d-engine/src/docs/performance/metrics-reference.md index 267ae443..1c06b251 100644 --- a/d-engine/src/docs/performance/metrics-reference.md +++ b/d-engine/src/docs/performance/metrics-reference.md @@ -21,31 +21,31 @@ rocksdb.*` vs `server.storage.file.*`). ## Write Pipeline Core -| Metric | Type | Normal Range | Answers | -|---|---|---|---| -| `core.raft.buffer.length{buffer}` | Gauge | Near 0, draining between ticks | Is the propose/linearizable/lease/eventual buffer backlogged? | -| `core.raft.fsync.duration_ms` | Histogram | p99 low single-digit ms on SSD | How long does one physical fsync take? | -| `core.raft.fsync.batch_entries` | Histogram | Grows with write concurrency | Is FsyncCoordinator coalescing concurrent writes? | -| `core.raft.fsync.inflight` | Gauge (0/1) | β€” | Is a fsync task running right now? | -| `core.raft.fsync.busy_nanos_total` | Counter | `rate(...)/1e9` should stay < 0.7 | fsync thread utilization | -| `server.storage.rocksdb.wal_flush_ms` | Histogram | p99 low single-digit ms | State machine's own RocksDB WAL flush duration (a separate DB from the Raft log β€” do not conflate with `fsync.duration_ms`). Only emitted when the `rocksdb` storage adaptor is active. | -| `server.storage.file.flush_ms` | Histogram | p99 low single-digit ms | Durability sync duration (`flush()` + `sync_all()`) for the default `file` storage adaptor. Only emitted when the `file` adaptor is active β€” no WAL concept, so this is the direct equivalent of `wal_flush_ms`. | -| `core.state_machine.apply_chunk.duration_ms` | Histogram | p99 low single-digit ms | How long does one apply_chunk call take? | -| `core.state_machine.apply_chunk.batch_size` | Histogram | Grows with write concurrency | Entries applied per chunk | -| `core.state_machine.apply_chunk.count` | Counter | β€” | Total apply_chunk invocations | -| `core.state_machine.apply_chunk.success` | Counter | β‰ˆ `.count` | Successful applies | -| `core.state_machine.apply_chunk.error{error_type}` | Counter | 0 | Failed applies, classified by error type | -| `core.state_machine.apply.busy_nanos_total` | Counter | `rate(...)/1e9` should stay < 0.7 | SM apply thread utilization | -| `core.raft.commit_index` | Gauge | Monotonically increasing | Highest log index this node has committed | -| `core.raft.apply_index` | Gauge | Tracks `commit_index` closely | Highest log index this node has applied | +| Metric | Type | Normal Range | Answers | +| -------------------------------------------------- | ----------- | --------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `core.raft.buffer.length{buffer}` | Gauge | Near 0, draining between ticks | Is the propose/linearizable/lease/eventual buffer backlogged? | +| `core.raft.fsync.duration_ms` | Histogram | p99 low single-digit ms on SSD | How long does one physical fsync take? | +| `core.raft.fsync.batch_entries` | Histogram | Grows with write concurrency | Is FsyncWorker coalescing concurrent writes? | +| `core.raft.fsync.inflight` | Gauge (0/1) | β€” | Is a fsync task running right now? | +| `core.raft.fsync.busy_nanos_total` | Counter | `rate(...)/1e9` should stay < 0.7 | fsync thread utilization | +| `server.storage.rocksdb.wal_flush_ms` | Histogram | p99 low single-digit ms | State machine's own RocksDB WAL flush duration (a separate DB from the Raft log β€” do not conflate with `fsync.duration_ms`). Only emitted when the `rocksdb` storage adaptor is active. | +| `server.storage.file.flush_ms` | Histogram | p99 low single-digit ms | Durability sync duration (`flush()` + `sync_all()`) for the default `file` storage adaptor. Only emitted when the `file` adaptor is active β€” no WAL concept, so this is the direct equivalent of `wal_flush_ms`. | +| `core.state_machine.apply_chunk.duration_ms` | Histogram | p99 low single-digit ms | How long does one apply_chunk call take? | +| `core.state_machine.apply_chunk.batch_size` | Histogram | Grows with write concurrency | Entries applied per chunk | +| `core.state_machine.apply_chunk.count` | Counter | β€” | Total apply_chunk invocations | +| `core.state_machine.apply_chunk.success` | Counter | β‰ˆ `.count` | Successful applies | +| `core.state_machine.apply_chunk.error{error_type}` | Counter | 0 | Failed applies, classified by error type | +| `core.state_machine.apply.busy_nanos_total` | Counter | `rate(...)/1e9` should stay < 0.7 | SM apply thread utilization | +| `core.raft.commit_index` | Gauge | Monotonically increasing | Highest log index this node has committed | +| `core.raft.apply_index` | Gauge | Tracks `commit_index` closely | Highest log index this node has applied | ## Write Latency Breakdown (leader-only) -| Metric | Type | Normal Range | Answers | -|---|---|---|---| -| `core.raft.write.propose_to_commit_ms` | Histogram | Low single-digit ms | Client write β†’ Raft commit | -| `core.raft.write.commit_to_apply_ms` | Histogram | Should stay well below `propose_to_commit_ms` | Raft commit β†’ state machine apply | -| `core.raft.write.propose_to_apply_ms` | Histogram | Sum of the two above | End-to-end write latency (what the client experiences) | +| Metric | Type | Normal Range | Answers | +| -------------------------------------- | --------- | --------------------------------------------- | ------------------------------------------------------ | +| `core.raft.write.propose_to_commit_ms` | Histogram | Low single-digit ms | Client write β†’ Raft commit | +| `core.raft.write.commit_to_apply_ms` | Histogram | Should stay well below `propose_to_commit_ms` | Raft commit β†’ state machine apply | +| `core.raft.write.propose_to_apply_ms` | Histogram | Sum of the two above | End-to-end write latency (what the client experiences) | ### How the segments add up @@ -56,8 +56,7 @@ propose_to_commit_ms + commit_to_apply_ms = propose_to_apply_ms ``` This is an exact per-request identity β€” the three timestamps are recorded -against the same log index. **Percentiles do not add**: `p99(propose_to_commit) -+ p99(commit_to_apply)` will not generally equal `p99(propose_to_apply)`, +against the same log index. **Percentiles do not add**: `p99(propose_to_commit) + p99(commit_to_apply)` will not generally equal `p99(propose_to_apply)`, because the slowest 1% of requests in each stage aren't necessarily the same requests. If you need to verify the identity, compare it per-request or via the mean, not by summing percentiles. @@ -84,18 +83,18 @@ together). ## Replication -| Metric | Type | Normal Range | Answers | -|---|---|---|---| -| `server.raft.replicate.rtt_ms{peer}` | Histogram | Should track your network's baseline RTT | AppendEntries round-trip time to a specific peer | -| `core.raft.snapshot.push_consecutive_failures` | Counter | 0 | Consecutive snapshot push failures to a peer | +| Metric | Type | Normal Range | Answers | +| ---------------------------------------------- | --------- | ---------------------------------------- | ------------------------------------------------ | +| `server.raft.replicate.rtt_ms{peer}` | Histogram | Should track your network's baseline RTT | AppendEntries round-trip time to a specific peer | +| `core.raft.snapshot.push_consecutive_failures` | Counter | 0 | Consecutive snapshot push failures to a peer | ## Cluster Health & Guardrails -| Metric | Type | Normal Range | Answers | -|---|---|---|---| -| `core.raft.backpressure.rejections{node_id,type}` | Counter | 0 | Requests rejected due to backpressure (write/read) | -| `core.membership.stale_learner_removed` | Counter | 0 | Learners auto-removed for falling too far behind | -| `core.cluster.unsafe_join_attempts` | Counter | 0 | Join requests rejected because they would create an even-voter cluster | +| Metric | Type | Normal Range | Answers | +| ------------------------------------------------- | ------- | ------------ | ---------------------------------------------------------------------- | +| `core.raft.backpressure.rejections{node_id,type}` | Counter | 0 | Requests rejected due to backpressure (write/read) | +| `core.membership.stale_learner_removed` | Counter | 0 | Learners auto-removed for falling too far behind | +| `core.cluster.unsafe_join_attempts` | Counter | 0 | Join requests rejected because they would create an even-voter cluster | --- @@ -104,10 +103,10 @@ together). d-engine's metrics operate at two different granularities. Comparing across them directly leads to wrong conclusions: -| Granularity | Metrics | -|---|---| +| Granularity | Metrics | +| ----------------------------------------------- | ----------------------------------------------------------------------------------------------- | | Per fsync/apply batch (may cover many requests) | `fsync.duration_ms`, `fsync.batch_entries`, `apply_chunk.duration_ms`, `apply_chunk.batch_size` | -| Per single client request | `write.propose_to_commit_ms`, `write.commit_to_apply_ms`, `write.propose_to_apply_ms` | +| Per single client request | `write.propose_to_commit_ms`, `write.commit_to_apply_ms`, `write.propose_to_apply_ms` | Example: `fsync.duration_ms` p99 of 5ms does not mean "any given `propose_to_commit_ms` sample near 5ms is explained by that fsync" β€” one fsync diff --git a/d-engine/src/docs/performance/throughput-optimization-guide.md b/d-engine/src/docs/performance/throughput-optimization-guide.md index 03bc4efc..8f1ffb98 100644 --- a/d-engine/src/docs/performance/throughput-optimization-guide.md +++ b/d-engine/src/docs/performance/throughput-optimization-guide.md @@ -23,39 +23,20 @@ pub(crate) enum ConnectionType { ## Persistence Strategy & Throughput/Latency Trade-offs -`MemFirst` is the only persistence strategy in v0.2.4+. It batches writes to OS page cache and flushes with fsync asynchronously β€” committing data to disk before notifying Raft. - -### Strategy Configuration - -```toml -[raft.persistence] -strategy = "MemFirst" -flush_policy = { Batch = { idle_flush_interval_ms = 1000 } } -``` - -### `MemFirst` Strategy - -- **Write Path**: Entries are written to OS page cache via `db.write()` / `file.write()`; the IO - thread batches them and calls fsync (`flush_wal(true)` / `sync_all()`) before advancing - `durable_index`. Raft only counts an entry toward quorum after fsync completes. +- **Write Path**: Entries are written to OS page cache via `db.write()` / `file.write()`, then + fsynced (`flush_wal(true)` / `sync_all()`) before `durable_index` advances. Raft only counts an + entry toward quorum after fsync completes. - **Durability**: - _Process crash_: OS page cache survives a process restart β†’ full recovery via WAL replay. βœ… - _Power loss (single node)_: In a multi-node cluster, Raft quorum ensures committed data survives a single-node power failure β€” the other quorum members retain the data. βœ… - - _Power loss (majority of nodes simultaneously)_: Entries in the current unflushed batch - (written since the last fsync) may be lost. This window is bounded by - `idle_flush_interval_ms`. Raft has not yet counted these entries toward quorum, so no - client-acknowledged write is lost β€” the leader will re-replicate after re-election. ⚠️ -- **Throughput**: High. Multiple writes share a single fsync; no per-write fsync overhead. - -### `FlushPolicy` Tuning + - _Power loss (majority of nodes simultaneously)_: Entries not yet fsynced on a quorum may be + lost. Raft has not yet counted these entries toward quorum, so no client-acknowledged write + is lost β€” the leader will re-replicate after re-election. ⚠️ +- **Throughput**: High. Concurrent writes are coalesced into a single fsync by `FsyncWorker`; no + per-write fsync overhead. -- **`Batch { idle_flush_interval_ms }`**: Flush (fsync) after this many milliseconds of idle time. - - Lower values reduce the unflushed batch window but increase IO pressure. - - Default `1000` ms is suitable for most workloads. - -> **Note**: `DiskFirst` strategy was removed in v0.2.4. `MemFirst` replaced it with batched -> fsync β€” multiple writes share one fsync call, reducing IO overhead while still providing +> **Note**: Writes are batched into a single fsync, reducing IO overhead while still providing > disk-level durability for all client-acknowledged (committed) writes. ## Batching Configuration @@ -179,7 +160,7 @@ tonic::transport::Server::builder() ``` -Inbound message size is the one setting that *is* per-service rather than +Inbound message size is the one setting that _is_ per-service rather than transport-wide, so it's applied on each `XxxServiceServer` individually: ```rust,ignore @@ -197,7 +178,7 @@ RaftReplicationServiceServer::from_arc(node.clone()) | p99.9 Latency | 14015 Β΅s | 11279 Β΅s | -19.5% | > **Key improvement**: 15% reduction in tail latency - critical for consensus stability -> **Note**: These metrics show the impact of connection pooling optimization. These results can be further improved by tuning the PersistenceStrategy for your specific workload. +> **Note**: These metrics show the impact of connection pooling optimization. > > For absolute performance benchmarks, see [v0.2.4 Performance Report](https://github.com/deventlab/d-engine/tree/main/benches/reports/v0.2.4/bench_report_v0.2.4.md) @@ -230,7 +211,7 @@ RaftReplicationServiceServer::from_arc(node.clone()) ``` -5. **Monitor Flush Lag**: When using `MemFirst`, monitor the difference between `last_log_index` and `durable_index`. A growing gap indicates the disk is not keeping up with writes, increasing potential data loss. +5. **Monitor Flush Lag**: Monitor the difference between `last_log_index` and `durable_index`. Raft only counts an entry toward quorum and acknowledges it to the client after fsync β€” so a growing gap does not put acknowledged writes at risk. It does mean client-facing write latency is growing, and (if the gap keeps growing) the amount of work an unflushed batch would need to redo on restart is growing too. ## Anti-Patterns to Avoid @@ -243,14 +224,6 @@ client.request_vote(...) // Control operation on data channel get_peer_channel(peer_id, ConnectionType::Control).await?; client.request_vote(...) -// DON'T: Set idle_flush_interval_ms too low β€” defeats batching. -[strategy = "MemFirst"] -flush_policy = { Batch = { idle_flush_interval_ms = 1 } } // Near-synchronous; low throughput - -// DO: Use a generous idle interval to amortize disk I/O cost. -[strategy = "MemFirst"] -flush_policy = { Batch = { idle_flush_interval_ms = 1000 } } - ``` ## Why Connection Isolation and Strategy Choice Matters @@ -261,8 +234,8 @@ flush_policy = { Batch = { idle_flush_interval_ms = 1000 } } Control: Low latency ↔ Data: High throughput ↔ Bulk: Bandwidth 3. **Improves fault containment** Connection issues affect only one operation type -4. **Decouples Performance from Durability** - `MemFirst` with tunable `idle_flush_interval_ms` lets you balance write throughput against flush frequency. +4. **Decouples Performance from Ack Latency** + Client-acknowledged writes are always fsync-durable β€” that's not tunable. ## Reference Deployment Configurations @@ -276,10 +249,6 @@ Adjust values based on snapshot size, log append rate, and cluster size. β€’ Network: Localhost ```toml -[raft.persistence] -strategy = "MemFirst" -flush_policy = { Batch = { idle_flush_interval_ms = 1000 } } - [network.control] connection_window_size = 1_048_576 # 1MB @@ -305,10 +274,6 @@ max_concurrent_streams = 128 β€’ Priority: Balanced throughput and durability ```toml -[raft.persistence] -strategy = "MemFirst" -flush_policy = { Batch = { idle_flush_interval_ms = 1000 } } - [network.control] connection_window_size = 2_097_152 # 2MB @@ -325,7 +290,7 @@ max_concurrent_streams = 256 ``` -**Tip**: For public cloud, moderate concurrency and 32MB bulk windows ensure stable snapshot streaming without affecting heartbeats. The batch policy is tuned for high throughput with a reasonable data loss window. +**Tip**: For public cloud, moderate concurrency and 32MB bulk windows ensure stable snapshot streaming without affecting heartbeats. Acknowledged writes are never at risk; only client-facing ack latency and the amount of in-flight (un-fsynced) data on restart scale with write concurrency. ### 3. 5-Node High-Durability Cluster (Production) @@ -334,10 +299,6 @@ max_concurrent_streams = 256 β€’ Priority: Data Integrity over Write Latency ```toml -[raft.persistence] -strategy = "MemFirst" -flush_policy = { Batch = { idle_flush_interval_ms = 100 } } # More frequent flush for durability - [network.control] connection_window_size = 4_194_304 # 4MB @@ -354,7 +315,7 @@ max_concurrent_streams = 512 ``` -**Tip**: For higher write persistence within a process lifecycle, lower `idle_flush_interval_ms` (e.g., 100ms). Note: `MemFirst` is not power-loss safe regardless of flush interval. +**Tip**: Client-facing write latency is fsync-bound. Acknowledged writes are power-loss safe β€” Raft only counts an entry toward quorum, and acknowledges it to the client, after fsync completes. ## Network Environment Tuning Recommendations @@ -362,12 +323,12 @@ These parameters are primarilyΒ **network-dependent**, not CPU/memory dependent. Adjust them based on latency, packet loss, and connection stability. -| **Environment** | **tcp_keepalive_in_secs** | **http2_keep_alive_interval_in_secs** | **http2_keep_alive_timeout_in_secs** | **Notes** | -| -------------------------------- | ------------------------- | ------------------------------------- | ------------------------------------ | ------------------------------------------------------- | -| **Local / In-Cluster (LAN)** | 60 | 10 | 5 | Low latency & stable; defaults are fine | -| **Cross-Region / Stable WAN** | 60 | 15 | 8 | Slightly longer keep-alive to avoid false disconnects | -| **Public Cloud / Moderate Loss** | 60 | 20 | 10 | Higher interval & timeout for lossy links | -| **High Latency / Unstable WAN** | 120 | 30 | 15 | Longer timeouts prevent spurious drops | +| **Environment** | **tcp_keepalive_in_secs** | **http2_keep_alive_interval_in_secs** | **http2_keep_alive_timeout_in_secs** | **Notes** | +| -------------------------------- | ------------------------- | ------------------------------------- | ------------------------------------ | ----------------------------------------------------- | +| **Local / In-Cluster (LAN)** | 60 | 10 | 5 | Low latency & stable; defaults are fine | +| **Cross-Region / Stable WAN** | 60 | 15 | 8 | Slightly longer keep-alive to avoid false disconnects | +| **Public Cloud / Moderate Loss** | 60 | 20 | 10 | Higher interval & timeout for lossy links | +| **High Latency / Unstable WAN** | 120 | 30 | 15 | Longer timeouts prevent spurious drops | **Guidelines:** @@ -558,19 +519,19 @@ Client write ### Metrics Reference -| Metric | Type | Answers | -| ------------------------------------------------- | ----------- | -------------------------------------- | -| `core.raft.buffer.length{buffer="propose"}` | Gauge | Is the proposal channel backlogged? | -| `core.raft.fsync.duration_ms` | Histogram | How long does each fsync take? | -| `core.raft.fsync.batch_entries` | Histogram | Is FsyncCoordinator coalescing writes? | -| `core.raft.fsync.inflight` | Gauge (0/1) | Is a fsync task currently running? | -| `core.raft.fsync.busy_nanos_total` | Counter | fsync thread utilization | -| `core.state_machine.apply_chunk.duration_ms` | Histogram | SM apply latency | -| `core.state_machine.apply_chunk.batch_size` | Histogram | Entries applied per chunk | -| `core.state_machine.apply.busy_nanos_total` | Counter | SM apply utilization | -| `core.raft.write.propose_to_apply_ms` | Histogram | End-to-end write latency | -| `core.raft.write.propose_to_commit_ms` | Histogram | Propose-to-commit latency | -| `core.raft.backpressure.rejections{node_id,type}` | Counter | Rejected requests (write/read) | +| Metric | Type | Answers | +| ------------------------------------------------- | ----------- | ----------------------------------- | +| `core.raft.buffer.length{buffer="propose"}` | Gauge | Is the proposal channel backlogged? | +| `core.raft.fsync.duration_ms` | Histogram | How long does each fsync take? | +| `core.raft.fsync.batch_entries` | Histogram | Is FsyncWorker coalescing writes? | +| `core.raft.fsync.inflight` | Gauge (0/1) | Is a fsync task currently running? | +| `core.raft.fsync.busy_nanos_total` | Counter | fsync thread utilization | +| `core.state_machine.apply_chunk.duration_ms` | Histogram | SM apply latency | +| `core.state_machine.apply_chunk.batch_size` | Histogram | Entries applied per chunk | +| `core.state_machine.apply.busy_nanos_total` | Counter | SM apply utilization | +| `core.raft.write.propose_to_apply_ms` | Histogram | End-to-end write latency | +| `core.raft.write.propose_to_commit_ms` | Histogram | Propose-to-commit latency | +| `core.raft.backpressure.rejections{node_id,type}` | Counter | Rejected requests (write/read) | ### Finding the Bottleneck: Utilization Ratio @@ -600,7 +561,7 @@ Example interpretation: **`fsync.batch_entries` p50 = 1 under high write load** -FsyncCoordinator is not coalescing. Each proposal triggers its own fsync. Under a single- +FsyncWorker is not coalescing. Each proposal triggers its own fsync. Under a single- client benchmark this is expected and correct β€” batching requires concurrent writers. Under multi-client load, if batch_entries stays at 1, investigate whether proposals are arriving in rapid bursts or at a steady trickle. @@ -608,7 +569,6 @@ in rapid bursts or at a steady trickle. **`fsync.duration_ms` p99 >> p50 (high tail latency)** Occasional long fsyncs (disk GC, cloud volume throttling). Check storage I/O metrics. -Increasing `idle_flush_interval_ms` allows larger batches that amortize these spikes. **End-to-end latency high, both utilization metrics low** diff --git a/d-engine/src/docs/server_guide/customize-storage-engine.md b/d-engine/src/docs/server_guide/customize-storage-engine.md index 0874acb0..c337141e 100644 --- a/d-engine/src/docs/server_guide/customize-storage-engine.md +++ b/d-engine/src/docs/server_guide/customize-storage-engine.md @@ -42,7 +42,7 @@ impl LogStore for CustomLogStore { } fn flush(&self) -> Result<(), Error> { - // Sync pending writes to stable storage (called by BufferedRaftLog when is_write_durable() == false) + // Sync pending writes to stable storage (called by RaftLogCore when is_write_durable() == false) Ok(()) } @@ -85,31 +85,31 @@ impl StorageEngine for CustomStorageEngine { ## 2. Key Implementation Notes - **Atomicity**: Ensure write operations are atomicβ€”use batch operations where possible -- **Durability**: Implement `is_write_durable()` accuratelyβ€”this controls whether `BufferedRaftLog` +- **Durability**: Implement `is_write_durable()` accuratelyβ€”this controls whether `RaftLogCore` calls `flush()` after each batch. A wrong `true` advances `durable_index` before data is crash-safe, silently breaking Raft's durability guarantee. - **Consistency**: Maintain exactly-once semantics for log entries - **Performance**: Target >100k ops/sec for log persistence. Do not call `fsync` inside - `persist_entries()`β€”the framework batches entries and calls `flush()` once per batch - (`MemFirst + FlushPolicy::Batch`), which amortises the `fsync` cost across many entries. + `persist_entries()`β€”the framework batches entries and calls `flush()` once per batch, + which amortises the `fsync` cost across many entries. - **Resource Management**: Clean up resources in `Drop` implementation ## 3. StorageEngine API Reference ### LogStore Methods -| Method | Purpose | Performance Target | -| ---------------------- | ---------------------------------------------------- | ---------------------- | -| `persist_entries()` | Batch persist log entries | >100k entries/sec | -| `entry()` | Get single entry by index | <1ms latency | -| `get_entries()` | Get entries in range | <1ms for 10k entries | -| `purge()` | Remove logs up to index | <100ms for 10k entries | -| `truncate()` | Remove entries from index | <100ms | -| `replace_range()` | Atomically truncate + persist new entries (default: truncate then persist; override for single-batch atomicity) | Varies by backend | -| `is_write_durable()` | Whether persist_entries() is crash-safe without flush() | sync, no I/O | -| `flush()` | Sync writes to stable storage (called when `is_write_durable() == false`) | Varies by backend | -| `reset()` | Clear all data | <1s | -| `last_index()` | Get highest persisted index | <100ΞΌs | +| Method | Purpose | Performance Target | +| -------------------- | --------------------------------------------------------------------------------------------------------------- | ---------------------- | +| `persist_entries()` | Batch persist log entries | >100k entries/sec | +| `entry()` | Get single entry by index | <1ms latency | +| `get_entries()` | Get entries in range | <1ms for 10k entries | +| `purge()` | Remove logs up to index | <100ms for 10k entries | +| `truncate()` | Remove entries from index | <100ms | +| `replace_range()` | Atomically truncate + persist new entries (default: truncate then persist; override for single-batch atomicity) | Varies by backend | +| `is_write_durable()` | Whether persist_entries() is crash-safe without flush() | sync, no I/O | +| `flush()` | Sync writes to stable storage (called when `is_write_durable() == false`) | Varies by backend | +| `reset()` | Clear all data | <1s | +| `last_index()` | Get highest persisted index | <100ΞΌs | ### MetaStore Methods diff --git a/examples/quick-start-standalone/go.mod b/examples/quick-start-standalone/go.mod index 79b1ef9c..9865cf61 100644 --- a/examples/quick-start-standalone/go.mod +++ b/examples/quick-start-standalone/go.mod @@ -6,13 +6,13 @@ replace github.com/deventlab/d-engine/proto => ../../d-engine-proto/go require ( github.com/deventlab/d-engine/proto v0.0.0 - google.golang.org/grpc v1.80.0 + google.golang.org/grpc v1.83.2 ) require ( - golang.org/x/net v0.53.0 // indirect - golang.org/x/sys v0.43.0 // indirect - golang.org/x/text v0.36.0 // indirect - google.golang.org/genproto/googleapis/rpc v0.0.0-20260406210006-6f92a3bedf2d // indirect + golang.org/x/net v0.58.0 // indirect + golang.org/x/sys v0.47.0 // indirect + golang.org/x/text v0.41.0 // indirect + google.golang.org/genproto/googleapis/rpc v0.0.0-20260526163538-3dc84a4a5aaa // indirect google.golang.org/protobuf v1.36.11 // indirect ) diff --git a/examples/quick-start-standalone/go.sum b/examples/quick-start-standalone/go.sum index adb3ad1c..2d3ad8d1 100644 --- a/examples/quick-start-standalone/go.sum +++ b/examples/quick-start-standalone/go.sum @@ -12,27 +12,27 @@ github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0= github.com/google/uuid v1.6.0/go.mod h1:TIyPZe4MgqvfeYDBFedMoGGpEw/LqOeaOT+nhxU+yHo= go.opentelemetry.io/auto/sdk v1.2.1 h1:jXsnJ4Lmnqd11kwkBV2LgLoFMZKizbCi5fNZ/ipaZ64= go.opentelemetry.io/auto/sdk v1.2.1/go.mod h1:KRTj+aOaElaLi+wW1kO/DZRXwkF4C5xPbEe3ZiIhN7Y= -go.opentelemetry.io/otel v1.39.0 h1:8yPrr/S0ND9QEfTfdP9V+SiwT4E0G7Y5MO7p85nis48= -go.opentelemetry.io/otel v1.39.0/go.mod h1:kLlFTywNWrFyEdH0oj2xK0bFYZtHRYUdv1NklR/tgc8= -go.opentelemetry.io/otel/metric v1.39.0 h1:d1UzonvEZriVfpNKEVmHXbdf909uGTOQjA0HF0Ls5Q0= -go.opentelemetry.io/otel/metric v1.39.0/go.mod h1:jrZSWL33sD7bBxg1xjrqyDjnuzTUB0x1nBERXd7Ftcs= -go.opentelemetry.io/otel/sdk v1.39.0 h1:nMLYcjVsvdui1B/4FRkwjzoRVsMK8uL/cj0OyhKzt18= -go.opentelemetry.io/otel/sdk v1.39.0/go.mod h1:vDojkC4/jsTJsE+kh+LXYQlbL8CgrEcwmt1ENZszdJE= -go.opentelemetry.io/otel/sdk/metric v1.39.0 h1:cXMVVFVgsIf2YL6QkRF4Urbr/aMInf+2WKg+sEJTtB8= -go.opentelemetry.io/otel/sdk/metric v1.39.0/go.mod h1:xq9HEVH7qeX69/JnwEfp6fVq5wosJsY1mt4lLfYdVew= -go.opentelemetry.io/otel/trace v1.39.0 h1:2d2vfpEDmCJ5zVYz7ijaJdOF59xLomrvj7bjt6/qCJI= -go.opentelemetry.io/otel/trace v1.39.0/go.mod h1:88w4/PnZSazkGzz/w84VHpQafiU4EtqqlVdxWy+rNOA= -golang.org/x/net v0.53.0 h1:d+qAbo5L0orcWAr0a9JweQpjXF19LMXJE8Ey7hwOdUA= -golang.org/x/net v0.53.0/go.mod h1:JvMuJH7rrdiCfbeHoo3fCQU24Lf5JJwT9W3sJFulfgs= -golang.org/x/sys v0.43.0 h1:Rlag2XtaFTxp19wS8MXlJwTvoh8ArU6ezoyFsMyCTNI= -golang.org/x/sys v0.43.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw= -golang.org/x/text v0.36.0 h1:JfKh3XmcRPqZPKevfXVpI1wXPTqbkE5f7JA92a55Yxg= -golang.org/x/text v0.36.0/go.mod h1:NIdBknypM8iqVmPiuco0Dh6P5Jcdk8lJL0CUebqK164= +go.opentelemetry.io/otel v1.44.0 h1:JjwHmHpA4iZ3wBxluu2fbbE7j4kqlE8jXyAyPXH7HqU= +go.opentelemetry.io/otel v1.44.0/go.mod h1:BMgjTHL9WPRlRjL2oZCBTL4whCGtXch2H4BhOPIAyYc= +go.opentelemetry.io/otel/metric v1.44.0 h1:1w0gILTcHdr3YI+ixLyjemwrVnsMURbTZFrSYCdDdmc= +go.opentelemetry.io/otel/metric v1.44.0/go.mod h1:8O7hanEPBNgEMmybD3s2VBKcgWOCsA6tzHBPODAiquo= +go.opentelemetry.io/otel/sdk v1.44.0 h1:nHYwb9lK+fJPU/dnT6s7W7Z8itMWyqrnVfbheVYrZ58= +go.opentelemetry.io/otel/sdk v1.44.0/go.mod h1:Osuydd3Se74nqjAKxid74N5eC+jfEqfTegHRnq58oK0= +go.opentelemetry.io/otel/sdk/metric v1.44.0 h1:3LlKgI+VjbVsjNRFZJZAJ30WjXC5VkNRks6si09iEfI= +go.opentelemetry.io/otel/sdk/metric v1.44.0/go.mod h1:5B5pMARnXxKhltooO4xUuCBorl65a4EpnTalObqOigA= +go.opentelemetry.io/otel/trace v1.44.0 h1:jxF5CsGYCe74MCRx2X4g7WsY/VBKRqqpNvXlX/6gtIk= +go.opentelemetry.io/otel/trace v1.44.0/go.mod h1:oLl1jrMQAVo6v3GAggN+1VH9VIz9iUSvW53sW1Q8PIE= +golang.org/x/net v0.58.0 h1:ynWG7rqYi4ccpTEuPZ2QGWHktVEM9DMCj9yzDE0Q7To= +golang.org/x/net v0.58.0/go.mod h1:YwCddHnFlT7eLQqVprV19OnhLGtc5xOKgE0RyqgfWAU= +golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs= +golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw= +golang.org/x/text v0.41.0 h1:vz/seA0lnX87Othu2f/0L24RcgrXD9/YFTSuGjj3rH8= +golang.org/x/text v0.41.0/go.mod h1:jvf1O8ajNzZqhSrQBPbutR/EB83Cc0CFrezNQIwbb5M= gonum.org/v1/gonum v0.17.0 h1:VbpOemQlsSMrYmn7T2OUvQ4dqxQXU+ouZFQsZOx50z4= gonum.org/v1/gonum v0.17.0/go.mod h1:El3tOrEuMpv2UdMrbNlKEh9vd86bmQ6vqIcDwxEOc1E= -google.golang.org/genproto/googleapis/rpc v0.0.0-20260406210006-6f92a3bedf2d h1:wT2n40TBqFY6wiwazVK9/iTWbsQrgk5ZfCSVFLO9LQA= -google.golang.org/genproto/googleapis/rpc v0.0.0-20260406210006-6f92a3bedf2d/go.mod h1:4Hqkh8ycfw05ld/3BWL7rJOSfebL2Q+DVDeRgYgxUU8= -google.golang.org/grpc v1.80.0 h1:Xr6m2WmWZLETvUNvIUmeD5OAagMw3FiKmMlTdViWsHM= -google.golang.org/grpc v1.80.0/go.mod h1:ho/dLnxwi3EDJA4Zghp7k2Ec1+c2jqup0bFkw07bwF4= +google.golang.org/genproto/googleapis/rpc v0.0.0-20260526163538-3dc84a4a5aaa h1:mZHHdPZl0dbGHCflZgAq/Q468DWVFcU2whhB2KAo8fk= +google.golang.org/genproto/googleapis/rpc v0.0.0-20260526163538-3dc84a4a5aaa/go.mod h1:4Hqkh8ycfw05ld/3BWL7rJOSfebL2Q+DVDeRgYgxUU8= +google.golang.org/grpc v1.83.2 h1:EManeRomTObA0BU7I8vXgg/78uE5MJ9M8B39EX2WscU= +google.golang.org/grpc v1.83.2/go.mod h1:YPI1hK3kDked6iHvgX3tR0y+nX/qpMFKhPgFsokw1S8= google.golang.org/protobuf v1.36.11 h1:fV6ZwhNocDyBLK0dj+fg8ektcVegBBuEolpbTQyBNVE= google.golang.org/protobuf v1.36.11/go.mod h1:HTf+CrKn2C3g5S8VImy6tdcUvCska2kB7j23XfzDpco= diff --git a/examples/single-node-expansion/Makefile b/examples/single-node-expansion/Makefile index a31b2775..a00218c4 100644 --- a/examples/single-node-expansion/Makefile +++ b/examples/single-node-expansion/Makefile @@ -8,12 +8,35 @@ # =============================== LOG_LEVEL ?= debug + +# On macOS with Homebrew: auto-detect compression lib paths to skip bundled C++ +# compilation of RocksDB dependencies, which fails under macOS 26 + Xcode 26 +# (Clang 16 lacks __builtin_ctzg/__builtin_clzg from LLVM 18+ SDK headers). +# brew --prefix resolves correctly on both Apple Silicon (/opt/homebrew) and +# Intel Mac (/usr/local). Silently no-ops when brew or a lib is absent. +SNAPPY_PREFIX := $(shell brew --prefix snappy 2>/dev/null) +LZ4_PREFIX := $(shell brew --prefix lz4 2>/dev/null) +ZSTD_PREFIX := $(shell brew --prefix zstd 2>/dev/null) +BREW_ROCKSDB_ENV := + +ifneq ($(SNAPPY_PREFIX),) +ifneq ($(wildcard $(SNAPPY_PREFIX)/lib),) + BREW_ROCKSDB_ENV += SNAPPY_LIB_DIR=$(SNAPPY_PREFIX)/lib +endif +endif +ifneq ($(LZ4_PREFIX),) + BREW_ROCKSDB_ENV += LZ4_LIB_DIR=$(LZ4_PREFIX)/lib +endif +ifneq ($(ZSTD_PREFIX),) + BREW_ROCKSDB_ENV += ZSTD_LIB_DIR=$(ZSTD_PREFIX)/lib +endif + # =============================== # Build Targets # =============================== build: @echo "Building release binary..." - cargo build --release --jobs 4 + $(BREW_ROCKSDB_ENV) cargo build --release --jobs 4 # =============================== # Single Node Bootstrap diff --git a/examples/single-node-expansion/config/n1.toml b/examples/single-node-expansion/config/n1.toml index bdb78715..1d961cb3 100644 --- a/examples/single-node-expansion/config/n1.toml +++ b/examples/single-node-expansion/config/n1.toml @@ -14,6 +14,8 @@ initial_cluster = [ [raft] general_raft_timeout_duration_in_ms = 100 +cmd_channel_capacity = 1024 +max_pending_append_responses = 1024 [raft.election] election_timeout_min = 1000 @@ -23,27 +25,49 @@ election_timeout_max = 2000 default_policy = "LeaseRead" lease_duration_ms = 500 +[raft.read_actor] +channel_capacity = 10240 +max_drain = 2000 + +[raft.batching] +# Maximum number of commands to accumulate in a single batch during drain operations +max_batch_size = 200 + +[raft.metrics] +enable_backpressure = false +enable_batch = false + +[raft.backpressure] +max_pending_writes = 1000 +max_pending_reads = 500 + + [raft.persistence] -strategy = "MemFirst" -flush_policy = { Batch = { idle_flush_interval_ms = 20 } } -max_buffered_entries = 10000 +flush_policy = { Batch = { idle_flush_interval_ms = 1000 } } [raft.snapshot] -enable = false -max_log_entries_before_snapshot = 10000 -retained_log_entries = 3 +enable = true +max_log_entries_before_snapshot = 5000 +retained_log_entries = 100 +cleanup_retain_count = 100 -# == Network Control Plane == +# == TTL Lease Configuration == +[raft.state_machine.lease] +cleanup_interval_ms = 1000 +max_cleanup_duration_ms = 1 + +# == Network Control Plane (voting, heartbeat, etc.) == [network.control] connection_window_size = 4_194_304 stream_window_size = 2_097_152 tcp_keepalive_in_secs = 60 -http2_keep_alive_interval_in_secs = 15 -http2_keep_alive_timeout_in_secs = 10 +http2_keep_alive_interval_in_secs = 15 # Slightly increase to reduce frequent keep-alives +http2_keep_alive_timeout_in_secs = 10 # Increase timeout +# New performance tuning parameters -# == Network Data Plane == +# == Network Data Plane (append_entries, etc.) == [network.data] connect_timeout_in_ms = 100 request_timeout_in_ms = 300 @@ -51,13 +75,16 @@ connection_window_size = 8_388_608 stream_window_size = 4_194_304 tcp_keepalive_in_secs = 60 -http2_keep_alive_interval_in_secs = 15 +http2_keep_alive_interval_in_secs = 15 # Same as control plane http2_keep_alive_timeout_in_secs = 10 - +# New data plane optimizations # == Server Transport (single listener serving every RPC type) == [network.server] concurrency_limit_per_connection = 100 # Increased for higher concurrent replication load max_concurrent_streams = 4096 # Increased to reduce stream creation overhead max_pending_accept_reset_streams = 2000 # Higher pending stream limit for Rapid Reset mitigation + +[storage] +unified_db = false diff --git a/examples/single-node-expansion/config/n2.toml b/examples/single-node-expansion/config/n2.toml index 5e5356c3..0e4bb985 100644 --- a/examples/single-node-expansion/config/n2.toml +++ b/examples/single-node-expansion/config/n2.toml @@ -28,9 +28,7 @@ lease_duration_ms = 500 [raft.persistence] -strategy = "MemFirst" flush_policy = { Batch = { idle_flush_interval_ms = 20 } } -max_buffered_entries = 10000 [raft.snapshot] enable = false diff --git a/examples/single-node-expansion/config/n3.toml b/examples/single-node-expansion/config/n3.toml index 75a585e3..32dc4c3b 100644 --- a/examples/single-node-expansion/config/n3.toml +++ b/examples/single-node-expansion/config/n3.toml @@ -30,9 +30,7 @@ lease_duration_ms = 500 [raft.persistence] -strategy = "MemFirst" flush_policy = { Batch = { idle_flush_interval_ms = 20 } } -max_buffered_entries = 10000 [raft.snapshot] enable = false diff --git a/examples/sled-cluster/config/n1.toml b/examples/sled-cluster/config/n1.toml index a5b391e8..5902a6ad 100644 --- a/examples/sled-cluster/config/n1.toml +++ b/examples/sled-cluster/config/n1.toml @@ -16,8 +16,6 @@ batch_size = 5000 [raft.persistence] -strategy = "MemFirst" -# strategy = "DiskFirst" flush_policy = { Batch = { idle_flush_interval_ms = 100 } } [raft.snapshot] diff --git a/examples/sled-cluster/config/n2.toml b/examples/sled-cluster/config/n2.toml index 87a5f29a..c7a89b70 100644 --- a/examples/sled-cluster/config/n2.toml +++ b/examples/sled-cluster/config/n2.toml @@ -16,8 +16,6 @@ batch_size = 5000 [raft.persistence] -strategy = "MemFirst" -# strategy = "DiskFirst" flush_policy = { Batch = { idle_flush_interval_ms = 100 } } [raft.snapshot] diff --git a/examples/sled-cluster/config/n3.toml b/examples/sled-cluster/config/n3.toml index 5fe5b628..c099227a 100644 --- a/examples/sled-cluster/config/n3.toml +++ b/examples/sled-cluster/config/n3.toml @@ -16,8 +16,6 @@ batch_size = 5000 [raft.persistence] -strategy = "MemFirst" -# strategy = "DiskFirst" flush_policy = { Batch = { idle_flush_interval_ms = 100 } } diff --git a/examples/sled-cluster/src/sled_storage_engine.rs b/examples/sled-cluster/src/sled_storage_engine.rs index 03948a64..1ecec903 100644 --- a/examples/sled-cluster/src/sled_storage_engine.rs +++ b/examples/sled-cluster/src/sled_storage_engine.rs @@ -147,7 +147,8 @@ impl LogStore for SledLogStore { &self, from_index: u64, new_entries: Vec, - ) -> Result<()> { + ) -> Result { + let new_last = new_entries.last().map(|e| e.index).unwrap_or(from_index.saturating_sub(1)); let mut batch = sled::Batch::default(); // collect and remove all keys >= from_index @@ -164,7 +165,7 @@ impl LogStore for SledLogStore { } self.tree.apply_batch(batch).map_err(|e| StorageError::DbError(e.to_string()))?; - Ok(()) + Ok(new_last) } fn is_write_durable(&self) -> bool { diff --git a/examples/three-nodes-embedded/README.md b/examples/three-nodes-embedded/README.md index 25255a59..b6f2dda2 100644 --- a/examples/three-nodes-embedded/README.md +++ b/examples/three-nodes-embedded/README.md @@ -84,7 +84,6 @@ default_policy = "LeaseRead" lease_duration_ms = 500 [raft.persistence] -strategy = "MemFirst" flush_policy = { Batch = { threshold = 100, interval_ms = 20 } } ``` diff --git a/examples/three-nodes-standalone/Makefile b/examples/three-nodes-standalone/Makefile index 81a14311..7f23d971 100644 --- a/examples/three-nodes-standalone/Makefile +++ b/examples/three-nodes-standalone/Makefile @@ -3,7 +3,8 @@ .PHONY: build start-cluster clean clean-log-db help \ start-node1 start-node2 start-node3 \ perf-node1 perf-node2 perf-node3 perf-cluster \ - tokio-console-node1 tokio-console-node2 tokio-console-node3 tokio-console-cluster + tokio-console-node1 tokio-console-node2 tokio-console-node3 tokio-console-cluster \ + start-ram-cluster ramdisk-create clean-ram-log-db ramdisk-release .DEFAULT_GOAL := help # =============================== @@ -43,10 +44,16 @@ build: # =============================== # Cluster Management (Normal Mode) # =============================== +# Overridable so start-ram-cluster can point these at /Volumes/RAMDiskN +# without duplicating the node targets. +DB_PATH_1 ?= ./db/1 +DB_PATH_2 ?= ./db/2 +DB_PATH_3 ?= ./db/3 + start-node1: @echo "πŸš€ Starting Node 1..." @CONFIG_PATH=config/n1 \ - DB_PATH="./db/1" \ + DB_PATH="$(DB_PATH_1)" \ LOG_DIR="./logs/1" \ METRICS_PORT=8081 \ RUST_LOG=demo=$(LOG_LEVEL),d_engine=$(LOG_LEVEL),timing=$(LOG_LEVEL) \ @@ -56,7 +63,7 @@ start-node1: start-node2: @echo "πŸš€ Starting Node 2..." @CONFIG_PATH=config/n2 \ - DB_PATH="./db/2" \ + DB_PATH="$(DB_PATH_2)" \ LOG_DIR="./logs/2" \ METRICS_PORT=8082 \ RUST_LOG=demo=$(LOG_LEVEL),d_engine=$(LOG_LEVEL),timing=$(LOG_LEVEL) \ @@ -65,7 +72,7 @@ start-node2: start-node3: @echo "πŸš€ Starting Node 3..." @CONFIG_PATH=config/n3 \ - DB_PATH="./db/3" \ + DB_PATH="$(DB_PATH_3)" \ LOG_DIR="./logs/3" \ METRICS_PORT=8083 \ RUST_LOG=demo=$(LOG_LEVEL),d_engine=$(LOG_LEVEL),timing=$(LOG_LEVEL) \ @@ -76,6 +83,40 @@ start-cluster: @echo "Starting 3-node cluster in parallel..." $(MAKE) -j3 start-node1 start-node2 start-node3 +# =============================== +# RAM Disk (macOS only, optional) +# =============================== +# Isolates each node's storage on its own independent RAM disk volume β€” use +# this to strip physical-disk latency out of a benchmark (e.g. to study +# fsync/scheduling behavior in isolation), not as a general fix for +# unrelated flakiness. See tickets/milestones/v0.2.5/446-perf-batching-measurement-2026-09-13.md. +RAMDISK_SIZE_MB ?= 1024 +RAMDISK_SECTORS := $(shell echo $$(( $(RAMDISK_SIZE_MB) * 2048 ))) +RAMDISK_VOLUMES := RAMDisk1 RAMDisk2 RAMDisk3 + +# One command: wipe stale node data, ensure the 3 RAM disks exist, start the +# cluster on them instead of ./db. +start-ram-cluster: clean-ram-log-db ramdisk-create + @echo "Starting 3-node cluster on RAM disk..." + $(MAKE) -j3 start-node1 start-node2 start-node3 \ + DB_PATH_1=/Volumes/RAMDisk1/n1 DB_PATH_2=/Volumes/RAMDisk2/n2 DB_PATH_3=/Volumes/RAMDisk3/n3 + +# Idempotent β€” mounts any of RAMDisk1/2/3 that aren't already present. +ramdisk-create: + @[ "$$(uname)" = "Darwin" ] || { echo "RAM disk targets need macOS."; exit 1; } + @for v in $(RAMDISK_VOLUMES); do \ + [ -d "/Volumes/$$v" ] || diskutil erasevolume HFS+ $$v `hdiutil attach -nomount ram://$(RAMDISK_SECTORS)` >/dev/null; \ + done + +# Wipes node data off the RAM disks; keeps the volumes mounted so the next +# start-ram-cluster doesn't pay to recreate them. +clean-ram-log-db: + @for v in $(RAMDISK_VOLUMES); do rm -rf /Volumes/$$v/*; done + +# Unmounts the RAM disks entirely, releasing the memory back to the OS. +ramdisk-release: + @for v in $(RAMDISK_VOLUMES); do diskutil eject /Volumes/$$v 2>/dev/null || true; done + # =============================== # Performance Profiling with Samply @@ -184,6 +225,10 @@ help: @echo " perf-node1..3 - Run individual nodes under samply profiler" @echo " perf-cluster - Run full 3-node cluster under samply profiling" @echo " tokio-console-cluster - Run full 3-node cluster under tokio console monitoring" + @echo " start-ram-cluster - (macOS) Clean+create RAM disks, start cluster on them" + @echo " ramdisk-create - (macOS) Create RAMDisk1/2/3 if not already mounted" + @echo " clean-ram-log-db - Wipe node data off the RAM disks (keeps them mounted)" + @echo " ramdisk-release - Unmount RAMDisk1/2/3, freeing the memory" @echo " clean - Remove build artifacts, logs, and profiles" @echo " clean-log-db - Remove only logs and database files" @echo " help - Show this help message" diff --git a/examples/three-nodes-standalone/config/n1.toml b/examples/three-nodes-standalone/config/n1.toml index a040e1ae..431f4e09 100644 --- a/examples/three-nodes-standalone/config/n1.toml +++ b/examples/three-nodes-standalone/config/n1.toml @@ -11,7 +11,7 @@ initial_cluster = [ [raft] general_raft_timeout_duration_in_ms = 100 cmd_channel_capacity = 1024 -ordered_channel_capacity = 1024 +max_pending_append_responses = 1024 [raft.election] election_timeout_min = 1000 @@ -39,15 +39,10 @@ max_pending_reads = 500 [raft.persistence] -# strategy = "DiskFirst" -strategy = "MemFirst" flush_policy = { Batch = { idle_flush_interval_ms = 1000 } } -# Maximum number of log entries to buffer in memory -# when using async persistence strategies (MemFirst/Batched) -max_buffered_entries = 10000 [raft.snapshot] -enable = true +enable = false max_log_entries_before_snapshot = 5000 retained_log_entries = 100 cleanup_retain_count = 100 diff --git a/examples/three-nodes-standalone/config/n2.toml b/examples/three-nodes-standalone/config/n2.toml index 20e816f2..0d100281 100644 --- a/examples/three-nodes-standalone/config/n2.toml +++ b/examples/three-nodes-standalone/config/n2.toml @@ -11,7 +11,7 @@ initial_cluster = [ [raft] general_raft_timeout_duration_in_ms = 100 cmd_channel_capacity = 1024 -ordered_channel_capacity = 1024 +max_pending_append_responses = 1024 [raft.election] election_timeout_min = 1000 @@ -39,15 +39,10 @@ max_pending_writes = 1000 max_pending_reads = 500 [raft.persistence] -# strategy = "DiskFirst" -strategy = "MemFirst" flush_policy = { Batch = { idle_flush_interval_ms = 1000 } } -# Maximum number of log entries to buffer in memory -# when using async persistence strategies (MemFirst/Batched) -max_buffered_entries = 10000 [raft.snapshot] -enable = true +enable = false max_log_entries_before_snapshot = 5000 retained_log_entries = 100 cleanup_retain_count = 100 diff --git a/examples/three-nodes-standalone/config/n3.toml b/examples/three-nodes-standalone/config/n3.toml index 0a29fe2d..5e3d80d2 100644 --- a/examples/three-nodes-standalone/config/n3.toml +++ b/examples/three-nodes-standalone/config/n3.toml @@ -11,7 +11,7 @@ initial_cluster = [ [raft] general_raft_timeout_duration_in_ms = 100 cmd_channel_capacity = 1024 -ordered_channel_capacity = 1024 +max_pending_append_responses = 1024 [raft.election] election_timeout_min = 1000 @@ -39,15 +39,10 @@ max_pending_writes = 1000 max_pending_reads = 500 [raft.persistence] -# strategy = "DiskFirst" -strategy = "MemFirst" flush_policy = { Batch = { idle_flush_interval_ms = 1000 } } -# Maximum number of log entries to buffer in memory -# when using async persistence strategies (MemFirst/Batched) -max_buffered_entries = 10000 [raft.snapshot] -enable = true +enable = false max_log_entries_before_snapshot = 5000 retained_log_entries = 100 cleanup_retain_count = 100 diff --git a/examples/three-nodes-standalone/docker/Dockerfile b/examples/three-nodes-standalone/docker/Dockerfile index df52fc4b..e971495e 100644 --- a/examples/three-nodes-standalone/docker/Dockerfile +++ b/examples/three-nodes-standalone/docker/Dockerfile @@ -55,6 +55,8 @@ RUN apt-get update && \ iptables \ iproute2 \ libc6 \ + libfuse3-3 \ + fuse3 \ tzdata && \ rm -rf /var/lib/apt/lists/* && \ mkdir -p /var/run/sshd && \ @@ -85,4 +87,4 @@ COPY examples/three-nodes-standalone/docker/monitoring/promtail/config.yml /etc/ WORKDIR /app -CMD ["sh", "-c", "/usr/sbin/sshd -D & CONFIG_PATH=$CONFIG_PATH LOG_DIR=$LOG_DIR METRICS_PORT=$METRICS_PORT RUST_LOG=demo=$LOG_LEVEL,d_engine=$LOG_LEVEL,hyper=warn,sled=warn demo & promtail --config.file=/etc/promtail/config.yml > /app/logs/promtail.log 2>&1"] +CMD ["sh", "-c", "/usr/sbin/sshd -D & CONFIG_PATH=$CONFIG_PATH LOG_DIR=$LOG_DIR METRICS_PORT=$METRICS_PORT DB_PATH=/app/db/$ID RUST_LOG=demo=$LOG_LEVEL,d_engine=$LOG_LEVEL,hyper=warn,sled=warn demo & promtail --config.file=/etc/promtail/config.yml > /app/logs/promtail.log 2>&1"] diff --git a/examples/three-nodes-standalone/docker/config/n1.toml b/examples/three-nodes-standalone/docker/config/n1.toml index efa8e941..a5b007dc 100644 --- a/examples/three-nodes-standalone/docker/config/n1.toml +++ b/examples/three-nodes-standalone/docker/config/n1.toml @@ -11,7 +11,7 @@ initial_cluster = [ [raft] general_raft_timeout_duration_in_ms = 100 cmd_channel_capacity = 1024 -ordered_channel_capacity = 1024 +max_pending_append_responses = 1024 [raft.election] election_timeout_min = 1000 @@ -39,12 +39,7 @@ max_pending_reads = 500 [raft.persistence] -# strategy = "DiskFirst" -strategy = "MemFirst" flush_policy = { Batch = { idle_flush_interval_ms = 1000 } } -# Maximum number of log entries to buffer in memory -# when using async persistence strategies (MemFirst/Batched) -max_buffered_entries = 10000 [raft.snapshot] enable = false diff --git a/examples/three-nodes-standalone/docker/config/n2.toml b/examples/three-nodes-standalone/docker/config/n2.toml index 78f66899..06c54f50 100644 --- a/examples/three-nodes-standalone/docker/config/n2.toml +++ b/examples/three-nodes-standalone/docker/config/n2.toml @@ -11,7 +11,7 @@ initial_cluster = [ [raft] general_raft_timeout_duration_in_ms = 100 cmd_channel_capacity = 1024 -ordered_channel_capacity = 1024 +max_pending_append_responses = 1024 [raft.election] election_timeout_min = 1000 @@ -39,12 +39,7 @@ max_pending_writes = 1000 max_pending_reads = 500 [raft.persistence] -# strategy = "DiskFirst" -strategy = "MemFirst" flush_policy = { Batch = { idle_flush_interval_ms = 1000 } } -# Maximum number of log entries to buffer in memory -# when using async persistence strategies (MemFirst/Batched) -max_buffered_entries = 10000 [raft.snapshot] enable = false diff --git a/examples/three-nodes-standalone/docker/config/n3.toml b/examples/three-nodes-standalone/docker/config/n3.toml index 674109a2..2cfe4ad9 100644 --- a/examples/three-nodes-standalone/docker/config/n3.toml +++ b/examples/three-nodes-standalone/docker/config/n3.toml @@ -11,7 +11,7 @@ initial_cluster = [ [raft] general_raft_timeout_duration_in_ms = 100 cmd_channel_capacity = 1024 -ordered_channel_capacity = 1024 +max_pending_append_responses = 1024 [raft.election] election_timeout_min = 1000 @@ -39,12 +39,7 @@ max_pending_writes = 1000 max_pending_reads = 500 [raft.persistence] -# strategy = "DiskFirst" -strategy = "MemFirst" flush_policy = { Batch = { idle_flush_interval_ms = 1000 } } -# Maximum number of log entries to buffer in memory -# when using async persistence strategies (MemFirst/Batched) -max_buffered_entries = 10000 [raft.snapshot] enable = false diff --git a/examples/three-nodes-standalone/src/main.rs b/examples/three-nodes-standalone/src/main.rs index 4a0fd403..c61dc79e 100644 --- a/examples/three-nodes-standalone/src/main.rs +++ b/examples/three-nodes-standalone/src/main.rs @@ -58,8 +58,11 @@ async fn main() { let (graceful_tx, graceful_rx) = watch::channel(()); // Start the server (wait for its initialization to complete) - let server_handler = - tokio::spawn(start_dengine_server(data_dir, config_path, graceful_rx.clone())); + let server_handler = tokio::spawn(start_dengine_server( + data_dir, + config_path, + graceful_rx.clone(), + )); // Wait for the server to initialize (adjust the waiting time according to the actual logic) tokio::time::sleep(Duration::from_secs(1)).await;